mirror of
https://github.com/techno-tim/k3s-ansible.git
synced 2026-08-09 07:23:19 +02:00
Compare commits
233 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 21f7317ab8 | |||
| fcdbf68ba3 | |||
| e0ac53dba3 | |||
| 287d8b7a27 | |||
| 56bb912bd3 | |||
| cf76292169 | |||
| 52c086d638 | |||
| 010551b8d2 | |||
| c82f2e0415 | |||
| ac1e3288c0 | |||
| db30128468 | |||
| 249238c7a4 | |||
| f5483cdabe | |||
| 88159b3875 | |||
| db85fa960c | |||
| 6aea6e71b6 | |||
| 890d43b339 | |||
| bb006cf157 | |||
| b6363cdfc5 | |||
| 83f205177d | |||
| 5c288e8f3e | |||
| dfcfbc1f3f | |||
| c3606a7847 | |||
| 35939315cd | |||
| 997ea63a3b | |||
| 68d03acc68 | |||
| 2babd39c89 | |||
| bbca35331d | |||
| 4c50fbbe10 | |||
| 10bde4eff0 | |||
| 57f234c8da | |||
| fb9a0bebd1 | |||
| 4990157d35 | |||
| 73fae0826c | |||
| 57a22e364d | |||
| 9b220c1629 | |||
| a0d78ff317 | |||
| 999cf3ee05 | |||
| 1402f33108 | |||
| 94dbffaef7 | |||
| a52e2ea72c | |||
| bb3843dbb1 | |||
| 87ea8160c1 | |||
| 5bc347aed4 | |||
| 665e274820 | |||
| 5747bfce0e | |||
| 29b7aa1b72 | |||
| 2fad0a8db6 | |||
| 5cbbf7371b | |||
| 422621c69c | |||
| 39988a9bee | |||
| 133a84b564 | |||
| 6b79057f6c | |||
| 4c0b1ee8f3 | |||
| 11f9505460 | |||
| 850301fbc4 | |||
| 983e11322e | |||
| a4df16cf87 | |||
| f8ababb7bf | |||
| 90eb5e4b41 | |||
| 97ed29b4a2 | |||
| fc2225ab8d | |||
| d99f6a96f2 | |||
| fab302fd91 | |||
| eddbcbfb76 | |||
| 03ae8de0d5 | |||
| d136fa4486 | |||
| b906cfbf72 | |||
| 2c04f38e2c | |||
| 3435f43748 | |||
| 924a2f528c | |||
| 2892ac3858 | |||
| df8e8dd591 | |||
| 3a0303d130 | |||
| b077a49e1f | |||
| 635f0b21b3 | |||
| 4a64ad42df | |||
| d0537736de | |||
| 2149827800 | |||
| 2d0596209e | |||
| 3a20500f9c | |||
| 9ce9fecc5b | |||
| 668d7fb896 | |||
| 6cee0e9051 | |||
| 6823ad51d5 | |||
| 1a521ea0d9 | |||
| e48bb6df26 | |||
| 36893c27fb | |||
| e8cd10d49b | |||
| b86156b995 | |||
| 072f1a321d | |||
| 2f46a54240 | |||
| bf0418d77f | |||
| d88eb80df0 | |||
| f50d335451 | |||
| d6597150c7 | |||
| 353f7ab641 | |||
| c7c727c3dc | |||
| 0422bfa2ac | |||
| 0333406725 | |||
| f4a19d368b | |||
| 02d212c007 | |||
| 80095250e9 | |||
| 4fe2c92795 | |||
| b3f2a4addc | |||
| cb03ee829e | |||
| 9e2e82faeb | |||
| 7c1f6cbe42 | |||
| 604eb7a6e6 | |||
| a204ed5169 | |||
| b6608ca3e4 | |||
| 8252a45dfd | |||
| c99f098c2e | |||
| 7867b87d85 | |||
| dfe19f3731 | |||
| a46d97a28d | |||
| dc9d571f17 | |||
| 6742551e5c | |||
| fb3478a086 | |||
| 518c5bb62a | |||
| 3f5d8dfe9f | |||
| efbfadcb93 | |||
| f81ec04ba2 | |||
| 8432d3bc66 | |||
| 14ae9df1bc | |||
| f175716339 | |||
| 955c6f6b4a | |||
| 3b74985767 | |||
| 9ace193ade | |||
| 83a0be3afd | |||
| 029eba6102 | |||
| 0c8253b3a5 | |||
| 326b71dfa2 | |||
| b95d6dd2cc | |||
| e4146b4ca9 | |||
| 1fb10faf7f | |||
| ea3b3c776a | |||
| 5beca87783 | |||
| 6ffc25dfe5 | |||
| bcd37a6904 | |||
| 8dd3ffc825 | |||
| f6ba208b5c | |||
| a22d8f7aaf | |||
| 05fb6b566d | |||
| 3aeb7d69ea | |||
| 61bf3971ef | |||
| 3f06a11c8d | |||
| 3888a29bb1 | |||
| 98ef696f31 | |||
| de26a79a4c | |||
| ab7ca9b551 | |||
| c5f71c9e2e | |||
| 0f23e7e258 | |||
| 121061d875 | |||
| db53f595fd | |||
| 7b6b24ce4d | |||
| a5728da35e | |||
| cda7c92203 | |||
| d910b83bf3 | |||
| 101313f880 | |||
| 12be355867 | |||
| aa09e3e9df | |||
| 511c410451 | |||
| df9c6f3014 | |||
| 5ae8fd1223 | |||
| e2e9881f0f | |||
| edf0c9eebd | |||
| 7669fd4721 | |||
| cddbfc8e40 | |||
| 70e658cf98 | |||
| 7badfbd7bd | |||
| e880f08d26 | |||
| 95b2836dfc | |||
| 505c2eeff2 | |||
| 9b6d551dd6 | |||
| a64e882fb7 | |||
| 38e773315b | |||
| 70ddf7b63c | |||
| fb3128a783 | |||
| 2e318e0862 | |||
| 0607eb8aa4 | |||
| a9904d1562 | |||
| 9707bc8a58 | |||
| e635bd2626 | |||
| 1aabb5a927 | |||
| 215690b55b | |||
| bd44a9b126 | |||
| 8d61fe81e5 | |||
| c0ff304f22 | |||
| 83077ecdd1 | |||
| 33ae0d4970 | |||
| edd4838407 | |||
| 5c79ea9b71 | |||
| 3d204ad851 | |||
| 13bd868faa | |||
| c564a8562a | |||
| 0d6d43e7ca | |||
| c0952288c2 | |||
| 1c9796e98b | |||
| 288c4089e0 | |||
| 49f0a2ce6b | |||
| 6c4621bd56 | |||
| 3e16ab6809 | |||
| 83fe50797c | |||
| 2db0b3024c | |||
| 6b2af77e74 | |||
| d1d1bc3d91 | |||
| 3a1a7a19aa | |||
| 030eeb4b75 | |||
| 4aeeb124ef | |||
| 511c020bec | |||
| c47da38b53 | |||
| 6448948e9f | |||
| 7bc198ab26 | |||
| 65bbc8e2ac | |||
| dc2976e7f6 | |||
| 5a7ba98968 | |||
| 10c6ef1d57 | |||
| ed4d888e3d | |||
| 49d6d484ae | |||
| 96c49c864e | |||
| 60adb1de42 | |||
| e023808f2f | |||
| 511ec493d6 | |||
| be3e72e173 | |||
| e33cbe52c1 | |||
| c06af919f3 | |||
| b86384c439 | |||
| bf2bd1edc5 | |||
| e98e3ee77c | |||
| 78f7a60378 | |||
| e64fea760d | |||
| 764e32c778 |
+22
-8
@@ -1,17 +1,31 @@
|
||||
---
|
||||
profile: production
|
||||
exclude_paths:
|
||||
# default paths
|
||||
- '.cache/'
|
||||
- '.github/'
|
||||
- 'test/fixtures/formatting-before/'
|
||||
- 'test/fixtures/formatting-prettier/'
|
||||
- .cache/
|
||||
- .ansible/
|
||||
- .github/
|
||||
- test/fixtures/formatting-before/
|
||||
- test/fixtures/formatting-prettier/
|
||||
|
||||
# The "converge" and "reset" playbooks use import_playbook in
|
||||
# conjunction with the "env" lookup plugin, which lets the
|
||||
# syntax check of ansible-lint fail.
|
||||
- 'molecule/**/converge.yml'
|
||||
- 'molecule/**/prepare.yml'
|
||||
- 'molecule/**/reset.yml'
|
||||
- molecule/**/converge.yml
|
||||
- molecule/**/prepare.yml
|
||||
- molecule/**/reset.yml
|
||||
|
||||
# Scenario verify inputs are plain variable files, not playbooks. They are
|
||||
# loaded as vars, not executed, so ansible-lint must not treat them as plays.
|
||||
- molecule/**/verify-vars.yml
|
||||
|
||||
# The file was generated by galaxy ansible - don't mess with it.
|
||||
- galaxy.yml
|
||||
|
||||
skip_list:
|
||||
- 'fqcn-builtins'
|
||||
- var-naming[no-role-prefix]
|
||||
|
||||
# The Molecule Vagrant driver injects this module at runtime. The custom create
|
||||
# playbook is syntax-checked separately against the exact pinned plugin module.
|
||||
mock_modules:
|
||||
- vagrant
|
||||
|
||||
@@ -1,77 +0,0 @@
|
||||
|
||||
<!-- It's a good idea to check this post first for general troubleshooting https://github.com/techno-tim/k3s-ansible/discussions/19 -->
|
||||
|
||||
<!--- Provide a general summary of the issue in the Title above -->
|
||||
|
||||
## Expected Behavior
|
||||
|
||||
<!--- Tell us what should happen -->
|
||||
|
||||
## Current Behavior
|
||||
<!--- Tell us what happens instead of the expected behavior -->
|
||||
|
||||
## Steps to Reproduce
|
||||
|
||||
<!--- reproduce this bug. Include code to reproduce, if relevant -->
|
||||
|
||||
1.
|
||||
2.
|
||||
3.
|
||||
4.
|
||||
|
||||
## Context (variables)
|
||||
<!--- please include which OS, along with the variables used when running the playbook -->
|
||||
|
||||
Operating system:
|
||||
|
||||
Hardware:
|
||||
|
||||
### Variables Used
|
||||
|
||||
`all.yml`
|
||||
|
||||
```yml
|
||||
k3s_version: ""
|
||||
ansible_user: NA
|
||||
systemd_dir: ""
|
||||
|
||||
flannel_iface: ""
|
||||
|
||||
apiserver_endpoint: ""
|
||||
|
||||
k3s_token: "NA"
|
||||
|
||||
extra_server_args: ""
|
||||
extra_agent_args: ""
|
||||
|
||||
kube_vip_tag_version: ""
|
||||
|
||||
metal_lb_speaker_tag_version: ""
|
||||
metal_lb_controller_tag_version: ""
|
||||
|
||||
metal_lb_ip_range: ""
|
||||
```
|
||||
|
||||
### Hosts
|
||||
|
||||
`host.ini`
|
||||
|
||||
```ini
|
||||
[master]
|
||||
IP.ADDRESS.ONE
|
||||
IP.ADDRESS.TWO
|
||||
IP.ADDRESS.THREE
|
||||
|
||||
[node]
|
||||
IP.ADDRESS.FOUR
|
||||
IP.ADDRESS.FIVE
|
||||
|
||||
[k3s_cluster:children]
|
||||
master
|
||||
node
|
||||
```
|
||||
|
||||
## Possible Solution
|
||||
<!--- Not obligatory, but suggest a fix/reason for the bug, -->
|
||||
|
||||
- [ ] I've checked the [General Troubleshooting Guide](https://github.com/techno-tim/k3s-ansible/discussions/20)
|
||||
@@ -0,0 +1,102 @@
|
||||
---
|
||||
name: Bug report
|
||||
description: Report a reproducible problem with the playbooks, roles, or generated resources.
|
||||
title: "[Bug]: "
|
||||
labels:
|
||||
- bug
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
Thanks for reporting a problem. Search existing issues and review the troubleshooting link first.
|
||||
Remove credentials, tokens, public IP addresses, and private hostnames from all fields and logs.
|
||||
|
||||
- type: checkboxes
|
||||
id: prerequisites
|
||||
attributes:
|
||||
label: Prerequisites
|
||||
options:
|
||||
- label: I searched existing issues and discussions for this problem.
|
||||
required: true
|
||||
- label: I reviewed the troubleshooting guidance linked from the issue chooser.
|
||||
required: true
|
||||
- label: I removed secrets and identifying infrastructure details from this report.
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: summary
|
||||
attributes:
|
||||
label: Problem summary
|
||||
description: Describe what failed and its impact.
|
||||
placeholder: A concise description of the problem and affected nodes or components.
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: expected
|
||||
attributes:
|
||||
label: Expected behavior
|
||||
description: What should have happened?
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: reproduction
|
||||
attributes:
|
||||
label: Steps to reproduce
|
||||
description: Provide the smallest reliable sequence that reproduces the issue.
|
||||
placeholder: |
|
||||
1. Configure ...
|
||||
2. Run ...
|
||||
3. Observe ...
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: input
|
||||
id: revision
|
||||
attributes:
|
||||
label: Repository revision
|
||||
description: Release, tag, branch, or commit SHA used.
|
||||
placeholder: v1.36.2+k3s1+tt1 or a commit SHA
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: input
|
||||
id: ansible-version
|
||||
attributes:
|
||||
label: Ansible version
|
||||
description: Output of `ansible --version`, shortened to version and Python details.
|
||||
placeholder: ansible-core 2.18.0, Python 3.12
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: environment
|
||||
attributes:
|
||||
label: Environment
|
||||
description: Include target OS and version, architecture, node counts, platform, and network provider.
|
||||
placeholder: Debian 13, amd64, 3 control nodes and 2 agents, bare metal, Cilium
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: configuration
|
||||
attributes:
|
||||
label: Relevant sanitized configuration
|
||||
description: Include only variables and inventory groups needed to reproduce the problem.
|
||||
render: yaml
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: logs
|
||||
attributes:
|
||||
label: Relevant logs or task output
|
||||
description: Include the failing task and surrounding output. Redact sensitive or identifying values.
|
||||
render: shell
|
||||
|
||||
- type: textarea
|
||||
id: context
|
||||
attributes:
|
||||
label: Additional context
|
||||
description: Add attempted fixes, suspected causes, regressions, or other useful context.
|
||||
@@ -0,0 +1,9 @@
|
||||
---
|
||||
blank_issues_enabled: false
|
||||
contact_links:
|
||||
- name: Troubleshooting and support
|
||||
url: https://github.com/timothystewart6/k3s-ansible/discussions/20
|
||||
about: Review common troubleshooting guidance and ask configuration or usage questions.
|
||||
- name: General discussions
|
||||
url: https://github.com/timothystewart6/k3s-ansible/discussions
|
||||
about: Discuss ideas and questions that are not confirmed bugs or concrete feature requests.
|
||||
@@ -0,0 +1,55 @@
|
||||
---
|
||||
name: Feature request
|
||||
description: Propose a focused improvement to supported repository behavior.
|
||||
title: "[Feature]: "
|
||||
labels:
|
||||
- enhancement
|
||||
body:
|
||||
- type: markdown
|
||||
attributes:
|
||||
value: |
|
||||
Describe the use case before proposing an implementation. Search existing issues and discussions first.
|
||||
|
||||
- type: checkboxes
|
||||
id: prerequisites
|
||||
attributes:
|
||||
label: Prerequisites
|
||||
options:
|
||||
- label: I searched existing issues and discussions for this request.
|
||||
required: true
|
||||
- label: This request is about reusable project behavior, not support for one private environment.
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: problem
|
||||
attributes:
|
||||
label: Problem or use case
|
||||
description: What limitation exists, who encounters it, and why does it matter?
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: proposal
|
||||
attributes:
|
||||
label: Proposed behavior
|
||||
description: Describe the desired user-visible result. Include example variables or commands when useful.
|
||||
validations:
|
||||
required: true
|
||||
|
||||
- type: textarea
|
||||
id: alternatives
|
||||
attributes:
|
||||
label: Alternatives considered
|
||||
description: Describe workarounds or other designs and their tradeoffs.
|
||||
|
||||
- type: textarea
|
||||
id: compatibility
|
||||
attributes:
|
||||
label: Compatibility and operational impact
|
||||
description: Note affected operating systems, architectures, CNIs, existing clusters, or reset behavior.
|
||||
|
||||
- type: textarea
|
||||
id: context
|
||||
attributes:
|
||||
label: Additional context
|
||||
description: Add relevant upstream documentation, examples, or prior discussion.
|
||||
@@ -1,15 +1,46 @@
|
||||
# Proposed Changes
|
||||
<!--- Provide a general summary of your changes -->
|
||||
## Summary
|
||||
|
||||
<!-- Explain the problem and the resulting behavior. Keep implementation details in the sections below. -->
|
||||
|
||||
## Changes
|
||||
|
||||
-
|
||||
-
|
||||
-
|
||||
|
||||
## Checklist
|
||||
## Related issues
|
||||
|
||||
- [ ] Tested locally
|
||||
- [ ] Ran `site.yml` playbook
|
||||
- [ ] Ran `reset.yml` playbook
|
||||
- [ ] Did not add any unnecessary changes
|
||||
- [ ] Ran pre-commit install at least once before committing
|
||||
- [ ] 🚀
|
||||
<!-- Use "Fixes #123" when this pull request should close an issue. Write "None" when not applicable. -->
|
||||
|
||||
## Testing
|
||||
|
||||
<!-- List exact commands, scenarios, and relevant manual checks. Do not check a box for a test that was not run. -->
|
||||
|
||||
- [ ] `pre-commit run --all-files`
|
||||
- [ ] Relevant Ansible syntax checks
|
||||
- [ ] Relevant focused regression tests
|
||||
- [ ] Relevant Molecule scenario
|
||||
- [ ] Provisioning tested against a non-production cluster
|
||||
- [ ] Reset behavior tested against a non-production cluster
|
||||
|
||||
Not run, with reason:
|
||||
|
||||
## Risk and compatibility
|
||||
|
||||
<!-- Cover existing clusters, upgrades, networking, supported platforms, security, and rollback. Write "None" when a
|
||||
category is not affected. -->
|
||||
|
||||
- Existing cluster or upgrade impact:
|
||||
- Networking or CNI impact:
|
||||
- Security impact:
|
||||
- Rollback plan:
|
||||
|
||||
## Documentation
|
||||
|
||||
<!-- Identify updated docs and sample configuration, or explain why no documentation change is needed. -->
|
||||
|
||||
## Final checklist
|
||||
|
||||
- [ ] The change is focused and contains no unrelated edits.
|
||||
- [ ] Tests cover new behavior or a regression, where applicable.
|
||||
- [ ] User-facing variables are documented in role defaults, sample inventory, and the README.
|
||||
- [ ] Logs, examples, and configuration contain no secrets or identifying infrastructure details.
|
||||
- [ ] Generated files and local environment files are not included.
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
---
|
||||
self-hosted-runner:
|
||||
labels:
|
||||
- k3s-ci
|
||||
- virtualbox
|
||||
- nested-virt
|
||||
@@ -0,0 +1,6 @@
|
||||
# GitHub Copilot instructions
|
||||
|
||||
Read and follow the repository's root-level `AGENTS.md` before proposing or making changes. It is the canonical guide
|
||||
for architecture, safety, implementation, validation, and documentation expectations.
|
||||
|
||||
Do not duplicate repository guidance here. If instructions need to change, update `AGENTS.md`.
|
||||
@@ -9,3 +9,18 @@ updates:
|
||||
ignore:
|
||||
- dependency-name: "*"
|
||||
update-types: ["version-update:semver-major"]
|
||||
|
||||
- package-ecosystem: "github-actions"
|
||||
directory: "/"
|
||||
schedule:
|
||||
interval: "daily"
|
||||
rebase-strategy: "auto"
|
||||
|
||||
- package-ecosystem: "docker"
|
||||
directory: "/"
|
||||
schedule:
|
||||
interval: "daily"
|
||||
rebase-strategy: "auto"
|
||||
ignore:
|
||||
- dependency-name: "*"
|
||||
update-types: ["version-update:semver-major"]
|
||||
|
||||
+75
-21
@@ -1,37 +1,91 @@
|
||||
#!/bin/bash
|
||||
|
||||
# download-boxes.sh
|
||||
# Check all molecule.yml files for required Vagrant boxes and download the ones that are not
|
||||
# already present on the system.
|
||||
# Validate the pinned Vagrant box set and download exact versions that are not
|
||||
# already present in VAGRANT_HOME.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
GIT_ROOT=$(git rev-parse --show-toplevel)
|
||||
PROVIDER=virtualbox
|
||||
LOCK_FILE="${VAGRANT_BOX_LOCK_FILE:-${GIT_ROOT}/.github/vagrant-boxes.lock}"
|
||||
|
||||
# Read all boxes for all platforms from the "molecule.yml" files
|
||||
all_boxes=$(cat "${GIT_ROOT}"/molecule/*/molecule.yml |
|
||||
yq -r '.platforms[].box' | # Read the "box" property of each node under "platforms"
|
||||
grep --invert-match --regexp=--- | # Filter out file separators
|
||||
sort |
|
||||
uniq)
|
||||
MOLECULE_YML_PATH=("${GIT_ROOT}"/molecule/*/molecule.yml)
|
||||
|
||||
# Read the boxes that are currently present on the system (for the current provider)
|
||||
# Extract the unique boxes referenced by the scenarios.
|
||||
declared_boxes=$(for file in "${MOLECULE_YML_PATH[@]}"; do
|
||||
yq -r '.platforms[].box' "$file"
|
||||
done | sort -u)
|
||||
|
||||
if [[ ! -r "$LOCK_FILE" ]]; then
|
||||
printf 'Vagrant box lock file is missing or unreadable: %s\n' "$LOCK_FILE" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
lock_entries=$(awk '
|
||||
/^[[:space:]]*#/ || NF == 0 { next }
|
||||
NF != 3 {
|
||||
printf "Invalid lock entry on line %d: expected box, version, architecture\n", NR > "/dev/stderr"
|
||||
invalid = 1
|
||||
next
|
||||
}
|
||||
{ print $1 " " $2 " " $3 }
|
||||
END { exit invalid }
|
||||
' "$LOCK_FILE")
|
||||
|
||||
duplicate_boxes=$(printf '%s\n' "$lock_entries" | awk '{ print $1 }' | sort | uniq -d)
|
||||
if [[ -n "$duplicate_boxes" ]]; then
|
||||
printf 'Duplicate Vagrant box lock entries:\n%s\n' "$duplicate_boxes" >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
locked_boxes=$(printf '%s\n' "$lock_entries" | sort)
|
||||
locked_names=$(printf '%s\n' "$locked_boxes" | awk '{ print $1 }')
|
||||
missing_locks=$(comm -23 <(printf '%s\n' "$declared_boxes") <(printf '%s\n' "$locked_names"))
|
||||
unused_locks=$(comm -13 <(printf '%s\n' "$declared_boxes") <(printf '%s\n' "$locked_names"))
|
||||
|
||||
if [[ -n "$missing_locks" || -n "$unused_locks" ]]; then
|
||||
if [[ -n "$missing_locks" ]]; then
|
||||
printf 'Scenario boxes missing from the lock file:\n%s\n' "$missing_locks" >&2
|
||||
fi
|
||||
if [[ -n "$unused_locks" ]]; then
|
||||
printf 'Lock entries not referenced by a scenario:\n%s\n' "$unused_locks" >&2
|
||||
fi
|
||||
exit 1
|
||||
fi
|
||||
|
||||
printf 'Pinned Vagrant boxes:\n%s\n' "$locked_boxes"
|
||||
|
||||
# Read exact box, provider, version, and architecture tuples already present.
|
||||
present_boxes=$(
|
||||
(vagrant box list |
|
||||
grep "${PROVIDER}" | # Filter by boxes available for the current provider
|
||||
awk '{print $1;}' | # The box name is the first word in each line
|
||||
sort |
|
||||
uniq) ||
|
||||
echo "" # In case any of these commands fails, just use an empty list
|
||||
vagrant box list --machine-readable |
|
||||
awk -F, -v expected_provider="$PROVIDER" '
|
||||
$3 == "box-name" { name = $4; next }
|
||||
$3 == "box-provider" { provider = $4; next }
|
||||
$3 == "box-version" { version = $4; next }
|
||||
$3 == "box-architecture" {
|
||||
architecture = $4
|
||||
if (provider == expected_provider) {
|
||||
print name " " version " " architecture
|
||||
}
|
||||
name = provider = version = architecture = ""
|
||||
}
|
||||
' |
|
||||
sort -u
|
||||
)
|
||||
|
||||
# The boxes that we need to download are the ones present in $all_boxes, but not $present_boxes.
|
||||
download_boxes=$(comm -2 -3 <(echo "${all_boxes}") <(echo "${present_boxes}"))
|
||||
download_boxes=$(comm -23 \
|
||||
<(printf '%s\n' "$locked_boxes") \
|
||||
<(printf '%s\n' "$present_boxes"))
|
||||
|
||||
# Actually download the necessary boxes
|
||||
if [ -n "${download_boxes}" ]; then
|
||||
echo "${download_boxes}" | while IFS= read -r box; do
|
||||
vagrant box add --provider "${PROVIDER}" "${box}"
|
||||
if [[ -n "$download_boxes" ]]; then
|
||||
printf '%s\n' "$download_boxes" | while read -r box version architecture; do
|
||||
vagrant box add \
|
||||
--provider "$PROVIDER" \
|
||||
--box-version "$version" \
|
||||
--architecture "$architecture" \
|
||||
"$box"
|
||||
done
|
||||
else
|
||||
printf 'All pinned Vagrant boxes are already present.\n'
|
||||
fi
|
||||
|
||||
Executable
+249
@@ -0,0 +1,249 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
usage() {
|
||||
printf '%s\n' \
|
||||
'Usage: cleanup-runner-resources.sh [--snapshot|--dry-run|--apply]' \
|
||||
'' \
|
||||
'Discover and, with --apply, remove only VirtualBox resources referenced by' \
|
||||
'repository-owned Molecule Vagrant state. The default is --dry-run.'
|
||||
}
|
||||
|
||||
mode="dry-run"
|
||||
case "${1:-}" in
|
||||
"") ;;
|
||||
--snapshot) mode="snapshot" ;;
|
||||
--dry-run) mode="dry-run" ;;
|
||||
--apply) mode="apply" ;;
|
||||
--help|-h) usage; exit 0 ;;
|
||||
*) usage >&2; exit 2 ;;
|
||||
esac
|
||||
|
||||
home_dir="${HOME:?HOME must be set}"
|
||||
molecule_root="${K3S_CI_MOLECULE_ROOT:-${home_dir}/.cache/molecule}"
|
||||
repository_name="${K3S_CI_MOLECULE_PROJECT:-k3s-ansible}"
|
||||
virtualbox_root="${K3S_CI_VIRTUALBOX_ROOT:-${home_dir}/VirtualBox VMs}"
|
||||
hostonly_marker="${K3S_CI_HOSTONLY_MARKER:-${home_dir}/.cache/k3s-ci/hostonly-interfaces}"
|
||||
|
||||
record_hostonly() {
|
||||
local marker_dir="${hostonly_marker%/*}"
|
||||
local marker_tmp="${hostonly_marker}.tmp"
|
||||
local hostonly_inventory
|
||||
if ! hostonly_inventory="$(VBoxManage list hostonlyifs)"; then
|
||||
fail_closed 'unable to inventory VirtualBox host-only interfaces'
|
||||
fi
|
||||
mkdir -p -- "$marker_dir"
|
||||
awk -F': ' '
|
||||
/^Name:/ { name=$2 }
|
||||
/^IPAddress:/ { print name "|" $2 }
|
||||
' <<< "$hostonly_inventory" > "$marker_tmp"
|
||||
mv -- "$marker_tmp" "$hostonly_marker"
|
||||
chmod 600 "$hostonly_marker"
|
||||
printf 'Recorded host-only interface baseline: %s\n' "$hostonly_marker"
|
||||
}
|
||||
|
||||
cleanup_hostonly() {
|
||||
local hostonly_inventory
|
||||
if [[ ! -f "$hostonly_marker" ]]; then
|
||||
printf 'No host-only interface baseline found; leaving interfaces unchanged.\n'
|
||||
return 0
|
||||
fi
|
||||
|
||||
if ! hostonly_inventory="$(VBoxManage list hostonlyifs)"; then
|
||||
fail_closed 'unable to inventory VirtualBox host-only interfaces'
|
||||
fi
|
||||
|
||||
while IFS='|' read -r interface_name interface_ip; do
|
||||
[[ "$interface_name" == vboxnet* ]] || continue
|
||||
[[ "$interface_ip" == 192.168.30.* || "$interface_ip" == fdad:bad:ba55:* ]] || continue
|
||||
if grep -Fqx "${interface_name}|${interface_ip}" "$hostonly_marker"; then
|
||||
continue
|
||||
fi
|
||||
if [[ "$mode" == apply ]]; then
|
||||
VBoxManage hostonlyif remove "$interface_name"
|
||||
printf 'Removed host-only interface %s (%s)\n' "$interface_name" "$interface_ip"
|
||||
else
|
||||
printf 'Would remove host-only interface %s (%s)\n' "$interface_name" "$interface_ip"
|
||||
fi
|
||||
done < <(awk -F': ' '
|
||||
/^Name:/ { name=$2 }
|
||||
/^IPAddress:/ { print name "|" $2 }
|
||||
' <<< "$hostonly_inventory")
|
||||
}
|
||||
|
||||
if [[ "$mode" == snapshot ]]; then
|
||||
record_hostonly
|
||||
exit 0
|
||||
fi
|
||||
|
||||
resolve_existing_dir() {
|
||||
local candidate="$1"
|
||||
if [[ ! -d "$candidate" ]]; then
|
||||
return 1
|
||||
fi
|
||||
readlink -f -- "$candidate"
|
||||
}
|
||||
|
||||
root_contains() {
|
||||
local root="$1"
|
||||
local path="$2"
|
||||
[[ "$path" == "$root"/* ]]
|
||||
}
|
||||
|
||||
is_supported_scenario() {
|
||||
case "$1" in
|
||||
default|single_node|calico|cilium|kube-vip|ipv6) return 0 ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
|
||||
is_unregistered_vm_error() {
|
||||
grep -Eq 'Could not find a registered machine|VBOX_E_OBJECT_NOT_FOUND'
|
||||
}
|
||||
|
||||
fail_closed() {
|
||||
printf 'cleanup refused: %s\n' "$1" >&2
|
||||
exit 3
|
||||
}
|
||||
|
||||
if [[ ! "$repository_name" =~ ^[A-Za-z0-9._-]+$ ]]; then
|
||||
fail_closed 'invalid Molecule repository name'
|
||||
fi
|
||||
|
||||
print_inventory() {
|
||||
local phase="$1"
|
||||
printf '%s VirtualBox inventory:\n' "$phase"
|
||||
VBoxManage list vms || true
|
||||
VBoxManage list hdds || true
|
||||
VBoxManage list hostonlyifs || true
|
||||
}
|
||||
|
||||
molecule_root_real="$(resolve_existing_dir "$molecule_root" || true)"
|
||||
if [[ -z "$molecule_root_real" ]]; then
|
||||
printf 'No Molecule root exists: %s\n' "$molecule_root"
|
||||
cleanup_hostonly
|
||||
exit 0
|
||||
fi
|
||||
|
||||
print_inventory before
|
||||
|
||||
repository_root_real="$(resolve_existing_dir "$molecule_root_real/$repository_name" || true)"
|
||||
if [[ -z "$repository_root_real" ]]; then
|
||||
printf 'No repository Molecule state root exists: %s\n' "$molecule_root_real/$repository_name"
|
||||
cleanup_hostonly
|
||||
print_inventory after
|
||||
exit 0
|
||||
fi
|
||||
if ! root_contains "$molecule_root_real" "$repository_root_real"; then
|
||||
fail_closed "repository Molecule state root is outside Molecule root: $repository_root_real"
|
||||
fi
|
||||
|
||||
declare -a state_files=()
|
||||
while IFS= read -r -d '' state_file; do
|
||||
state_files+=("$state_file")
|
||||
done < <(find "$repository_root_real" -mindepth 6 -maxdepth 6 -type f \
|
||||
-path '*/.vagrant/machines/*/virtualbox/id' -print0 2>/dev/null)
|
||||
|
||||
if ((${#state_files[@]} == 0)); then
|
||||
printf 'No repository-owned Molecule Vagrant state found under %s\n' "$repository_root_real"
|
||||
cleanup_hostonly
|
||||
print_inventory after
|
||||
exit 0
|
||||
fi
|
||||
|
||||
virtualbox_root_real="$(resolve_existing_dir "$virtualbox_root" || true)"
|
||||
|
||||
declare -a vm_records=()
|
||||
for state_file in "${state_files[@]}"; do
|
||||
if [[ ! -f "$state_file" ]]; then
|
||||
printf 'Skipping Vagrant state removed with its stale scenario directory: %s\n' "$state_file"
|
||||
continue
|
||||
fi
|
||||
state_file_real="$(readlink -f -- "$state_file")"
|
||||
state_dir="${state_file_real%/.vagrant/machines/*/virtualbox/id}"
|
||||
machine_dir="${state_file_real%/virtualbox/id}"
|
||||
machine_name="${machine_dir##*/}"
|
||||
scenario_name="${state_dir##*/}"
|
||||
|
||||
if ! root_contains "$repository_root_real" "$state_dir"; then
|
||||
fail_closed "state path is outside the repository Molecule root: $state_file_real"
|
||||
fi
|
||||
if ! is_supported_scenario "$scenario_name"; then
|
||||
fail_closed "unexpected Molecule scenario: $scenario_name"
|
||||
fi
|
||||
if [[ "$machine_name" != control* && "$machine_name" != node* ]]; then
|
||||
fail_closed "unexpected Molecule machine name: $machine_name"
|
||||
fi
|
||||
|
||||
vm_uuid="$(tr -d '[:space:]' < "$state_file_real")"
|
||||
if [[ ! "$vm_uuid" =~ ^[0-9a-fA-F-]{36}$ ]]; then
|
||||
fail_closed "invalid VirtualBox UUID in $state_file_real"
|
||||
fi
|
||||
|
||||
if ! vm_info="$(VBoxManage showvminfo "$vm_uuid" --machinereadable 2>&1)"; then
|
||||
if is_unregistered_vm_error <<< "$vm_info"; then
|
||||
printf 'Stale Vagrant state without a registered VM: %s (%s)\n' "$machine_name" "$vm_uuid"
|
||||
if [[ "$mode" == apply ]]; then
|
||||
rm -rf -- "${state_dir}/.vagrant"
|
||||
printf 'Removed stale Vagrant state: %s\n' "${state_dir}/.vagrant"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
fail_closed "unable to inspect VirtualBox VM $vm_uuid: $vm_info"
|
||||
fi
|
||||
|
||||
if [[ -z "$virtualbox_root_real" ]]; then
|
||||
fail_closed "VirtualBox VM root does not exist: $virtualbox_root"
|
||||
fi
|
||||
|
||||
cfg_file="$(awk -F= '$1 == "CfgFile" {gsub(/"/, "", $2); print $2; exit}' <<< "$vm_info")"
|
||||
if [[ -z "$cfg_file" ]]; then
|
||||
fail_closed "VirtualBox configuration path missing for $vm_uuid"
|
||||
fi
|
||||
cfg_file_real="$(readlink -f -- "$cfg_file")"
|
||||
if ! root_contains "$virtualbox_root_real" "$cfg_file_real"; then
|
||||
fail_closed "VM configuration is outside VirtualBox root: $cfg_file_real"
|
||||
fi
|
||||
|
||||
while IFS= read -r disk_path; do
|
||||
[[ -z "$disk_path" ]] && continue
|
||||
disk_path_real="$(readlink -f -- "$disk_path" 2>/dev/null || true)"
|
||||
if [[ -z "$disk_path_real" ]] || ! root_contains "$virtualbox_root_real" "$disk_path_real"; then
|
||||
fail_closed "attached disk is outside VirtualBox root: $disk_path"
|
||||
fi
|
||||
done < <(awk -F= '$1 ~ /^(SATA|IDE|SCSI|SAS|VirtioSCSI|NVMe)-[0-9]+-[0-9]+$/ {gsub(/"/, "", $2); print $2}' <<< "$vm_info")
|
||||
|
||||
vm_records+=("$vm_uuid|$machine_name|$cfg_file_real")
|
||||
done
|
||||
|
||||
if ((${#vm_records[@]} == 0)); then
|
||||
printf 'No live repository-owned VirtualBox resources found\n'
|
||||
cleanup_hostonly
|
||||
print_inventory after
|
||||
exit 0
|
||||
fi
|
||||
|
||||
for record in "${vm_records[@]}"; do
|
||||
IFS='|' read -r vm_uuid machine_name cfg_file_real <<< "$record"
|
||||
if [[ "$mode" == dry-run ]]; then
|
||||
printf 'Would remove VM %s (%s) config=%s\n' "$machine_name" "$vm_uuid" "$cfg_file_real"
|
||||
continue
|
||||
fi
|
||||
|
||||
vm_state="$(VBoxManage showvminfo "$vm_uuid" --machinereadable | awk -F= '$1 == "VMState" {gsub(/"/, "", $2); print $2; exit}')"
|
||||
if [[ "$vm_state" != poweroff && "$vm_state" != saved ]]; then
|
||||
VBoxManage controlvm "$vm_uuid" poweroff
|
||||
fi
|
||||
VBoxManage unregistervm "$vm_uuid" --delete
|
||||
printf 'Removed VM %s (%s)\n' "$machine_name" "$vm_uuid"
|
||||
done
|
||||
|
||||
cleanup_hostonly
|
||||
print_inventory after
|
||||
|
||||
if [[ "$mode" == apply ]]; then
|
||||
printf 'Repository-owned VM and host-only interface cleanup complete.\n'
|
||||
else
|
||||
printf 'Dry run complete. No resources were modified.\n'
|
||||
fi
|
||||
+47
@@ -0,0 +1,47 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
output_dir="${1:-${RUNNER_TEMP:-/tmp}/k3s-ci-diagnostics}"
|
||||
mkdir -p -- "$output_dir"
|
||||
umask 077
|
||||
|
||||
run_capture() {
|
||||
local output_file="$1"
|
||||
shift
|
||||
{
|
||||
printf '$'
|
||||
printf ' %q' "$@"
|
||||
printf '\n'
|
||||
"$@"
|
||||
} > "$output_dir/$output_file" 2>&1 || true
|
||||
}
|
||||
|
||||
run_capture system.txt uname -a
|
||||
run_capture runner-user.txt id
|
||||
run_capture memory.txt free -h
|
||||
run_capture disk.txt df -h
|
||||
run_capture virtualbox-version VBoxManage --version
|
||||
run_capture virtualbox-vms VBoxManage list vms
|
||||
run_capture virtualbox-running-vms VBoxManage list runningvms
|
||||
run_capture virtualbox-disks VBoxManage list hdds
|
||||
run_capture virtualbox-hostonlyifs VBoxManage list hostonlyifs
|
||||
run_capture virtualbox-groups VBoxManage list groups
|
||||
run_capture vagrant-status vagrant global-status
|
||||
run_capture molecule-state find "${HOME}/.cache/molecule" -maxdepth 6 -type f -path '*/.vagrant/machines/*/virtualbox/id' -print
|
||||
|
||||
scenario_name="${K3S_CI_SCENARIO_NAME:-}"
|
||||
if [[ "$scenario_name" =~ ^[A-Za-z0-9_-]+$ ]]; then
|
||||
molecule_state_dir="${HOME}/.cache/molecule/k3s-ansible/${scenario_name}"
|
||||
for log_name in vagrant.out vagrant.err; do
|
||||
if [[ -r "${molecule_state_dir}/${log_name}" ]]; then
|
||||
cp -- "${molecule_state_dir}/${log_name}" "$output_dir/${scenario_name}-${log_name}"
|
||||
fi
|
||||
done
|
||||
fi
|
||||
|
||||
if [[ -r /etc/vbox/networks.conf ]]; then
|
||||
cp -- /etc/vbox/networks.conf "$output_dir/virtualbox-networks.conf"
|
||||
fi
|
||||
|
||||
printf 'Diagnostics written to %s\n' "$output_dir"
|
||||
Executable
+44
@@ -0,0 +1,44 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
output_dir="${1:?output directory is required}"
|
||||
interval="${2:-10}"
|
||||
[[ "$interval" =~ ^[1-9][0-9]*$ ]] || {
|
||||
printf 'monitor interval must be a positive integer\n' >&2
|
||||
exit 2
|
||||
}
|
||||
|
||||
mkdir -p -- "$output_dir"
|
||||
umask 077
|
||||
|
||||
free -h > "$output_dir/memory-before.txt"
|
||||
df -h > "$output_dir/disk-before.txt"
|
||||
vmstat -w "$interval" > "$output_dir/vmstat.txt" &
|
||||
vmstat_pid=$!
|
||||
|
||||
iostat_pid=""
|
||||
if command -v iostat >/dev/null 2>&1; then
|
||||
iostat -dx "$interval" > "$output_dir/iostat.txt" &
|
||||
iostat_pid=$!
|
||||
else
|
||||
printf 'iostat is not installed on this runner\n' > "$output_dir/iostat-unavailable.txt"
|
||||
fi
|
||||
|
||||
cleanup() {
|
||||
local rc=$?
|
||||
trap - EXIT INT TERM
|
||||
kill "$vmstat_pid" 2>/dev/null || true
|
||||
[[ -z "$iostat_pid" ]] || kill "$iostat_pid" 2>/dev/null || true
|
||||
wait "$vmstat_pid" 2>/dev/null || true
|
||||
[[ -z "$iostat_pid" ]] || wait "$iostat_pid" 2>/dev/null || true
|
||||
free -h > "$output_dir/memory-after.txt"
|
||||
df -h > "$output_dir/disk-after.txt"
|
||||
exit "$rc"
|
||||
}
|
||||
trap cleanup EXIT INT TERM
|
||||
|
||||
while :; do
|
||||
sleep 3600 &
|
||||
wait $!
|
||||
done
|
||||
+211
@@ -0,0 +1,211 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
fail() {
|
||||
printf 'Vagrant box master preparation refused: %s\n' "$1" >&2
|
||||
exit 3
|
||||
}
|
||||
|
||||
root_contains() {
|
||||
local root="$1"
|
||||
local path="$2"
|
||||
[[ "$path" == "$root"/* ]]
|
||||
}
|
||||
|
||||
read_machine_value() {
|
||||
local machine_info="$1"
|
||||
local key="$2"
|
||||
awk -F= -v key="$key" '$1 == key {gsub(/"/, "", $2); print $2; exit}' <<< "$machine_info"
|
||||
}
|
||||
|
||||
read_extra_data() {
|
||||
local uuid="$1"
|
||||
local key="$2"
|
||||
local value
|
||||
value="$(VBoxManage getextradata "$uuid" "$key" 2>/dev/null || true)"
|
||||
[[ "$value" == 'Value: '* ]] || return 1
|
||||
printf '%s\n' "${value#Value: }"
|
||||
}
|
||||
|
||||
validate_owned_master() {
|
||||
local uuid="$1"
|
||||
local box="$2"
|
||||
local version="$3"
|
||||
local architecture="$4"
|
||||
local machine_info cfg_file cfg_file_real vm_state groups disk_path disk_path_real
|
||||
|
||||
[[ "$uuid" =~ ^[0-9a-fA-F-]{36}$ ]] || return 1
|
||||
machine_info="$(VBoxManage showvminfo "$uuid" --machinereadable 2>/dev/null)" || return 1
|
||||
vm_state="$(read_machine_value "$machine_info" VMState)"
|
||||
groups="$(read_machine_value "$machine_info" groups)"
|
||||
cfg_file="$(read_machine_value "$machine_info" CfgFile)"
|
||||
[[ "$vm_state" == poweroff ]] || return 1
|
||||
[[ ",$groups," == *,/k3s-ansible/box-masters,* ]] || return 1
|
||||
[[ -n "$cfg_file" ]] || return 1
|
||||
cfg_file_real="$(readlink -f -- "$cfg_file" 2>/dev/null || true)"
|
||||
[[ -n "$cfg_file_real" ]] || return 1
|
||||
root_contains "$virtualbox_root_real" "$cfg_file_real" || return 1
|
||||
|
||||
[[ "$(read_extra_data "$uuid" k3s-ansible/owner || true)" == box-master ]] || return 1
|
||||
[[ "$(read_extra_data "$uuid" k3s-ansible/box || true)" == "$box" ]] || return 1
|
||||
[[ "$(read_extra_data "$uuid" k3s-ansible/version || true)" == "$version" ]] || return 1
|
||||
[[ "$(read_extra_data "$uuid" k3s-ansible/architecture || true)" == "$architecture" ]] || return 1
|
||||
|
||||
while IFS= read -r disk_path; do
|
||||
[[ -z "$disk_path" || "$disk_path" == none ]] && continue
|
||||
disk_path_real="$(readlink -f -- "$disk_path" 2>/dev/null || true)"
|
||||
[[ -n "$disk_path_real" ]] || return 1
|
||||
root_contains "$virtualbox_root_real" "$disk_path_real" || return 1
|
||||
done < <(awk -F= '$1 ~ /^(SATA|IDE|SCSI|SAS|VirtioSCSI|NVMe)-[0-9]+-[0-9]+$/ {
|
||||
gsub(/"/, "", $2); print $2
|
||||
}' <<< "$machine_info")
|
||||
}
|
||||
|
||||
write_prewarm_vagrantfile() {
|
||||
local destination="$1"
|
||||
local box="$2"
|
||||
local version="$3"
|
||||
{
|
||||
printf '%s\n' "Vagrant.configure('2') do |config|"
|
||||
printf ' config.vm.box = "%s"\n' "$box"
|
||||
printf ' config.vm.box_version = "%s"\n' "$version"
|
||||
printf '%s\n' \
|
||||
' config.vm.synced_folder ".", "/vagrant", disabled: true' \
|
||||
' config.vm.hostname = "k3s-ansible-box-prewarm"' \
|
||||
' config.vm.boot_timeout = 600' \
|
||||
' config.vm.provider "virtualbox" do |virtualbox|' \
|
||||
' virtualbox.linked_clone = true' \
|
||||
' virtualbox.memory = 1024' \
|
||||
' virtualbox.cpus = 2' \
|
||||
' end' \
|
||||
'end'
|
||||
} > "$destination"
|
||||
}
|
||||
|
||||
cleanup_prewarm() {
|
||||
local rc=$?
|
||||
trap - EXIT
|
||||
if [[ -n "${prewarm_dir:-}" && -d "$prewarm_dir" ]]; then
|
||||
VAGRANT_CWD="$prewarm_dir" vagrant destroy --force >/dev/null 2>&1 || true
|
||||
rm -rf -- "$prewarm_dir"
|
||||
fi
|
||||
exit "$rc"
|
||||
}
|
||||
|
||||
create_owned_master() {
|
||||
local box="$1"
|
||||
local version="$2"
|
||||
local architecture="$3"
|
||||
local master_id_file="$4"
|
||||
local mapping_file="$5"
|
||||
local uuid machine_info cfg_file cfg_file_real vm_state mapping_tmp
|
||||
|
||||
# A master_id restored from an immutable cache is only a hint. Without the
|
||||
# runner-local ownership record and matching VirtualBox metadata it is not
|
||||
# trusted, adopted, modified, or deleted.
|
||||
rm -f -- "$master_id_file"
|
||||
|
||||
prewarm_dir="$(mktemp -d "${master_root}/prewarm.XXXXXX")"
|
||||
trap cleanup_prewarm EXIT
|
||||
write_prewarm_vagrantfile "$prewarm_dir/Vagrantfile" "$box" "$version"
|
||||
|
||||
printf 'Creating runner-owned linked-clone master for %s %s %s\n' \
|
||||
"$box" "$version" "$architecture"
|
||||
VAGRANT_CWD="$prewarm_dir" vagrant up --provider virtualbox --no-provision
|
||||
|
||||
[[ -r "$master_id_file" ]] || fail "Vagrant did not record a master UUID for $box"
|
||||
uuid="$(tr -d '[:space:]' < "$master_id_file")"
|
||||
[[ "$uuid" =~ ^[0-9a-fA-F-]{36}$ ]] || fail "Vagrant recorded an invalid master UUID for $box"
|
||||
|
||||
machine_info="$(VBoxManage showvminfo "$uuid" --machinereadable 2>/dev/null)" || \
|
||||
fail "Vagrant master $uuid for $box is not registered"
|
||||
vm_state="$(read_machine_value "$machine_info" VMState)"
|
||||
cfg_file="$(read_machine_value "$machine_info" CfgFile)"
|
||||
cfg_file_real="$(readlink -f -- "$cfg_file" 2>/dev/null || true)"
|
||||
[[ "$vm_state" == poweroff ]] || fail "new Vagrant master $uuid is not powered off"
|
||||
if [[ -z "$cfg_file_real" ]] || ! root_contains "$virtualbox_root_real" "$cfg_file_real"; then
|
||||
fail "new Vagrant master $uuid is outside the runner VirtualBox root"
|
||||
fi
|
||||
|
||||
VAGRANT_CWD="$prewarm_dir" vagrant destroy --force
|
||||
VBoxManage modifyvm "$uuid" --groups /k3s-ansible/box-masters
|
||||
VBoxManage setextradata "$uuid" k3s-ansible/owner box-master
|
||||
VBoxManage setextradata "$uuid" k3s-ansible/box "$box"
|
||||
VBoxManage setextradata "$uuid" k3s-ansible/version "$version"
|
||||
VBoxManage setextradata "$uuid" k3s-ansible/architecture "$architecture"
|
||||
validate_owned_master "$uuid" "$box" "$version" "$architecture" || \
|
||||
fail "new Vagrant master $uuid failed ownership validation"
|
||||
|
||||
rm -rf -- "$prewarm_dir"
|
||||
prewarm_dir=""
|
||||
trap - EXIT
|
||||
|
||||
mapping_tmp="${mapping_file}.tmp"
|
||||
printf '%s\n' "$uuid" > "$mapping_tmp"
|
||||
chmod 600 "$mapping_tmp"
|
||||
mv -- "$mapping_tmp" "$mapping_file"
|
||||
printf '%s\n' "$uuid" > "$master_id_file"
|
||||
chmod 600 "$master_id_file"
|
||||
printf 'Created and recorded owned master %s for %s\n' "$uuid" "$box"
|
||||
}
|
||||
|
||||
repository_root="${K3S_CI_REPOSITORY_ROOT:-$(git rev-parse --show-toplevel)}"
|
||||
lock_file="${VAGRANT_BOX_LOCK_FILE:-${repository_root}/.github/vagrant-boxes.lock}"
|
||||
vagrant_home="${VAGRANT_HOME:?VAGRANT_HOME must be set}"
|
||||
master_root="${K3S_CI_VAGRANT_MASTER_ROOT:-${HOME:?HOME must be set}/.cache/k3s-ci/vagrant-masters}"
|
||||
virtualbox_root="${K3S_CI_VIRTUALBOX_ROOT:-${HOME}/VirtualBox VMs}"
|
||||
|
||||
[[ -r "$lock_file" ]] || fail "box lock file is missing or unreadable: $lock_file"
|
||||
[[ -d "$vagrant_home/boxes" ]] || fail "Vagrant box directory is missing: $vagrant_home/boxes"
|
||||
[[ -d "$virtualbox_root" ]] || fail "VirtualBox root is missing: $virtualbox_root"
|
||||
|
||||
box_root_real="$(readlink -f -- "$vagrant_home/boxes")"
|
||||
virtualbox_root_real="$(readlink -f -- "$virtualbox_root")"
|
||||
mkdir -p -- "$master_root"
|
||||
chmod 700 "$master_root"
|
||||
|
||||
exec 9> "${master_root}/prepare.lock"
|
||||
flock 9
|
||||
|
||||
lock_entries="$(awk '
|
||||
/^[[:space:]]*#/ || NF == 0 { next }
|
||||
NF != 3 { invalid = 1; next }
|
||||
{ print $1 " " $2 " " $3 }
|
||||
END { exit invalid }
|
||||
' "$lock_file")" || fail 'invalid Vagrant box lock entry'
|
||||
[[ -n "$lock_entries" ]] || fail 'Vagrant box lock is empty'
|
||||
|
||||
while read -r box version architecture; do
|
||||
[[ "$box" =~ ^[A-Za-z0-9._-]+/[A-Za-z0-9._-]+$ ]] || fail "invalid box name: $box"
|
||||
[[ "$version" =~ ^[A-Za-z0-9._-]+$ ]] || fail "invalid box version: $version"
|
||||
[[ "$architecture" =~ ^[A-Za-z0-9._-]+$ ]] || fail "invalid box architecture: $architecture"
|
||||
|
||||
box_slug="${box//\//-VAGRANTSLASH-}"
|
||||
record_slug="${box//\//_}-${version}-${architecture}"
|
||||
box_dir="${vagrant_home}/boxes/${box_slug}/${version}/${architecture}/virtualbox"
|
||||
[[ -d "$box_dir" ]] || fail "pinned box is not installed: $box $version $architecture"
|
||||
box_dir_real="$(readlink -f -- "$box_dir")"
|
||||
root_contains "$box_root_real" "$box_dir_real" || fail "box directory is outside VAGRANT_HOME: $box_dir_real"
|
||||
|
||||
master_id_file="${box_dir_real}/master_id"
|
||||
mapping_file="${master_root}/${record_slug}.uuid"
|
||||
uuid=""
|
||||
if [[ -r "$mapping_file" ]]; then
|
||||
uuid="$(tr -d '[:space:]' < "$mapping_file")"
|
||||
fi
|
||||
|
||||
if [[ -n "$uuid" ]] && validate_owned_master "$uuid" "$box" "$version" "$architecture"; then
|
||||
printf '%s\n' "$uuid" > "$master_id_file"
|
||||
chmod 600 "$master_id_file"
|
||||
printf 'Reusing owned master %s for %s %s %s\n' "$uuid" "$box" "$version" "$architecture"
|
||||
continue
|
||||
fi
|
||||
|
||||
if [[ -e "$mapping_file" ]]; then
|
||||
printf 'Owned master record is stale for %s; rebuilding without deleting any VM or disk.\n' "$box"
|
||||
fi
|
||||
create_owned_master "$box" "$version" "$architecture" "$master_id_file" "$mapping_file"
|
||||
done <<< "$lock_entries"
|
||||
|
||||
printf 'All pinned Vagrant box masters are ready.\n'
|
||||
@@ -0,0 +1,134 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Render the Cilium BGP CRD template and assert it uses the v2 API.
|
||||
|
||||
This is a manifest-only regression test used where no real BGP peer is
|
||||
available. It renders roles/k3s_server_post/templates/cilium.crs.j2 with
|
||||
zero, one, and multiple neighbors, then checks that the output:
|
||||
- never contains CiliumBGPPeeringPolicy or cilium.io/v2alpha1
|
||||
- emits the Cilium v2 BGP resources
|
||||
- emits deterministic DNS-safe peer and instance names
|
||||
- advertises Pod CIDRs only when cilium_exportPodCIDR is true
|
||||
"""
|
||||
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
from jinja2 import Environment, FileSystemLoader, StrictUndefined
|
||||
|
||||
|
||||
def repo_root():
|
||||
return subprocess.check_output(
|
||||
["git", "rev-parse", "--show-toplevel"], text=True
|
||||
).strip()
|
||||
|
||||
|
||||
def fail(message):
|
||||
raise SystemExit("Cilium BGP manifest test failed: " + message)
|
||||
|
||||
|
||||
def render(env, extra_vars):
|
||||
base_vars = {
|
||||
"cilium_bgp_my_asn": "64513",
|
||||
"cilium_bgp_peer_asn": "64512",
|
||||
"cilium_bgp_peer_address": "192.168.30.1",
|
||||
"cilium_exportPodCIDR": True,
|
||||
"cilium_bgp_lb_cidr": "192.168.31.0/24",
|
||||
}
|
||||
base_vars.update(extra_vars)
|
||||
template = env.get_template("cilium.crs.j2")
|
||||
return template.render(**base_vars)
|
||||
|
||||
|
||||
def check_common(output):
|
||||
if "cilium.io/v2alpha1" in output:
|
||||
fail("rendered output still contains cilium.io/v2alpha1")
|
||||
if "kind: CiliumBGPPeeringPolicy" in output:
|
||||
fail("rendered output still contains CiliumBGPPeeringPolicy")
|
||||
for kind in (
|
||||
"CiliumBGPPeerConfig",
|
||||
"CiliumBGPClusterConfig",
|
||||
"CiliumBGPAdvertisement",
|
||||
"CiliumLoadBalancerIPPool",
|
||||
):
|
||||
if ("kind: " + kind) not in output:
|
||||
fail("rendered output is missing kind: " + kind)
|
||||
|
||||
|
||||
def main():
|
||||
root = repo_root()
|
||||
template_dir = os.path.join(
|
||||
root, "roles", "k3s_server_post", "templates"
|
||||
)
|
||||
env = Environment(
|
||||
loader=FileSystemLoader(template_dir), undefined=StrictUndefined
|
||||
)
|
||||
|
||||
# Zero neighbors -> fall back to the single default peer.
|
||||
output = render(env, {"_cilium_bgp_neighbors": []})
|
||||
check_common(output)
|
||||
if "peer-64512-1" not in output:
|
||||
fail("default single peer name was not rendered")
|
||||
if "peerAddress: 192.168.30.1" not in output:
|
||||
fail("default peer address was not rendered")
|
||||
if 'advertisementType: "PodCIDR"' not in output:
|
||||
fail("PodCIDR advertisement missing when exportPodCIDR is true")
|
||||
|
||||
# One neighbor via the merged list.
|
||||
output = render(
|
||||
env,
|
||||
{"_cilium_bgp_neighbors": [{"peer_address": "10.0.0.1", "peer_asn": "65001"}]},
|
||||
)
|
||||
check_common(output)
|
||||
if "peer-65001-1" not in output:
|
||||
fail("single merged peer name was not rendered")
|
||||
if "peerAddress: 10.0.0.1" not in output:
|
||||
fail("single merged peer address was not rendered")
|
||||
|
||||
# Multiple neighbors.
|
||||
output = render(
|
||||
env,
|
||||
{
|
||||
"_cilium_bgp_neighbors": [
|
||||
{"peer_address": "10.0.0.1", "peer_asn": "65001"},
|
||||
{"peer_address": "10.0.0.2", "peer_asn": "65002"},
|
||||
]
|
||||
},
|
||||
)
|
||||
check_common(output)
|
||||
if "peer-65001-1" not in output or "peer-65002-2" not in output:
|
||||
fail("multiple merged peer names were not rendered")
|
||||
if "peerAddress: 10.0.0.2" not in output:
|
||||
fail("second merged peer address was not rendered")
|
||||
|
||||
# exportPodCIDR false -> no PodCIDR advertisement, service remains.
|
||||
output = render(
|
||||
env, {"_cilium_bgp_neighbors": [], "cilium_exportPodCIDR": False}
|
||||
)
|
||||
check_common(output)
|
||||
if 'advertisementType: "PodCIDR"' in output:
|
||||
fail("PodCIDR advertisement present when exportPodCIDR is false")
|
||||
if 'advertisementType: "Service"' not in output:
|
||||
fail("Service advertisement missing when exportPodCIDR is false")
|
||||
|
||||
# Load balancer pools: CIDR and start/stop forms.
|
||||
output = render(env, {"_cilium_bgp_neighbors": []})
|
||||
if "cidr: 192.168.31.0/24" not in output:
|
||||
fail("CIDR load balancer pool was not rendered")
|
||||
output = render(
|
||||
env,
|
||||
{
|
||||
"_cilium_bgp_neighbors": [],
|
||||
"cilium_bgp_lb_cidr": "192.168.31.80-192.168.31.90",
|
||||
},
|
||||
)
|
||||
check_common(output)
|
||||
if "start: 192.168.31.80" not in output or "stop: 192.168.31.90" not in output:
|
||||
fail("start/stop load balancer pool was not rendered")
|
||||
|
||||
print("Cilium BGP manifest regression test passed")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,120 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Regression test for the Cilium Envoy toggle.
|
||||
|
||||
The `cilium_envoy` variable lets users enable or disable the Cilium Envoy
|
||||
proxy. The Install/upgrade Cilium task in
|
||||
roles/k3s_server_post/tasks/cilium.yml passes the value through to Helm as
|
||||
`envoy.enabled`. This test:
|
||||
|
||||
- loads the real "Install Cilium" task and confirms the install/upgrade
|
||||
command actually contains the `envoy.enabled` Helm value,
|
||||
- renders the conditional that computes the Helm value and confirms it
|
||||
produces `true` when cilium_envoy is enabled and `false` when disabled,
|
||||
- confirms the task stays forward/backward compatible (no raw `true` /
|
||||
`false` hardcoded in place of the conditional).
|
||||
"""
|
||||
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
|
||||
import yaml
|
||||
from jinja2 import Environment
|
||||
|
||||
ENVOY_EXPRESSION = '{{ "true" if cilium_envoy else "false" }}'
|
||||
|
||||
|
||||
def repo_root():
|
||||
return subprocess.check_output(
|
||||
["git", "rev-parse", "--show-toplevel"], text=True
|
||||
).strip()
|
||||
|
||||
|
||||
def fail(message):
|
||||
raise SystemExit("Cilium Envoy toggle test failed: " + message)
|
||||
|
||||
|
||||
def extract_install_command(path):
|
||||
"""Return the command string for the 'Install Cilium' task.
|
||||
|
||||
Walks both top-level tasks and tasks nested inside a `block`/`always`/
|
||||
`rescue` list, since the Cilium deploy steps are grouped under the
|
||||
'Prepare Cilium CLI on first master and deploy CNI' block.
|
||||
"""
|
||||
with open(path, encoding="utf-8") as handle:
|
||||
doc = yaml.safe_load(handle)
|
||||
|
||||
def find_command(tasks):
|
||||
for task in tasks:
|
||||
if not isinstance(task, dict):
|
||||
continue
|
||||
if task.get("name") == "Install Cilium":
|
||||
command = task.get("ansible.builtin.command")
|
||||
if command is None:
|
||||
raise SystemExit(
|
||||
"Cilium Envoy toggle test failed: "
|
||||
"'Install Cilium' task has no ansible.builtin.command"
|
||||
)
|
||||
return command
|
||||
# Recurse into block/always/rescue sub-lists.
|
||||
for key in ("block", "always", "rescue"):
|
||||
nested = task.get(key)
|
||||
if isinstance(nested, list):
|
||||
found = find_command(nested)
|
||||
if found is not None:
|
||||
return found
|
||||
return None
|
||||
|
||||
command = find_command(doc)
|
||||
if command is None:
|
||||
raise SystemExit(
|
||||
"Cilium Envoy toggle test failed: could not find 'Install Cilium' task"
|
||||
)
|
||||
return command
|
||||
|
||||
|
||||
def assert_envoy_in_command(command):
|
||||
if "envoy.enabled" not in command:
|
||||
fail("install command is missing --helm-set envoy.enabled")
|
||||
if ENVOY_EXPRESSION not in command:
|
||||
fail(
|
||||
"install command does not use the cilium_envoy conditional: "
|
||||
"expected {0!r}".format(ENVOY_EXPRESSION)
|
||||
)
|
||||
# The conditional must be a WYSIWYG helm-set value, not a pre-rendered
|
||||
# true/false literal (which would ignore the cilium_envoy variable).
|
||||
if re.search(r"--helm-set envoy\.enabled=true(?:$|\s)", command):
|
||||
fail("install command hardcodes envoy.enabled=true")
|
||||
if re.search(r"--helm-set envoy\.enabled=false(?:$|\s)", command):
|
||||
fail("install command hardcodes envoy.enabled=false")
|
||||
|
||||
|
||||
def assert_render():
|
||||
env = Environment()
|
||||
|
||||
def render_for(value):
|
||||
template = env.from_string(ENVOY_EXPRESSION)
|
||||
return template.render(cilium_envoy=value)
|
||||
|
||||
if render_for(True) != "true":
|
||||
fail("envoy conditional did not render 'true' when enabled")
|
||||
if render_for(False) != "false":
|
||||
fail("envoy conditional did not render 'false' when disabled")
|
||||
|
||||
|
||||
def main():
|
||||
root = repo_root()
|
||||
cilium_tasks = os.path.join(
|
||||
root, "roles", "k3s_server_post", "tasks", "cilium.yml"
|
||||
)
|
||||
command = extract_install_command(cilium_tasks)
|
||||
assert_envoy_in_command(command)
|
||||
assert_render()
|
||||
|
||||
print("Cilium Envoy toggle regression test passed")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
+129
@@ -0,0 +1,129 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
# shellcheck disable=SC2016
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
repo_root="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/../.." && pwd)"
|
||||
test_root="$(mktemp -d "${TMPDIR:-/tmp}/k3s-ci-cleanup-test.XXXXXX")"
|
||||
trap 'rm -rf -- "$test_root"' EXIT
|
||||
|
||||
molecule_root="$test_root/molecule"
|
||||
virtualbox_root="$test_root/VirtualBox VMs"
|
||||
fake_bin="$test_root/bin"
|
||||
mkdir -p -- "$molecule_root/k3s-ansible/single_node/.vagrant/machines/control1/virtualbox" \
|
||||
"$molecule_root/k3s-ansible/single_node/.vagrant/machines/control2/virtualbox" \
|
||||
"$virtualbox_root/control1" "$virtualbox_root/unmarked" "$fake_bin"
|
||||
printf '%s\n' '11111111-1111-1111-1111-111111111111' \
|
||||
> "$molecule_root/k3s-ansible/single_node/.vagrant/machines/control1/virtualbox/id"
|
||||
printf '%s\n' '22222222-2222-2222-2222-222222222222' \
|
||||
> "$molecule_root/k3s-ansible/single_node/.vagrant/machines/control2/virtualbox/id"
|
||||
touch "$virtualbox_root/control1/control1.vbox" "$virtualbox_root/control1/disk.vdi" \
|
||||
"$virtualbox_root/unmarked/unmarked.vbox"
|
||||
|
||||
printf '%s\n' \
|
||||
'#!/usr/bin/env bash' \
|
||||
'set -Eeuo pipefail' \
|
||||
'case "${1:-}" in' \
|
||||
' list)' \
|
||||
' if [[ "${2:-}" == hostonlyifs && "${FAKE_HOSTONLY_FAIL:-false}" == true ]]; then exit 1; fi' \
|
||||
' exit 0' \
|
||||
' ;;' \
|
||||
' showvminfo)' \
|
||||
' if [[ "${FAKE_VM_MODE:-normal}" == missing ]]; then' \
|
||||
' printf '\''VBoxManage: error: Could not find a registered machine named "missing"\n'\'' >&2' \
|
||||
' exit 1' \
|
||||
' fi' \
|
||||
' if [[ "${FAKE_VM_MODE:-normal}" == fault ]]; then' \
|
||||
' printf '\''VBoxManage: error: VirtualBox service is unavailable\n'\'' >&2' \
|
||||
' exit 1' \
|
||||
' fi' \
|
||||
' printf '\''CfgFile="%s"\n'\'' "${FAKE_VBOX_ROOT}/control1/control1.vbox"' \
|
||||
' printf '\''SATA-0-0="%s"\n'\'' "${FAKE_VBOX_ROOT}/control1/disk.vdi"' \
|
||||
' printf '\''VMState="running"\n'\''' \
|
||||
' ;;' \
|
||||
' controlvm) printf '\''controlvm %s\n'\'' "$*" >> "${FAKE_LOG}" ;;' \
|
||||
' unregistervm)' \
|
||||
' printf '\''unregistervm %s\n'\'' "$*" >> "${FAKE_LOG}"' \
|
||||
' rm -f -- "${FAKE_VBOX_ROOT}/control1/control1.vbox" "${FAKE_VBOX_ROOT}/control1/disk.vdi"' \
|
||||
' ;;' \
|
||||
' *) : ;;' \
|
||||
'esac' > "$fake_bin/VBoxManage"
|
||||
chmod 700 "$fake_bin/VBoxManage"
|
||||
|
||||
output="$test_root/output.txt"
|
||||
if PATH="$fake_bin:$PATH" \
|
||||
HOME="$test_root/home" \
|
||||
K3S_CI_MOLECULE_ROOT="$molecule_root" \
|
||||
K3S_CI_MOLECULE_PROJECT=k3s-ansible \
|
||||
K3S_CI_VIRTUALBOX_ROOT="$virtualbox_root" \
|
||||
K3S_CI_HOSTONLY_MARKER="$test_root/hostonly-baseline" \
|
||||
FAKE_VM_MODE=fault \
|
||||
FAKE_VBOX_ROOT="$virtualbox_root" \
|
||||
FAKE_LOG="$test_root/vbox.log" \
|
||||
bash "$repo_root/.github/scripts/cleanup-runner-resources.sh" --apply > "$output" 2>&1; then
|
||||
printf '%s\n' 'cleanup unexpectedly accepted a VirtualBox inspection failure' >&2
|
||||
exit 1
|
||||
fi
|
||||
grep -Fq 'cleanup refused: unable to inspect VirtualBox VM' "$output"
|
||||
[[ -f "$molecule_root/k3s-ansible/single_node/.vagrant/machines/control1/virtualbox/id" ]]
|
||||
|
||||
printf '%s\n' 'vboxnet0|192.168.30.1' > "$test_root/hostonly-baseline"
|
||||
if PATH="$fake_bin:$PATH" \
|
||||
HOME="$test_root/home" \
|
||||
K3S_CI_MOLECULE_ROOT="$molecule_root" \
|
||||
K3S_CI_MOLECULE_PROJECT=k3s-ansible \
|
||||
K3S_CI_VIRTUALBOX_ROOT="$virtualbox_root" \
|
||||
K3S_CI_HOSTONLY_MARKER="$test_root/hostonly-baseline" \
|
||||
FAKE_HOSTONLY_FAIL=true \
|
||||
FAKE_VBOX_ROOT="$virtualbox_root" \
|
||||
FAKE_LOG="$test_root/vbox.log" \
|
||||
bash "$repo_root/.github/scripts/cleanup-runner-resources.sh" --dry-run > "$output" 2>&1; then
|
||||
printf '%s\n' 'cleanup unexpectedly accepted a host-only inventory failure' >&2
|
||||
exit 1
|
||||
fi
|
||||
grep -Fq 'cleanup refused: unable to inventory VirtualBox host-only interfaces' "$output"
|
||||
|
||||
PATH="$fake_bin:$PATH" \
|
||||
HOME="$test_root/home" \
|
||||
K3S_CI_MOLECULE_ROOT="$molecule_root" \
|
||||
K3S_CI_MOLECULE_PROJECT=k3s-ansible \
|
||||
K3S_CI_VIRTUALBOX_ROOT="$virtualbox_root" \
|
||||
K3S_CI_HOSTONLY_MARKER="$test_root/hostonly-baseline" \
|
||||
FAKE_VBOX_ROOT="$virtualbox_root" \
|
||||
FAKE_LOG="$test_root/vbox.log" \
|
||||
bash "$repo_root/.github/scripts/cleanup-runner-resources.sh" --dry-run > "$output"
|
||||
|
||||
grep -Fq 'Would remove VM control1 (11111111-1111-1111-1111-111111111111)' "$output"
|
||||
[[ ! -e "$test_root/vbox.log" ]]
|
||||
|
||||
PATH="$fake_bin:$PATH" \
|
||||
HOME="$test_root/home" \
|
||||
K3S_CI_MOLECULE_ROOT="$molecule_root" \
|
||||
K3S_CI_MOLECULE_PROJECT=k3s-ansible \
|
||||
K3S_CI_VIRTUALBOX_ROOT="$virtualbox_root" \
|
||||
K3S_CI_HOSTONLY_MARKER="$test_root/hostonly-baseline" \
|
||||
FAKE_VBOX_ROOT="$virtualbox_root" \
|
||||
FAKE_LOG="$test_root/vbox.log" \
|
||||
bash "$repo_root/.github/scripts/cleanup-runner-resources.sh" --apply > "$output"
|
||||
|
||||
grep -Fq 'controlvm 11111111-1111-1111-1111-111111111111 poweroff' "$test_root/vbox.log"
|
||||
grep -Fq 'unregistervm unregistervm 11111111-1111-1111-1111-111111111111 --delete' "$test_root/vbox.log"
|
||||
[[ ! -e "$virtualbox_root/control1/control1.vbox" ]]
|
||||
[[ -e "$virtualbox_root/unmarked/unmarked.vbox" ]]
|
||||
|
||||
PATH="$fake_bin:$PATH" \
|
||||
HOME="$test_root/home" \
|
||||
K3S_CI_MOLECULE_ROOT="$molecule_root" \
|
||||
K3S_CI_MOLECULE_PROJECT=k3s-ansible \
|
||||
K3S_CI_VIRTUALBOX_ROOT="$virtualbox_root" \
|
||||
K3S_CI_HOSTONLY_MARKER="$test_root/hostonly-baseline" \
|
||||
FAKE_VM_MODE=missing \
|
||||
FAKE_VBOX_ROOT="$virtualbox_root" \
|
||||
FAKE_LOG="$test_root/vbox.log" \
|
||||
bash "$repo_root/.github/scripts/cleanup-runner-resources.sh" --apply > "$output"
|
||||
|
||||
grep -Fq 'Stale Vagrant state without a registered VM' "$output"
|
||||
[[ ! -d "$molecule_root/k3s-ansible/single_node/.vagrant" ]]
|
||||
|
||||
printf 'cleanup-runner-resources fixture test passed\n'
|
||||
@@ -0,0 +1,90 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Assert the sample inventory resolves the flannel interface per host.
|
||||
|
||||
flannel_iface defaults to the host's default IPv4 interface rather than a
|
||||
hardcoded eth0. This test extracts the flannel_iface expression from the
|
||||
sample inventory and proves that a host whose primary interface is not named
|
||||
eth0 (e.g. enp1s0, ens3) resolves the interface from ansible facts.
|
||||
"""
|
||||
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
|
||||
from jinja2 import Environment, StrictUndefined
|
||||
|
||||
|
||||
def repo_root():
|
||||
return subprocess.check_output(
|
||||
["git", "rev-parse", "--show-toplevel"], text=True
|
||||
).strip()
|
||||
|
||||
|
||||
def fail(message):
|
||||
raise SystemExit("default-interface test failed: " + message)
|
||||
|
||||
|
||||
class FakeAnsibleFacts(object):
|
||||
"""Stand-in for the per-host ``ansible_facts`` dict."""
|
||||
|
||||
def __init__(self, default_iface, iface_ip):
|
||||
self._default = {"interface": default_iface, "address": iface_ip}
|
||||
self._ifaces = {
|
||||
default_iface: {"ipv4": {"address": iface_ip}},
|
||||
}
|
||||
|
||||
@property
|
||||
def default_ipv4(self):
|
||||
return self._default
|
||||
|
||||
def __getitem__(self, key):
|
||||
return self._ifaces[key]
|
||||
|
||||
|
||||
def read_all_yml(root):
|
||||
path = os.path.join(root, "inventory", "sample", "group_vars", "all.yml")
|
||||
with open(path, "r") as handle:
|
||||
return handle.read()
|
||||
|
||||
|
||||
def extract_value(content, key):
|
||||
# Match a quoted value assigned to the key, e.g. flannel_iface: "...".
|
||||
match = re.search(r"^%s:\s*\"(.+)\"\s*$" % re.escape(key), content, re.M)
|
||||
if not match:
|
||||
fail("could not find %s in the sample inventory" % key)
|
||||
return match.group(1)
|
||||
|
||||
|
||||
def resolve(env, expression, facts):
|
||||
template = env.from_string(expression)
|
||||
return template.render(ansible_facts=facts)
|
||||
|
||||
|
||||
def main():
|
||||
root = repo_root()
|
||||
content = read_all_yml(root)
|
||||
env = Environment(undefined=StrictUndefined)
|
||||
|
||||
flannel_expr = extract_value(content, "flannel_iface")
|
||||
if "default_ipv4.interface" not in flannel_expr:
|
||||
fail("flannel_iface no longer defaults from ansible facts")
|
||||
|
||||
# A host whose primary interface is enp1s0 (the core #621 scenario).
|
||||
facts = FakeAnsibleFacts("enp1s0", "192.168.30.11")
|
||||
resolved = resolve(env, flannel_expr, facts)
|
||||
if resolved != "enp1s0":
|
||||
fail("flannel_iface resolved to %r, expected enp1s0" % resolved)
|
||||
|
||||
# A different host with a different interface must resolve independently.
|
||||
facts2 = FakeAnsibleFacts("ens3", "192.168.30.12")
|
||||
resolved2 = resolve(env, flannel_expr, facts2)
|
||||
if resolved2 != "ens3":
|
||||
fail("flannel_iface resolved to %r, expected ens3" % resolved2)
|
||||
|
||||
print("default-interface regression test passed")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Executable
+37
@@ -0,0 +1,37 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
repo_root="$(git rev-parse --show-toplevel)"
|
||||
prereq_defaults="$repo_root/roles/prereq/defaults/main.yml"
|
||||
prereq_tasks="$repo_root/roles/prereq/tasks/main.yml"
|
||||
|
||||
# #670: k3s recommends swap be disabled on all nodes. The prereq role must expose
|
||||
# a disable_swap toggle (defaulting to true) that turns swap off now and comments
|
||||
# out the /etc/fstab swap entries so swap stays off across reboots.
|
||||
grep -Eq -- '^disable_swap: true' "$prereq_defaults" || {
|
||||
printf 'prereq defaults are missing disable_swap: true\n' >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
grep -Fq -- 'Disable swap on all cluster nodes' "$prereq_tasks" || {
|
||||
printf 'prereq tasks are missing the swap-disable block\n' >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
grep -Fq -- 'swapoff -a' "$prereq_tasks" || {
|
||||
printf 'swap-disable block does not run swapoff -a\n' >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
grep -Fq -- '/etc/fstab' "$prereq_tasks" || {
|
||||
printf 'swap-disable block does not comment out /etc/fstab swap entries\n' >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
if ! grep -Eq -- 'when: disable_swap' "$prereq_tasks"; then
|
||||
printf 'swap-disable block is not gated on the disable_swap toggle\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
printf 'Swap disable regression test passed\n'
|
||||
Executable
+4
@@ -0,0 +1,4 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
[[ "${1:-}" =~ ^[0-9]+$ ]]
|
||||
Executable
+30
@@ -0,0 +1,30 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
printf '%s\n' "$*" >> "$MOCK_VAGRANT_LOG"
|
||||
case "${1:-}" in
|
||||
up)
|
||||
if [[ -n "${MOCK_VAGRANTFILE_CAPTURE:-}" ]]; then
|
||||
cp -- "$VAGRANT_CWD/Vagrantfile" "$MOCK_VAGRANTFILE_CAPTURE"
|
||||
fi
|
||||
counter_file="$MOCK_VBOX_STATE/counter"
|
||||
counter=0
|
||||
[[ ! -r "$counter_file" ]] || counter="$(cat "$counter_file")"
|
||||
counter=$((counter + 1))
|
||||
printf '%s\n' "$counter" > "$counter_file"
|
||||
uuid="00000000-0000-4000-8000-$(printf '%012d' "$counter")"
|
||||
vm_dir="$MOCK_VBOX_ROOT/master-$counter"
|
||||
mkdir -p -- "$vm_dir"
|
||||
: > "$vm_dir/master.vbox"
|
||||
: > "$vm_dir/master.vdi"
|
||||
printf '%s\n' \
|
||||
'VMState="poweroff"' \
|
||||
'groups="/"' \
|
||||
"CfgFile=\"$vm_dir/master.vbox\"" \
|
||||
"SATA-0-0=\"$vm_dir/master.vdi\"" > "$MOCK_VBOX_STATE/vm-$uuid"
|
||||
printf '%s\n' "$uuid" > "$MOCK_BOX_DIR/master_id"
|
||||
;;
|
||||
destroy) ;;
|
||||
*) exit 2 ;;
|
||||
esac
|
||||
+40
@@ -0,0 +1,40 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
printf '%s\n' "$*" >> "$MOCK_VBOX_LOG"
|
||||
command_name="${1:-}"
|
||||
uuid="${2:-}"
|
||||
|
||||
case "$command_name" in
|
||||
showvminfo)
|
||||
[[ -r "$MOCK_VBOX_STATE/vm-$uuid" ]] || exit 1
|
||||
cat "$MOCK_VBOX_STATE/vm-$uuid"
|
||||
;;
|
||||
getextradata)
|
||||
key_slug="${3//\//_}"
|
||||
if [[ ! -r "$MOCK_VBOX_STATE/extra-$uuid-$key_slug" ]]; then
|
||||
printf '%s\n' 'No value set!'
|
||||
exit 0
|
||||
fi
|
||||
printf 'Value: '
|
||||
cat "$MOCK_VBOX_STATE/extra-$uuid-$key_slug"
|
||||
;;
|
||||
modifyvm)
|
||||
[[ "${3:-}" == --groups ]]
|
||||
awk -v groups="${4:-}" '
|
||||
$1 !~ /^groups=/ { print }
|
||||
END { printf "groups=\"%s\"\n", groups }
|
||||
' "$MOCK_VBOX_STATE/vm-$uuid" > "$MOCK_VBOX_STATE/vm-$uuid.tmp"
|
||||
mv "$MOCK_VBOX_STATE/vm-$uuid.tmp" "$MOCK_VBOX_STATE/vm-$uuid"
|
||||
;;
|
||||
setextradata)
|
||||
key_slug="${3//\//_}"
|
||||
printf '%s\n' "${4:-}" > "$MOCK_VBOX_STATE/extra-$uuid-$key_slug"
|
||||
;;
|
||||
unregistervm|closemedium)
|
||||
printf '%s\n' 'destructive VirtualBox command invoked' >&2
|
||||
exit 99
|
||||
;;
|
||||
*) exit 2 ;;
|
||||
esac
|
||||
Executable
+35
@@ -0,0 +1,35 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
repo_root="$(git rev-parse --show-toplevel)"
|
||||
main_tasks="$repo_root/roles/k3s_server/tasks/main.yml"
|
||||
join_tasks="$repo_root/roles/k3s_server/tasks/join_master.yml"
|
||||
|
||||
for task_file in "$main_tasks" "$join_tasks"; do
|
||||
for property in \
|
||||
'Delegate=yes' \
|
||||
'TasksMax=infinity' \
|
||||
'KillMode=process' \
|
||||
'LimitNOFILE=1048576' \
|
||||
'LimitNPROC=infinity' \
|
||||
'LimitCORE=infinity'; do
|
||||
grep -Fq -- "$property" "$task_file" || {
|
||||
printf '%s is missing transient K3s property %s\n' "$task_file" "$property" >&2
|
||||
exit 1
|
||||
}
|
||||
done
|
||||
done
|
||||
|
||||
if grep -Fq -- "node-role.kubernetes.io/master=true' -o=jsonpath" "$main_tasks"; then
|
||||
printf 'control-plane registration still depends on the optional legacy master role key\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
grep -Fq -- "map('extract', hostvars, 'ansible_hostname')" "$main_tasks"
|
||||
grep -Fq -- 'difference(nodes.stdout.split())' "$main_tasks"
|
||||
grep -Fq -- 'crd/addons.k3s.cattle.io' "$main_tasks"
|
||||
grep -Fq -- 'crd/helmcharts.helm.cattle.io' "$main_tasks"
|
||||
grep -Fq -- 'crd/helmchartconfigs.helm.cattle.io' "$main_tasks"
|
||||
|
||||
printf 'K3s transient bootstrap regression test passed\n'
|
||||
@@ -0,0 +1,162 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Render the kube-vip DaemonSet template and assert env key correctness.
|
||||
|
||||
kube-vip v1.2.2 reads `bgp_peers` and `vip_subnet`; it ignores the older
|
||||
`bgppeers` and `vip_cidr` names. This test proves the rendered manifest uses
|
||||
the keys the target image actually parses.
|
||||
"""
|
||||
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
from jinja2 import Environment, FileSystemLoader, StrictUndefined
|
||||
|
||||
|
||||
def repo_root():
|
||||
return subprocess.check_output(
|
||||
["git", "rev-parse", "--show-toplevel"], text=True
|
||||
).strip()
|
||||
|
||||
|
||||
def fail(message):
|
||||
raise SystemExit("kube-vip manifest test failed: " + message)
|
||||
|
||||
|
||||
def fake_ipsubnet(value):
|
||||
# ansible.utils.ipsubnet -> network of the address as x.y.z.0/24
|
||||
parts = value.split(".")
|
||||
return ".".join(parts[:3]) + ".0/24"
|
||||
|
||||
|
||||
def fake_ipaddr(_value, expr=None):
|
||||
# ansible.utils.ipaddr('prefix') -> prefix length
|
||||
return "24"
|
||||
|
||||
|
||||
def fake_bool(value):
|
||||
# Minimal stand-in for Ansible's truthiness filter used by the template.
|
||||
if isinstance(value, bool):
|
||||
return value
|
||||
return str(value).lower() in ("1", "true", "yes", "on")
|
||||
|
||||
|
||||
def fake_map(seq, *args, **kwargs):
|
||||
# Minimal stand-in for Ansible's map() filter in the two forms used by the
|
||||
# template: map(attribute='x') on a list of dicts, and map('join', sep) on
|
||||
# a list of sequences.
|
||||
if "attribute" in kwargs:
|
||||
return [item[kwargs["attribute"]] for item in seq]
|
||||
if kwargs:
|
||||
# e.g. map(default='x') not used here; ignore unknown kwargs.
|
||||
return list(seq)
|
||||
if args:
|
||||
filter_name = args[0]
|
||||
sep = args[1] if len(args) > 1 else ""
|
||||
if filter_name == "join":
|
||||
return [sep.join(str(x) for x in item) for item in seq]
|
||||
return list(seq)
|
||||
|
||||
|
||||
def fake_zip(*seqs):
|
||||
return list(zip(*seqs))
|
||||
|
||||
|
||||
def render(env, extra_vars):
|
||||
base_vars = {
|
||||
"apiserver_endpoint": "192.168.30.222",
|
||||
"kube_vip_iface": "",
|
||||
"kube_vip_arp": True,
|
||||
"kube_vip_bgp": True,
|
||||
"kube_vip_bgp_routerid": "127.0.0.1",
|
||||
"_kube_vip_bgp_peers": [
|
||||
{"peer_address": "192.168.30.1", "peer_asn": "64512"},
|
||||
{"peer_address": "192.168.30.2", "peer_asn": "64513"},
|
||||
],
|
||||
"kube_vip_tag_version": "v1.2.2",
|
||||
}
|
||||
base_vars.update(extra_vars)
|
||||
template = env.get_template("vip.yaml.j2")
|
||||
return template.render(**base_vars)
|
||||
|
||||
|
||||
def main():
|
||||
root = repo_root()
|
||||
template_dir = os.path.join(root, "roles", "k3s_server", "templates")
|
||||
env = Environment(
|
||||
loader=FileSystemLoader(template_dir), undefined=StrictUndefined
|
||||
)
|
||||
env.filters["ansible.utils.ipsubnet"] = fake_ipsubnet
|
||||
env.filters["ansible.utils.ipaddr"] = fake_ipaddr
|
||||
env.filters["bool"] = fake_bool
|
||||
env.filters["map"] = fake_map
|
||||
env.filters["zip"] = fake_zip
|
||||
|
||||
# Multi-peer BGP armed: must emit bgp_peers, never bgppeers.
|
||||
output = render(env, {})
|
||||
if "name: bgp_peers" not in output:
|
||||
fail("rendered manifest is missing bgp_peers")
|
||||
if "name: bgppeers" in output:
|
||||
fail("rendered manifest still uses the ignored bgppeers key")
|
||||
if "name: vip_subnet" not in output:
|
||||
fail("rendered manifest is missing vip_subnet")
|
||||
if "name: vip_cidr" in output:
|
||||
fail("rendered manifest still uses the ignored vip_cidr key")
|
||||
if "192.168.30.1:64512,192.168.30.2:64513" not in output:
|
||||
fail("bgp_peers value is not comma-separated address:ASN entries")
|
||||
if "ghcr.io/kube-vip/kube-vip:v1.2.2" not in output:
|
||||
fail("kube-vip image tag is not v1.2.2")
|
||||
|
||||
# BGP enabled with no merged peers: single-peer fallback vars, no bgp_peers.
|
||||
output = render(
|
||||
env,
|
||||
{
|
||||
"_kube_vip_bgp_peers": [],
|
||||
"kube_vip_bgp_as": "64513",
|
||||
"kube_vip_bgp_peeraddress": "192.168.30.1",
|
||||
"kube_vip_bgp_peeras": "64512",
|
||||
},
|
||||
)
|
||||
if "name: bgp_as" not in output:
|
||||
fail("single-peer bgp_as was not rendered")
|
||||
if "name: bgp_peers" in output:
|
||||
fail("bgp_peers present even though the peer list is empty")
|
||||
|
||||
# kube_vip_endpoint defaults to null (defined in role defaults): the
|
||||
# address and subnet must fall back to the apiserver endpoint. default()
|
||||
# without a truthy flag does NOT fall back on null, only on undefined, so
|
||||
# this case pins the null runtime condition to prevent that regression.
|
||||
output = render(
|
||||
env,
|
||||
{
|
||||
"_kube_vip_bgp_peers": [],
|
||||
"kube_vip_endpoint": None,
|
||||
"kube_vip_arp": True,
|
||||
"kube_vip_bgp": False,
|
||||
},
|
||||
)
|
||||
if "value: 192.168.30.222" not in output:
|
||||
fail("null kube_vip_endpoint does not fall back to apiserver_endpoint")
|
||||
|
||||
# kube_vip_endpoint set: overrides the internal listening address AND the
|
||||
# subnet derivation while the advertised apiserver_endpoint stays separate.
|
||||
output = render(
|
||||
env,
|
||||
{
|
||||
"_kube_vip_bgp_peers": [],
|
||||
"kube_vip_endpoint": "10.66.1.5",
|
||||
"kube_vip_arp": True,
|
||||
"kube_vip_bgp": False,
|
||||
},
|
||||
)
|
||||
if "value: 10.66.1.5" not in output:
|
||||
fail("kube_vip_endpoint did not override the address")
|
||||
if "value: 192.168.30.222" in output:
|
||||
fail("apiserver_endpoint leaked into address when kube_vip_endpoint set")
|
||||
|
||||
print("kube-vip manifest regression test passed")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,170 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Regression test for the MetalLB deploy conditions.
|
||||
|
||||
The MetalLB manifest (roles/k3s_server/tasks/main.yml) and the MetalLB pool
|
||||
(roles/k3s_server_post/tasks/main.yml) are included under a `when` condition
|
||||
that decides whether MetalLB provides load balancing. A previous change (#683)
|
||||
guarded `cilium_bgp` but accidentally skipped MetalLB whenever a non-BGP
|
||||
Cilium CNI was in use (`cilium_iface` defined), breaking the cilium + MetalLB
|
||||
scenario.
|
||||
|
||||
This test loads the real `when` expressions from both task files and evaluates
|
||||
them against representative variable sets, asserting MetalLB is deployed in
|
||||
every topology except when kube-vip owns the VIP range or Cilium BGP is enabled.
|
||||
"""
|
||||
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
|
||||
import yaml
|
||||
from jinja2 import Environment
|
||||
|
||||
METALLB_WHEN = (
|
||||
"kube_vip_lb_ip_range is not defined and "
|
||||
"not (cilium_bgp | default(false) | bool)"
|
||||
)
|
||||
|
||||
|
||||
def repo_root():
|
||||
return subprocess.check_output(
|
||||
["git", "rev-parse", "--show-toplevel"], text=True
|
||||
).strip()
|
||||
|
||||
|
||||
def fail(message):
|
||||
raise SystemExit("MetalLB deploy condition test failed: " + message)
|
||||
|
||||
|
||||
def extract_when(path, task_name):
|
||||
"""Return the `when:` expression string for the named task."""
|
||||
with open(path, encoding="utf-8") as handle:
|
||||
doc = yaml.safe_load(handle)
|
||||
for task in doc:
|
||||
if task.get("name") == task_name:
|
||||
when = task.get("when")
|
||||
return (when or "").strip()
|
||||
return None
|
||||
|
||||
|
||||
def evaluate(when, variables):
|
||||
"""Evaluate a `when` expression against variables using Jinja2."""
|
||||
env = Environment()
|
||||
|
||||
def fake_bool(value):
|
||||
# Minimal stand-in for Ansible's truthiness filter used by `| bool`.
|
||||
if isinstance(value, bool):
|
||||
return value
|
||||
if value is None:
|
||||
return False
|
||||
return str(value).lower() in ("1", "true", "yes", "on")
|
||||
|
||||
env.filters["bool"] = fake_bool
|
||||
template = env.from_string("{{ " + when + " }}")
|
||||
rendered = template.render(**variables)
|
||||
# The expression renders to the literal strings "True"/"False".
|
||||
if rendered == "True":
|
||||
return True
|
||||
if rendered == "False":
|
||||
return False
|
||||
fail("condition did not render to a boolean: {0!r}".format(rendered))
|
||||
|
||||
|
||||
def assert_deployment(when, variables, expected, label):
|
||||
result = evaluate(when, variables)
|
||||
verdict = "deploy" if result else "skip"
|
||||
expected_verdict = "deploy" if expected else "skip"
|
||||
if result != expected:
|
||||
fail(
|
||||
"{0}: expected to {1} MetalLB but the condition chose to {2} "
|
||||
"(vars: {3})".format(label, expected_verdict, verdict, variables)
|
||||
)
|
||||
|
||||
|
||||
def scenarios():
|
||||
"""Yield (variables, expected_deploy, label) pairs."""
|
||||
yield (
|
||||
# Default Flannel inventory (all.yml sets cilium_bgp: false).
|
||||
{
|
||||
"cilium_bgp": False,
|
||||
"cilium_iface": None,
|
||||
},
|
||||
True,
|
||||
"flannel default (cilium_bgp: false)",
|
||||
)
|
||||
yield (
|
||||
# Calico CNI with no Cilium variable in scope (issue #644): cilium_bgp
|
||||
# is genuinely undefined, so `default(false)` must keep MetalLB on.
|
||||
{
|
||||
"calico_iface": "eth1",
|
||||
},
|
||||
True,
|
||||
"calico, cilium_bgp undefined (#644)",
|
||||
)
|
||||
yield (
|
||||
# Cilium CNI with BGP disabled: MetalLB must still be deployed.
|
||||
{
|
||||
"cilium_bgp": False,
|
||||
"cilium_iface": "eth1",
|
||||
},
|
||||
True,
|
||||
"cilium non-BGP (regression catch)",
|
||||
)
|
||||
yield (
|
||||
# Cilium CNI with BGP enabled: Cilium provides the LB, skip MetalLB.
|
||||
{
|
||||
"cilium_bgp": True,
|
||||
"cilium_iface": "eth1",
|
||||
},
|
||||
False,
|
||||
"cilium BGP enabled",
|
||||
)
|
||||
yield (
|
||||
# kube-vip is the load balancer provider: skip MetalLB.
|
||||
{
|
||||
"kube_vip_lb_ip_range": "192.168.30.80-192.168.30.90",
|
||||
"cilium_bgp": False,
|
||||
},
|
||||
False,
|
||||
"kube-vip owns the VIP range",
|
||||
)
|
||||
|
||||
|
||||
def main():
|
||||
root = repo_root()
|
||||
server_tasks = os.path.join(root, "roles", "k3s_server", "tasks", "main.yml")
|
||||
server_post_tasks = os.path.join(
|
||||
root, "roles", "k3s_server_post", "tasks", "main.yml"
|
||||
)
|
||||
|
||||
server_when = extract_when(server_tasks, "Deploy metallb manifest")
|
||||
server_post_when = extract_when(server_post_tasks, "Deploy metallb pool")
|
||||
|
||||
if server_when is None:
|
||||
fail("could not find 'Deploy metallb manifest' when condition")
|
||||
if server_post_when is None:
|
||||
fail("could not find 'Deploy metallb pool' when condition")
|
||||
|
||||
for when, source in (
|
||||
(server_when, "k3s_server/tasks/main.yml"),
|
||||
(server_post_when, "k3s_server_post/tasks/main.yml"),
|
||||
):
|
||||
if when != METALLB_WHEN:
|
||||
fail(
|
||||
"{0} when condition changed unexpectedly:\n"
|
||||
" expected: {1}\n got: {2}".format(source, METALLB_WHEN, when)
|
||||
)
|
||||
|
||||
for variables, expected, label in scenarios():
|
||||
assert_deployment(server_when, variables, expected, "server " + label)
|
||||
assert_deployment(
|
||||
server_post_when, variables, expected, "server_post " + label
|
||||
)
|
||||
|
||||
print("MetalLB deploy condition regression test passed for all scenarios")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,88 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Regression test for the MetalLB L2Advertisement interfaces.
|
||||
|
||||
`metal_lb_interfaces` restricts which network interfaces MetalLB announces
|
||||
load balancer IPs on in layer2 mode. When the list is non-empty, the
|
||||
L2Advertisement in roles/k3s_server_post/templates/metallb.crs.j2 must render
|
||||
a `spec.interfaces` block; when it is empty (the default), no spec is rendered
|
||||
so MetalLB announces on all interfaces.
|
||||
|
||||
This renders the template and asserts both cases plus the BGP path (which must
|
||||
not be affected by the L2 interfaces variable).
|
||||
"""
|
||||
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
from jinja2 import Environment, FileSystemLoader, StrictUndefined
|
||||
|
||||
|
||||
def repo_root():
|
||||
return subprocess.check_output(
|
||||
["git", "rev-parse", "--show-toplevel"], text=True
|
||||
).strip()
|
||||
|
||||
|
||||
def fail(message):
|
||||
raise SystemExit("MetalLB interfaces test failed: " + message)
|
||||
|
||||
|
||||
def render(env, extra_vars):
|
||||
base_vars = {
|
||||
"metal_lb_mode": "layer2",
|
||||
"metal_lb_ip_range": "192.168.30.80-192.168.30.90",
|
||||
}
|
||||
base_vars.update(extra_vars)
|
||||
template = env.get_template("metallb.crs.j2")
|
||||
return template.render(**base_vars)
|
||||
|
||||
|
||||
def main():
|
||||
root = repo_root()
|
||||
template_dir = os.path.join(
|
||||
root, "roles", "k3s_server_post", "templates"
|
||||
)
|
||||
env = Environment(
|
||||
loader=FileSystemLoader(template_dir), undefined=StrictUndefined
|
||||
)
|
||||
|
||||
# Empty list (default): no spec.interfaces in the L2Advertisement.
|
||||
output = render(env, {"metal_lb_interfaces": []})
|
||||
if "spec:\n interfaces:" in output:
|
||||
fail("spec.interfaces rendered with an empty metal_lb_interfaces")
|
||||
if "kind: L2Advertisement" not in output:
|
||||
fail("L2Advertisement missing in layer2 mode")
|
||||
|
||||
# Single interface.
|
||||
output = render(env, {"metal_lb_interfaces": ["eth1"]})
|
||||
if "spec:\n interfaces:\n - eth1" not in output:
|
||||
fail("single interface was not rendered in spec.interfaces")
|
||||
|
||||
# Multiple interfaces.
|
||||
output = render(env, {"metal_lb_interfaces": ["eth1", "eth2"]})
|
||||
if "spec:\n interfaces:\n - eth1\n - eth2" not in output:
|
||||
fail("multiple interfaces were not rendered in spec.interfaces")
|
||||
|
||||
# BGP mode must not emit an L2Advertisement spec at all.
|
||||
output = render(
|
||||
env,
|
||||
{
|
||||
"metal_lb_mode": "bgp",
|
||||
"metal_lb_interfaces": ["eth1"],
|
||||
"metal_lb_bgp_my_asn": "64513",
|
||||
"metal_lb_bgp_peer_asn": "64512",
|
||||
"metal_lb_bgp_peer_address": "192.168.30.1",
|
||||
},
|
||||
)
|
||||
if "kind: L2Advertisement" in output:
|
||||
fail("L2Advertisement rendered in bgp mode")
|
||||
if "interfaces:" in output:
|
||||
fail("interfaces rendered in bgp mode")
|
||||
|
||||
print("MetalLB interfaces regression test passed")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,108 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Regression test for the MetalLB converge checks.
|
||||
|
||||
The MetalLB tasks in roles/k3s_server_post/tasks/metallb.yml must actually
|
||||
verify resources through an explicit kubectl get, and must retry on a
|
||||
transient kube API error while MetalLB converges.
|
||||
|
||||
The "Test metallb-system namespace" task previously ran `k3s kubectl -n
|
||||
metallb-system` with no subcommand, which only printed a usage page and always
|
||||
exited 0, so it always succeeded even when the namespace did not exist (issue
|
||||
#350). It must instead run an explicit `get namespace metallb-system`, which
|
||||
returns non-zero when the namespace is absent.
|
||||
|
||||
An explicit get actually contacts the API server, so these tasks need the same
|
||||
retry wiring as their siblings (register, until rc == 0, retries, delay). A
|
||||
bare get with no retry would otherwise abort the converge play on a transient
|
||||
kube API error while MetalLB converges.
|
||||
"""
|
||||
|
||||
from __future__ import print_function
|
||||
|
||||
import os
|
||||
import subprocess
|
||||
|
||||
import yaml
|
||||
|
||||
|
||||
def repo_root():
|
||||
return subprocess.check_output(
|
||||
["git", "rev-parse", "--show-toplevel"], text=True
|
||||
).strip()
|
||||
|
||||
|
||||
def fail(message):
|
||||
raise SystemExit("MetalLB namespace test failed: " + message)
|
||||
|
||||
|
||||
def find_task(tasks, name):
|
||||
for entry in tasks:
|
||||
if entry.get("name") == name:
|
||||
return entry
|
||||
fail("could not find the '{0}' task".format(name))
|
||||
return None
|
||||
|
||||
|
||||
def command_text(task):
|
||||
cmd = task.get("ansible.builtin.command")
|
||||
if not cmd:
|
||||
cmd = task.get("command")
|
||||
if not cmd:
|
||||
fail("task does not use ansible.builtin.command")
|
||||
return cmd if isinstance(cmd, str) else " ".join(cmd)
|
||||
|
||||
|
||||
def check_explicit_get(task, name, needle):
|
||||
text = command_text(task)
|
||||
if needle not in text:
|
||||
fail(
|
||||
"command does not run '{0}'; the task would only print usage and "
|
||||
"never verify the resource (got: {1!r})".format(needle, text)
|
||||
)
|
||||
|
||||
|
||||
def check_retry_wiring(task, name):
|
||||
# The sibling k3s_server_post metallb tasks retry kubectl because the kube
|
||||
# API can briefly be unavailable while MetalLB converges. Without the same
|
||||
# retry, a transient API error aborts the whole converge play.
|
||||
if not task.get("register"):
|
||||
fail(
|
||||
"{0} does not register a result; without retry wiring a transient "
|
||||
"kube API error aborts the converge play".format(name)
|
||||
)
|
||||
if not isinstance(task.get("until"), str) or "rc == 0" not in task["until"]:
|
||||
fail(
|
||||
"{0} does not retry on rc == 0; the kube API can transiently fail "
|
||||
"while MetalLB converges and abort the play".format(name)
|
||||
)
|
||||
if task.get("retries") is None:
|
||||
fail("{0} is missing retries".format(name))
|
||||
if task.get("delay") is None:
|
||||
fail("{0} is missing delay".format(name))
|
||||
|
||||
|
||||
def main():
|
||||
task_file = os.path.join(
|
||||
repo_root(), "roles", "k3s_server_post", "tasks", "metallb.yml"
|
||||
)
|
||||
with open(task_file, encoding="utf-8") as handle:
|
||||
tasks = yaml.safe_load(handle)
|
||||
|
||||
namespace_task = find_task(tasks, "Test metallb-system namespace")
|
||||
# A bare `-n metallb-system` with no subcommand prints kubectl usage and
|
||||
# always exits 0, so it never proves the namespace exists. The fix must
|
||||
# use an explicit get.
|
||||
check_explicit_get(namespace_task, "Test metallb-system namespace",
|
||||
"get namespace metallb-system")
|
||||
check_retry_wiring(namespace_task, "Test metallb-system namespace")
|
||||
|
||||
webhook_task = find_task(tasks, "Test metallb-system webhook-service endpoint")
|
||||
check_explicit_get(webhook_task, "Test metallb-system webhook-service endpoint",
|
||||
"get endpoints")
|
||||
check_retry_wiring(webhook_task, "Test metallb-system webhook-service endpoint")
|
||||
|
||||
print("MetalLB namespace check regression test passed")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Executable
+29
@@ -0,0 +1,29 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
repo_root="$(git rev-parse --show-toplevel)"
|
||||
metallb_task="$repo_root/roles/k3s_server/tasks/metallb.yml"
|
||||
|
||||
# The speaker tag verification must read the rendered manifest on the managed
|
||||
# host with slurp. A controller-side lookup('ansible.builtin.file', ...) would
|
||||
# read from the Ansible control node, which does not have the file, and would
|
||||
# fail on every MetalLB scenario.
|
||||
grep -Fq -- 'ansible.builtin.slurp' "$metallb_task" || {
|
||||
printf 'MetalLB speaker tag check does not use slurp on the managed host\n' >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
grep -Eq -- 'lookup\(.?ansible\.builtin\.file' "$metallb_task" && {
|
||||
printf 'MetalLB speaker tag check uses a controller-side file lookup\n' >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
# The check must reference the full image reference, not just a bare version
|
||||
# string that could appear anywhere in the manifest.
|
||||
grep -Fq -- 'quay.io/metallb/speaker:' "$metallb_task" || {
|
||||
printf 'MetalLB speaker tag check does not match the full image reference\n' >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
printf 'MetalLB remote manifest read regression test passed\n'
|
||||
+89
@@ -0,0 +1,89 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
repo_root="$(git rev-parse --show-toplevel)"
|
||||
fixture="$(mktemp -d)"
|
||||
trap 'rm -rf -- "$fixture"' EXIT
|
||||
|
||||
mock_bin="$fixture/bin"
|
||||
mock_state="$fixture/state"
|
||||
mock_home="$fixture/home"
|
||||
mock_vagrant_home="$fixture/vagrant-home"
|
||||
mock_box_dir="$mock_vagrant_home/boxes/bento-VAGRANTSLASH-ubuntu-26.04/202606.01.0/amd64/virtualbox"
|
||||
mock_vbox_root="$mock_home/VirtualBox VMs"
|
||||
mock_master_root="$mock_home/.cache/k3s-ci/vagrant-masters"
|
||||
lock_file="$fixture/vagrant-boxes.lock"
|
||||
mkdir -p -- "$mock_bin" "$mock_state" "$mock_box_dir" "$mock_vbox_root"
|
||||
printf '%s\n' 'bento/ubuntu-26.04 202606.01.0 amd64' > "$lock_file"
|
||||
ln -s "$repo_root/.github/scripts/test-fixtures/mock-vboxmanage" "$mock_bin/VBoxManage"
|
||||
ln -s "$repo_root/.github/scripts/test-fixtures/mock-vagrant" "$mock_bin/vagrant"
|
||||
ln -s "$repo_root/.github/scripts/test-fixtures/mock-flock" "$mock_bin/flock"
|
||||
|
||||
export PATH="$mock_bin:$PATH"
|
||||
export HOME="$mock_home"
|
||||
export VAGRANT_HOME="$mock_vagrant_home"
|
||||
export VAGRANT_BOX_LOCK_FILE="$lock_file"
|
||||
export K3S_CI_REPOSITORY_ROOT="$repo_root"
|
||||
export K3S_CI_VAGRANT_MASTER_ROOT="$mock_master_root"
|
||||
export K3S_CI_VIRTUALBOX_ROOT="$mock_vbox_root"
|
||||
export MOCK_VBOX_STATE="$mock_state"
|
||||
export MOCK_VBOX_ROOT="$mock_vbox_root"
|
||||
export MOCK_BOX_DIR="$mock_box_dir"
|
||||
export MOCK_VBOX_LOG="$fixture/vbox.log"
|
||||
export MOCK_VAGRANT_LOG="$fixture/vagrant.log"
|
||||
export MOCK_VAGRANTFILE_CAPTURE="$fixture/prewarm-Vagrantfile"
|
||||
: > "$MOCK_VBOX_LOG"
|
||||
: > "$MOCK_VAGRANT_LOG"
|
||||
|
||||
unowned_uuid='99999999-9999-4999-8999-999999999999'
|
||||
printf '%s\n' "$unowned_uuid" > "$mock_box_dir/master_id"
|
||||
|
||||
script="$repo_root/.github/scripts/prepare-vagrant-box-masters.sh"
|
||||
first_output="$fixture/first-output"
|
||||
second_output="$fixture/second-output"
|
||||
third_output="$fixture/third-output"
|
||||
|
||||
"$script" > "$first_output"
|
||||
if grep -Fq 'config.ssh.insert_key' "$MOCK_VAGRANTFILE_CAPTURE"; then
|
||||
printf 'prewarm Vagrantfile unexpectedly overrides Vagrant SSH key insertion\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
grep -Fq 'virtualbox.memory = 1024' "$MOCK_VAGRANTFILE_CAPTURE"
|
||||
grep -Fq 'virtualbox.cpus = 2' "$MOCK_VAGRANTFILE_CAPTURE"
|
||||
grep -Fq 'config.vm.boot_timeout = 600' "$MOCK_VAGRANTFILE_CAPTURE"
|
||||
mapping_file="$mock_master_root/bento_ubuntu-26.04-202606.01.0-amd64.uuid"
|
||||
test -s "$mapping_file"
|
||||
cmp -s "$mapping_file" "$mock_box_dir/master_id"
|
||||
grep -Fq 'Created and recorded owned master' "$first_output"
|
||||
grep -Fq 'modifyvm' "$MOCK_VBOX_LOG"
|
||||
grep -Fq 'setextradata' "$MOCK_VBOX_LOG"
|
||||
if grep -Fq "$unowned_uuid" "$MOCK_VBOX_LOG"; then
|
||||
printf 'unowned cached master UUID was unexpectedly inspected or modified\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
if grep -Eq 'unregistervm|closemedium' "$MOCK_VBOX_LOG"; then
|
||||
printf 'master preparation invoked a destructive VirtualBox command\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
: > "$MOCK_VAGRANT_LOG"
|
||||
"$script" > "$second_output"
|
||||
grep -Fq 'Reusing owned master' "$second_output"
|
||||
if grep -Fq 'up ' "$MOCK_VAGRANT_LOG"; then
|
||||
printf 'valid owned master was unexpectedly rebuilt\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
stale_uuid="$(tr -d '[:space:]' < "$mapping_file")"
|
||||
rm -f -- "$mock_state/vm-$stale_uuid"
|
||||
: > "$MOCK_VAGRANT_LOG"
|
||||
"$script" > "$third_output"
|
||||
grep -Fq 'rebuilding without deleting any VM or disk' "$third_output"
|
||||
grep -Fq 'up ' "$MOCK_VAGRANT_LOG"
|
||||
if grep -Eq 'unregistervm|closemedium' "$MOCK_VBOX_LOG"; then
|
||||
printf 'stale master recovery invoked a destructive VirtualBox command\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
printf 'Vagrant box master preparation fixture test passed\n'
|
||||
+28
@@ -0,0 +1,28 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
repo_root="$(git rev-parse --show-toplevel)"
|
||||
site_play="$repo_root/site.yml"
|
||||
|
||||
# #636: verify the "Pre tasks" play asserts that all k3s_cluster hosts have
|
||||
# unique hostnames, so a duplicate-hostname inventory fails fast instead of
|
||||
# silently breaking node registration/joining.
|
||||
grep -Fq -- 'Verify all cluster nodes have unique hostnames' "$site_play" || {
|
||||
printf 'site.yml is missing the unique-hostname preflight check\n' >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
# The check must deduplicate the cluster hostname list via the `unique` filter
|
||||
# and compare lengths, i.e. groups['k3s_cluster'] must be referenced.
|
||||
grep -Fq -- "groups['k3s_cluster']" "$site_play" || {
|
||||
printf 'unique-hostname check does not iterate the k3s_cluster group\n' >&2
|
||||
exit 1
|
||||
}
|
||||
|
||||
if ! grep -Eq -- 'cluster_hostnames.*\|.*unique|\| unique' "$site_play"; then
|
||||
printf 'unique-hostname check does not deduplicate the hostname list\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
printf 'Unique hostname preflight regression test passed\n'
|
||||
Executable
+24
@@ -0,0 +1,24 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
if (($# < 2)); then
|
||||
printf 'Usage: vagrant-up-timed.sh WORKDIR MACHINE [MACHINE ...]\n' >&2
|
||||
exit 2
|
||||
fi
|
||||
|
||||
workdir="$1"
|
||||
shift
|
||||
timing_log="${K3S_CI_CREATE_TIMING_LOG:-${RUNNER_TEMP:-/tmp}/k3s-ci-create-timing.log}"
|
||||
mkdir -p -- "${timing_log%/*}"
|
||||
|
||||
printf '%s batch-start machines=%s\n' "$(date --iso-8601=ns)" "$*" | tee -a "$timing_log"
|
||||
set +e
|
||||
VAGRANT_CWD="$workdir" vagrant up "$@" --provider virtualbox --no-provision 2>&1 |
|
||||
while IFS= read -r line; do
|
||||
printf '%s %s\n' "$(date --iso-8601=ns)" "$line"
|
||||
done | tee -a "$timing_log"
|
||||
rc=${PIPESTATUS[0]}
|
||||
set -e
|
||||
printf '%s batch-end rc=%d machines=%s\n' "$(date --iso-8601=ns)" "$rc" "$*" | tee -a "$timing_log"
|
||||
exit "$rc"
|
||||
Executable
+98
@@ -0,0 +1,98 @@
|
||||
#!/usr/bin/env bash
|
||||
|
||||
# The single-quoted expressions below are written into fake executables and
|
||||
# intentionally expand only when those executables run.
|
||||
# shellcheck disable=SC2016
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
repo_root=$(git rev-parse --show-toplevel)
|
||||
test_root=$(mktemp -d)
|
||||
fake_bin="$test_root/bin"
|
||||
fake_log="$test_root/vagrant.log"
|
||||
output="$test_root/output.txt"
|
||||
mkdir -p "$fake_bin"
|
||||
trap 'rm -rf "$test_root"' EXIT
|
||||
|
||||
printf '%s\n' \
|
||||
'#!/usr/bin/env bash' \
|
||||
'set -Eeuo pipefail' \
|
||||
'printf "%s\n" bento/debian-13 bento/rockylinux-10.1 bento/ubuntu-26.04' \
|
||||
>"$fake_bin/yq"
|
||||
|
||||
printf '%s\n' \
|
||||
'#!/usr/bin/env bash' \
|
||||
'set -Eeuo pipefail' \
|
||||
'emit_box() {' \
|
||||
' case "$1" in' \
|
||||
' bento/debian-13) version=202510.26.0 ;;' \
|
||||
' bento/rockylinux-10.1) version=202512.01.0 ;;' \
|
||||
' bento/ubuntu-26.04) version=202606.01.0 ;;' \
|
||||
' *) printf "Unexpected box: %s\n" "$1" >&2; exit 1 ;;' \
|
||||
' esac' \
|
||||
' printf "0,,box-name,%s\n" "$1"' \
|
||||
' printf "0,,box-provider,virtualbox\n"' \
|
||||
' printf "0,,box-version,%s\n" "$version"' \
|
||||
' printf "0,,box-architecture,amd64\n"' \
|
||||
'}' \
|
||||
'if [[ "${1:-}" == box && "${2:-}" == list ]]; then' \
|
||||
' emit_box bento/debian-13' \
|
||||
' emit_box bento/rockylinux-10.1' \
|
||||
' if [[ "${FAKE_PRESENT_MODE:-all}" == all ]]; then' \
|
||||
' emit_box bento/ubuntu-26.04' \
|
||||
' fi' \
|
||||
'elif [[ "${1:-}" == box && "${2:-}" == add ]]; then' \
|
||||
' printf "%s\n" "$*" >>"${FAKE_VAGRANT_LOG:?}"' \
|
||||
'else' \
|
||||
' printf "Unexpected vagrant arguments: %s\n" "$*" >&2' \
|
||||
' exit 1' \
|
||||
'fi' \
|
||||
>"$fake_bin/vagrant"
|
||||
chmod +x "$fake_bin/yq" "$fake_bin/vagrant"
|
||||
|
||||
PATH="$fake_bin:$PATH" \
|
||||
FAKE_VAGRANT_LOG="$fake_log" \
|
||||
"$repo_root/.github/download-boxes.sh" >"$output"
|
||||
grep -Fq 'All pinned Vagrant boxes are already present.' "$output"
|
||||
[[ ! -e "$fake_log" ]]
|
||||
|
||||
PATH="$fake_bin:$PATH" \
|
||||
FAKE_PRESENT_MODE=partial \
|
||||
FAKE_VAGRANT_LOG="$fake_log" \
|
||||
"$repo_root/.github/download-boxes.sh" >"$output"
|
||||
grep -Fxq \
|
||||
'box add --provider virtualbox --box-version 202606.01.0 --architecture amd64 bento/ubuntu-26.04' \
|
||||
"$fake_log"
|
||||
|
||||
incomplete_lock="$test_root/incomplete.lock"
|
||||
printf '%s\n' \
|
||||
'bento/debian-13 202510.26.0 amd64' \
|
||||
'bento/rockylinux-10.1 202512.01.0 amd64' \
|
||||
>"$incomplete_lock"
|
||||
if PATH="$fake_bin:$PATH" \
|
||||
VAGRANT_BOX_LOCK_FILE="$incomplete_lock" \
|
||||
FAKE_VAGRANT_LOG="$fake_log" \
|
||||
"$repo_root/.github/download-boxes.sh" >"$output" 2>&1; then
|
||||
printf 'Download script accepted a lock missing a scenario box.\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
grep -Fq 'Scenario boxes missing from the lock file:' "$output"
|
||||
grep -Fq 'bento/ubuntu-26.04' "$output"
|
||||
|
||||
duplicate_lock="$test_root/duplicate.lock"
|
||||
printf '%s\n' \
|
||||
'bento/debian-13 202510.26.0 amd64' \
|
||||
'bento/debian-13 202508.10.0 amd64' \
|
||||
'bento/rockylinux-10.1 202512.01.0 amd64' \
|
||||
'bento/ubuntu-26.04 202606.01.0 amd64' \
|
||||
>"$duplicate_lock"
|
||||
if PATH="$fake_bin:$PATH" \
|
||||
VAGRANT_BOX_LOCK_FILE="$duplicate_lock" \
|
||||
FAKE_VAGRANT_LOG="$fake_log" \
|
||||
"$repo_root/.github/download-boxes.sh" >"$output" 2>&1; then
|
||||
printf 'Download script accepted duplicate box lock entries.\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
grep -Fq 'Duplicate Vagrant box lock entries:' "$output"
|
||||
|
||||
printf 'Vagrant box download tests passed.\n'
|
||||
@@ -0,0 +1,4 @@
|
||||
# box version architecture
|
||||
bento/debian-13 202510.26.0 amd64
|
||||
bento/rockylinux-10.1 202512.01.0 amd64
|
||||
bento/ubuntu-26.04 202606.01.0 amd64
|
||||
@@ -0,0 +1,57 @@
|
||||
---
|
||||
name: "Cache"
|
||||
on:
|
||||
workflow_call:
|
||||
jobs:
|
||||
molecule:
|
||||
name: cache
|
||||
runs-on: [self-hosted, linux, x64, k3s-ci, virtualbox, nested-virt]
|
||||
env:
|
||||
PYTHON_VERSION: "3.11"
|
||||
VAGRANT_DEFAULT_PROVIDER: virtualbox
|
||||
VAGRANT_HOME: ${{ github.workspace }}/.vagrant-home
|
||||
|
||||
steps:
|
||||
- name: Check out the codebase
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # 7.0.1
|
||||
with:
|
||||
ref: ${{ github.event.pull_request.head.sha || github.sha }}
|
||||
|
||||
- name: Check nested VirtualBox platform
|
||||
run: |
|
||||
set -Eeuo pipefail
|
||||
grep -Eq 'vmx|svm' /proc/cpuinfo
|
||||
test -c /dev/kvm
|
||||
test -c /dev/vboxdrv
|
||||
VBoxManage --version
|
||||
vagrant --version
|
||||
test -r /etc/vbox/networks.conf
|
||||
test "$(stat -c '%u' /etc/vbox/networks.conf)" -eq 0
|
||||
free -h
|
||||
df -Pk "${RUNNER_TEMP}"
|
||||
|
||||
- name: Set up Python ${{ env.PYTHON_VERSION }}
|
||||
uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # 7.0.0
|
||||
with:
|
||||
python-version: ${{ env.PYTHON_VERSION }}
|
||||
cache: 'pip' # caching pip dependencies
|
||||
|
||||
- name: Cache Vagrant boxes
|
||||
id: cache-vagrant
|
||||
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # 6.1.0
|
||||
with:
|
||||
# This producer only needs to know whether the immutable cache exists.
|
||||
# Molecule jobs restore it after this job completes.
|
||||
lookup-only: true
|
||||
path: |
|
||||
.vagrant-home/boxes
|
||||
key: vagrant-boxes-${{ runner.name }}-${{ runner.os }}-${{ runner.arch }}-virtualbox-7.2-vagrant-2.4-${{ hashFiles('.github/vagrant-boxes.lock') }} # yamllint disable-line rule:line-length
|
||||
|
||||
- name: Download Vagrant boxes for all scenarios
|
||||
# An exact hit skips both cache restoration and upstream downloads.
|
||||
# A lock change builds and saves one clean, version-pinned cache.
|
||||
if: steps.cache-vagrant.outputs.cache-hit != 'true'
|
||||
run: |
|
||||
./.github/download-boxes.sh
|
||||
./.github/scripts/prepare-vagrant-box-masters.sh
|
||||
vagrant box list
|
||||
@@ -2,14 +2,43 @@
|
||||
name: "CI"
|
||||
on:
|
||||
pull_request:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
types:
|
||||
- opened
|
||||
- synchronize
|
||||
- reopened
|
||||
- ready_for_review
|
||||
paths-ignore:
|
||||
- '**/README.md'
|
||||
- '**/.gitignore'
|
||||
- '**/FUNDING.yml'
|
||||
- '**/host.ini'
|
||||
- '**/*.md'
|
||||
- '.github/ISSUE_TEMPLATE/**'
|
||||
- '**/.editorconfig'
|
||||
- '**/ansible.example.cfg'
|
||||
- '**/deploy.sh'
|
||||
- '**/LICENSE'
|
||||
- '**/reboot.sh'
|
||||
- '**/reset.sh'
|
||||
workflow_dispatch:
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
concurrency:
|
||||
group: ci-${{ github.event.pull_request.number || github.run_id }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
pre:
|
||||
if: github.actor != 'dependabot[bot]'
|
||||
uses: ./.github/workflows/cache.yml
|
||||
needs: [lint]
|
||||
lint:
|
||||
if: github.actor != 'dependabot[bot]'
|
||||
uses: ./.github/workflows/lint.yml
|
||||
test:
|
||||
if: github.actor != 'dependabot[bot]'
|
||||
uses: ./.github/workflows/test.yml
|
||||
needs: [lint]
|
||||
needs: [pre, lint]
|
||||
|
||||
+30
-23
@@ -7,35 +7,26 @@ jobs:
|
||||
name: Pre-Commit
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
PYTHON_VERSION: "3.10"
|
||||
PYTHON_VERSION: "3.12"
|
||||
|
||||
steps:
|
||||
- name: Check out the codebase
|
||||
uses: actions/checkout@e2f20e631ae6d7dd3b768f56a5d2af784dd54791 # v3 2.5.0
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # 7.0.1
|
||||
with:
|
||||
ref: ${{ github.event.pull_request.head.sha }}
|
||||
ref: ${{ github.event.pull_request.head.sha || github.sha }}
|
||||
|
||||
- name: Set up Python ${{ env.PYTHON_VERSION }}
|
||||
uses: actions/setup-python@75f3110429a8c05be0e1bf360334e4cced2b63fa # 2.3.3
|
||||
uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # 7.0.0
|
||||
with:
|
||||
python-version: ${{ env.PYTHON_VERSION }}
|
||||
cache: 'pip' # caching pip dependencies
|
||||
|
||||
- name: Cache pip
|
||||
uses: actions/cache@9b0c1fce7a93df8e3bb8926b0d6e9d89e92f20a7 # 3.0.11
|
||||
with:
|
||||
path: ~/.cache/pip
|
||||
key: ${{ runner.os }}-pip-${{ hashFiles('./requirements.txt') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-pip-
|
||||
|
||||
- name: Cache Ansible
|
||||
uses: actions/cache@9b0c1fce7a93df8e3bb8926b0d6e9d89e92f20a7 # 3.0.11
|
||||
- name: Restore Ansible cache
|
||||
id: cache-ansible
|
||||
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # 6.1.0
|
||||
with:
|
||||
path: ~/.ansible/collections
|
||||
key: ${{ runner.os }}-ansible-${{ hashFiles('collections/requirements.txt') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-ansible-
|
||||
key: ansible-${{ hashFiles('collections/requirements.yml') }}
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
@@ -47,21 +38,37 @@ jobs:
|
||||
python3 -m pip install -r requirements.txt
|
||||
echo "::endgroup::"
|
||||
|
||||
echo "::group::Install Ansible role requirements from collections/requirements.yml"
|
||||
ansible-galaxy install -r collections/requirements.yml
|
||||
echo "::endgroup::"
|
||||
- name: Install Ansible collections with retries
|
||||
if: steps.cache-ansible.outputs.cache-hit != 'true'
|
||||
run: |
|
||||
set -Eeuo pipefail
|
||||
for attempt in 1 2 3 4 5; do
|
||||
if ansible-galaxy collection install -r collections/requirements.yml; then
|
||||
exit 0
|
||||
fi
|
||||
echo "Ansible Galaxy attempt ${attempt} failed; retrying."
|
||||
sleep $((attempt * 10))
|
||||
done
|
||||
exit 1
|
||||
|
||||
- name: Save Ansible collection cache
|
||||
if: steps.cache-ansible.outputs.cache-hit != 'true'
|
||||
uses: actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # 6.1.0
|
||||
with:
|
||||
path: ~/.ansible/collections
|
||||
key: ansible-${{ hashFiles('collections/requirements.yml') }}
|
||||
|
||||
- name: Run pre-commit
|
||||
uses: pre-commit/action@646c83fcd040023954eafda54b4db0192ce70507 # 3.0.0
|
||||
uses: pre-commit/action@2c7b3805fd2a0fd8c1884dcaebf91fc102a13ecd # 3.0.1
|
||||
|
||||
ensure-pinned-actions:
|
||||
name: Ensure SHA Pinned Actions
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout code
|
||||
uses: actions/checkout@e2f20e631ae6d7dd3b768f56a5d2af784dd54791 # v3 2.5.0
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # 7.0.1
|
||||
- name: Ensure SHA pinned actions
|
||||
uses: zgosalvez/github-actions-ensure-sha-pinned-actions@af2eb3226618e2494e3d9084f515ad6dcf16e229 # 2.0.1
|
||||
uses: zgosalvez/github-actions-ensure-sha-pinned-actions@46cfe808a5f1588656ef299eedd0ce2fd7ec0dcc # 5.0.6
|
||||
with:
|
||||
allowlist: |
|
||||
aws-actions/
|
||||
|
||||
+70
-43
@@ -5,60 +5,64 @@ on:
|
||||
jobs:
|
||||
molecule:
|
||||
name: Molecule
|
||||
runs-on: macos-12
|
||||
runs-on: [self-hosted, linux, x64, k3s-ci, virtualbox, nested-virt]
|
||||
strategy:
|
||||
matrix:
|
||||
scenario:
|
||||
- default
|
||||
- ipv6
|
||||
- single_node
|
||||
fail-fast: false
|
||||
- calico
|
||||
- cilium
|
||||
- kube-vip
|
||||
# - ipv6
|
||||
fail-fast: true
|
||||
max-parallel: 1
|
||||
env:
|
||||
PYTHON_VERSION: "3.10"
|
||||
PYTHON_VERSION: "3.11"
|
||||
VAGRANT_DEFAULT_PROVIDER: virtualbox
|
||||
VAGRANT_HOME: ${{ github.workspace }}/.vagrant-home
|
||||
|
||||
steps:
|
||||
- name: Check out the codebase
|
||||
uses: actions/checkout@e2f20e631ae6d7dd3b768f56a5d2af784dd54791 # v3 2.5.0
|
||||
uses: actions/checkout@3d3c42e5aac5ba805825da76410c181273ba90b1 # 7.0.1
|
||||
with:
|
||||
ref: ${{ github.event.pull_request.head.sha }}
|
||||
ref: ${{ github.event.pull_request.head.sha || github.sha }}
|
||||
|
||||
- name: Configure VirtualBox
|
||||
run: |-
|
||||
sudo mkdir -p /etc/vbox
|
||||
cat <<EOF | sudo tee -a /etc/vbox/networks.conf > /dev/null
|
||||
* 192.168.30.0/24
|
||||
* fdad:bad:ba55::/64
|
||||
EOF
|
||||
- name: Clean repository-owned resources before testing
|
||||
run: ./.github/scripts/cleanup-runner-resources.sh --apply
|
||||
|
||||
- name: Cache pip
|
||||
uses: actions/cache@9b0c1fce7a93df8e3bb8926b0d6e9d89e92f20a7 # 3.0.11
|
||||
with:
|
||||
path: ~/.cache/pip
|
||||
key: ${{ runner.os }}-pip-${{ hashFiles('./requirements.txt') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-pip-
|
||||
- name: Record host-only network baseline
|
||||
run: ./.github/scripts/cleanup-runner-resources.sh --snapshot
|
||||
|
||||
- name: Cache Vagrant boxes
|
||||
uses: actions/cache@9b0c1fce7a93df8e3bb8926b0d6e9d89e92f20a7 # 3.0.11
|
||||
with:
|
||||
path: |
|
||||
~/.vagrant.d/boxes
|
||||
key: vagrant-boxes-${{ hashFiles('**/molecule.yml') }}
|
||||
restore-keys: |
|
||||
vagrant-boxes
|
||||
|
||||
- name: Download Vagrant boxes for all scenarios
|
||||
# To save some cache space, all scenarios share the same cache key.
|
||||
# On the other hand, this means that the cache contents should be
|
||||
# the same across all scenarios. This step ensures that.
|
||||
run: ./.github/download-boxes.sh
|
||||
- name: Check nested VirtualBox platform
|
||||
run: |
|
||||
set -Eeuo pipefail
|
||||
grep -Eq 'vmx|svm' /proc/cpuinfo
|
||||
test -c /dev/kvm
|
||||
test -c /dev/vboxdrv
|
||||
VBoxManage --version
|
||||
vagrant --version
|
||||
test -r /etc/vbox/networks.conf
|
||||
test "$(stat -c '%u' /etc/vbox/networks.conf)" -eq 0
|
||||
free -h
|
||||
df -Pk "${RUNNER_TEMP}"
|
||||
|
||||
- name: Set up Python ${{ env.PYTHON_VERSION }}
|
||||
uses: actions/setup-python@75f3110429a8c05be0e1bf360334e4cced2b63fa # 2.3.3
|
||||
uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # 7.0.0
|
||||
with:
|
||||
python-version: ${{ env.PYTHON_VERSION }}
|
||||
cache: 'pip' # caching pip dependencies
|
||||
|
||||
- name: Restore vagrant Boxes cache
|
||||
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # 6.1.0
|
||||
with:
|
||||
path: .vagrant-home/boxes
|
||||
key: vagrant-boxes-${{ runner.name }}-${{ runner.os }}-${{ runner.arch }}-virtualbox-7.2-vagrant-2.4-${{ hashFiles('.github/vagrant-boxes.lock') }} # yamllint disable-line rule:line-length
|
||||
fail-on-cache-miss: true
|
||||
|
||||
- name: Prepare runner-owned Vagrant box masters
|
||||
run: ./.github/scripts/prepare-vagrant-box-masters.sh
|
||||
|
||||
- name: Install dependencies
|
||||
run: |
|
||||
echo "::group::Upgrade pip"
|
||||
@@ -70,22 +74,45 @@ jobs:
|
||||
echo "::endgroup::"
|
||||
|
||||
- name: Test with molecule
|
||||
run: molecule test --scenario-name ${{ matrix.scenario }}
|
||||
run: |
|
||||
set -Eeuo pipefail
|
||||
resource_dir="${RUNNER_TEMP}/logs/resources/${{ matrix.scenario }}"
|
||||
timing_file="${RUNNER_TEMP}/logs/timing/${{ matrix.scenario }}.txt"
|
||||
mkdir -p -- "${timing_file%/*}"
|
||||
./.github/scripts/monitor-runner-resources.sh "$resource_dir" 10 &
|
||||
monitor_pid=$!
|
||||
stop_monitor() {
|
||||
kill -TERM "$monitor_pid" 2>/dev/null || true
|
||||
wait "$monitor_pid" 2>/dev/null || true
|
||||
}
|
||||
trap stop_monitor EXIT
|
||||
/usr/bin/time -v -o "$timing_file" \
|
||||
molecule test --scenario-name ${{ matrix.scenario }}
|
||||
timeout-minutes: 180
|
||||
env:
|
||||
ANSIBLE_K3S_LOG_DIR: ${{ runner.temp }}/logs/k3s-ansible/${{ matrix.scenario }}
|
||||
ANSIBLE_SSH_RETRIES: 4
|
||||
ANSIBLE_TIMEOUT: 60
|
||||
ANSIBLE_TIMEOUT: 120
|
||||
PY_COLORS: 1
|
||||
ANSIBLE_FORCE_COLOR: 1
|
||||
K3S_CI_CREATE_TIMING_LOG: ${{ runner.temp }}/logs/timing/${{ matrix.scenario }}-create.log
|
||||
|
||||
- name: Collect runner diagnostics
|
||||
if: always()
|
||||
run: ./.github/scripts/collect-runner-diagnostics.sh "${RUNNER_TEMP}/logs/runner"
|
||||
env:
|
||||
K3S_CI_SCENARIO_NAME: ${{ matrix.scenario }}
|
||||
|
||||
- name: Clean repository-owned resources after testing
|
||||
if: always()
|
||||
run: ./.github/scripts/cleanup-runner-resources.sh --apply
|
||||
|
||||
- name: Upload log files
|
||||
if: always() # do this even if a step before has failed
|
||||
uses: actions/upload-artifact@83fd05a356d7e2593de66fc9913b3002723633cb # 3.1.1
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # 7.0.1
|
||||
with:
|
||||
name: logs
|
||||
name: logs-${{ matrix.scenario }}-${{ github.run_id }}-${{ github.run_attempt }}
|
||||
path: |
|
||||
${{ runner.temp }}/logs
|
||||
|
||||
- name: Delete old box versions
|
||||
if: always() # do this even if a step before has failed
|
||||
run: vagrant box prune --force
|
||||
if-no-files-found: warn
|
||||
retention-days: 14
|
||||
|
||||
@@ -1,2 +1,6 @@
|
||||
.env/
|
||||
*.log
|
||||
ansible.cfg
|
||||
.ansible/
|
||||
kubeconfig
|
||||
zIgnore/
|
||||
|
||||
+111
-6
@@ -1,7 +1,7 @@
|
||||
---
|
||||
repos:
|
||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||
rev: 3298ddab3c13dd77d6ce1fc0baf97691430d84b0 # v4.3.0
|
||||
rev: v4.5.0
|
||||
hooks:
|
||||
- id: requirements-txt-fixer
|
||||
- id: sort-simple-yaml
|
||||
@@ -12,24 +12,129 @@ repos:
|
||||
- id: trailing-whitespace
|
||||
args: [--markdown-linebreak-ext=md]
|
||||
- repo: https://github.com/adrienverge/yamllint.git
|
||||
rev: 9cce2940414e9560ae4c8518ddaee2ac1863a4d2 # v1.28.0
|
||||
rev: v1.33.0
|
||||
hooks:
|
||||
- id: yamllint
|
||||
args: [-c=.yamllint]
|
||||
- repo: https://github.com/ansible-community/ansible-lint.git
|
||||
rev: a058554b9bcf88f12ad09ab9fb93b267a214368f # v6.8.6
|
||||
rev: v6.22.2
|
||||
hooks:
|
||||
- id: ansible-lint
|
||||
additional_dependencies: [ansible-core==2.18.0]
|
||||
language_version: python3.12
|
||||
args: [--offline]
|
||||
- repo: https://github.com/shellcheck-py/shellcheck-py
|
||||
rev: 4c7c3dd7161ef39e984cb295e93a968236dc8e8a # v0.8.0.4
|
||||
rev: v0.9.0.6
|
||||
hooks:
|
||||
- id: shellcheck
|
||||
- repo: https://github.com/Lucas-C/pre-commit-hooks
|
||||
rev: 04618e68aa2380828a36a23ff5f65a06ae8f59b9 # v1.3.1
|
||||
rev: v1.5.4
|
||||
hooks:
|
||||
- id: remove-crlf
|
||||
- id: remove-tabs
|
||||
- repo: https://github.com/sirosen/texthooks
|
||||
rev: 30d9af95631de0d7cff4e282bde9160d38bb0359 # 0.4.0
|
||||
rev: 0.6.4
|
||||
hooks:
|
||||
- id: fix-smartquotes
|
||||
- repo: local
|
||||
hooks:
|
||||
- id: cleanup-runner-resources-test
|
||||
name: cleanup runner resources test
|
||||
entry: .github/scripts/test-cleanup-runner-resources.sh
|
||||
language: system
|
||||
pass_filenames: false
|
||||
files: ^\.github/scripts/(cleanup-runner-resources|test-cleanup-runner-resources)\.sh$
|
||||
- id: download-vagrant-boxes-test
|
||||
name: Vagrant box download test
|
||||
entry: .github/test-download-boxes.sh
|
||||
language: system
|
||||
pass_filenames: false
|
||||
files: ^\.github/(download-boxes|test-download-boxes)\.sh$|^\.github/vagrant-boxes\.lock$
|
||||
- id: prepare-vagrant-box-masters-test
|
||||
name: Vagrant box master preparation test
|
||||
entry: .github/scripts/test-prepare-vagrant-box-masters.sh
|
||||
language: system
|
||||
pass_filenames: false
|
||||
files: ^\.github/scripts/(prepare-vagrant-box-masters|test-prepare-vagrant-box-masters)\.sh$
|
||||
- id: k3s-server-bootstrap-test
|
||||
name: K3s transient bootstrap test
|
||||
entry: .github/scripts/test-k3s-server-bootstrap.sh
|
||||
language: system
|
||||
pass_filenames: false
|
||||
files: ^roles/k3s_server/tasks/(main|join_master)\.yml$|^\.github/scripts/test-k3s-server-bootstrap\.sh$
|
||||
- id: unique-hostname-precheck-test
|
||||
name: Unique hostname precheck test
|
||||
entry: .github/scripts/test-unique-hostname-precheck.sh
|
||||
language: system
|
||||
pass_filenames: false
|
||||
files: ^site\.yml$|^\.github/scripts/test-unique-hostname-precheck\.sh$
|
||||
- id: disable-swap-test
|
||||
name: Disable swap test
|
||||
entry: .github/scripts/test-disable-swap.sh
|
||||
language: system
|
||||
pass_filenames: false
|
||||
files: ^roles/prereq/(tasks/main|defaults/main)\.yml$|^\.github/scripts/test-disable-swap\.sh$
|
||||
- id: cilium-bgp-manifest-test
|
||||
name: Cilium BGP manifest test
|
||||
entry: python3 .github/scripts/test-cilium-bgp-manifest.py
|
||||
language: python
|
||||
additional_dependencies:
|
||||
- Jinja2>=3.1
|
||||
pass_filenames: false
|
||||
files: ^roles/k3s_server_post/templates/cilium\.crs\.j2$|^\.github/scripts/test-cilium-bgp-manifest\.py$
|
||||
- id: cilium-envoy-toggle-test
|
||||
name: Cilium Envoy toggle test
|
||||
entry: python3 .github/scripts/test-cilium-envoy-toggle.py
|
||||
language: python
|
||||
additional_dependencies:
|
||||
- Jinja2>=3.1
|
||||
- PyYAML
|
||||
pass_filenames: false
|
||||
files: ^roles/k3s_server_post/tasks/cilium\.yml$|^\.github/scripts/test-cilium-envoy-toggle\.py$
|
||||
- id: kube-vip-manifest-test
|
||||
name: kube-vip manifest test
|
||||
entry: python3 .github/scripts/test-kube-vip-manifest.py
|
||||
language: python
|
||||
additional_dependencies:
|
||||
- Jinja2>=3.1
|
||||
pass_filenames: false
|
||||
files: ^roles/k3s_server/templates/vip\.yaml\.j2$|^\.github/scripts/test-kube-vip-manifest\.py$
|
||||
- id: metallb-remote-read-test
|
||||
name: MetalLB remote read test
|
||||
entry: .github/scripts/test-metallb-remote-read.sh
|
||||
language: system
|
||||
pass_filenames: false
|
||||
files: ^roles/k3s_server/tasks/metallb\.yml$|^\.github/scripts/test-metallb-remote-read\.sh$
|
||||
- id: metallb-interfaces-test
|
||||
name: MetalLB interfaces test
|
||||
entry: python3 .github/scripts/test-metallb-interfaces.py
|
||||
language: python
|
||||
additional_dependencies:
|
||||
- Jinja2>=3.1
|
||||
pass_filenames: false
|
||||
files: ^roles/k3s_server_post/templates/metallb\.crs\.j2$|^\.github/scripts/test-metallb-interfaces\.py$
|
||||
- id: metallb-namespace-test
|
||||
name: MetalLB namespace test
|
||||
entry: python3 .github/scripts/test-metallb-namespace.py
|
||||
language: python
|
||||
additional_dependencies:
|
||||
- PyYAML
|
||||
pass_filenames: false
|
||||
files: ^roles/k3s_server_post/tasks/metallb\.yml$|^\.github/scripts/test-metallb-namespace\.py$
|
||||
- id: metallb-deploy-condition-test
|
||||
name: MetalLB deploy condition test
|
||||
entry: python3 .github/scripts/test-metallb-deploy-condition.py
|
||||
language: python
|
||||
additional_dependencies:
|
||||
- Jinja2>=3.1
|
||||
- PyYAML
|
||||
pass_filenames: false
|
||||
files: ^roles/k3s_server/tasks/main\.yml$|^roles/k3s_server_post/tasks/main\.yml$|^\.github/scripts/test-metallb-deploy-condition\.py$ # noqa yaml[line-length]
|
||||
- id: default-interface-test
|
||||
name: default interface test
|
||||
entry: python3 .github/scripts/test-default-interface.py
|
||||
language: python
|
||||
additional_dependencies:
|
||||
- Jinja2>=3.1
|
||||
pass_filenames: false
|
||||
files: ^inventory/sample/group_vars/all\.yml$|^\.github/scripts/test-default-interface\.py$
|
||||
|
||||
@@ -2,8 +2,19 @@
|
||||
extends: default
|
||||
|
||||
rules:
|
||||
comments:
|
||||
min-spaces-from-content: 1
|
||||
comments-indentation: false
|
||||
braces:
|
||||
max-spaces-inside: 1
|
||||
octal-values:
|
||||
forbid-implicit-octal: true
|
||||
forbid-explicit-octal: true
|
||||
line-length:
|
||||
max: 120
|
||||
level: warning
|
||||
truthy:
|
||||
allowed-values: ['true', 'false', 'yes', 'no']
|
||||
allowed-values: ["true", "false"]
|
||||
|
||||
ignore:
|
||||
- galaxy.yml
|
||||
|
||||
@@ -0,0 +1,130 @@
|
||||
# Agent Guide
|
||||
|
||||
This file is the canonical repository guide for coding agents and automated contributors. Read it before making
|
||||
changes. Human contributors should also review [CONTRIBUTING.md](CONTRIBUTING.md).
|
||||
|
||||
## Project overview
|
||||
|
||||
This repository is an Ansible collection that provisions and resets highly available k3s clusters. It supports
|
||||
multiple networking choices, including Flannel, Calico, Cilium, kube-vip, and MetalLB.
|
||||
|
||||
The main entry points are:
|
||||
|
||||
- `site.yml`: provision or update a cluster.
|
||||
- `reset.yml`: remove k3s from a cluster.
|
||||
- `reboot.yml`: reboot cluster nodes.
|
||||
- `inventory/sample/`: example inventory and variables.
|
||||
- `roles/`: reusable Ansible roles used by the playbooks.
|
||||
- `molecule/`: integration scenarios run by CI.
|
||||
- `.github/scripts/`: CI support scripts and focused regression tests.
|
||||
|
||||
## Source of truth
|
||||
|
||||
- Role defaults belong in `roles/<role>/defaults/main.yml`.
|
||||
- Tasks belong in `roles/<role>/tasks/` and handlers in `roles/<role>/handlers/`.
|
||||
- Example user configuration belongs in `inventory/sample/`.
|
||||
- User-facing setup and variable documentation belongs in `README.md`.
|
||||
- Contributor workflows and review expectations belong in `CONTRIBUTING.md`.
|
||||
- Agent-specific repository instructions belong in this file.
|
||||
|
||||
Keep `CLAUDE.md` and `.github/copilot-instructions.md` as small pointers to this file. Do not duplicate these
|
||||
instructions in tool-specific files.
|
||||
|
||||
## Development setup
|
||||
|
||||
Use a Python virtual environment. Do not commit the environment, generated logs, inventories, kubeconfigs, or
|
||||
credentials.
|
||||
|
||||
```bash
|
||||
python3 -m venv .env
|
||||
source .env/bin/activate
|
||||
python3 -m pip install -r requirements.txt
|
||||
ansible-galaxy collection install -r collections/requirements.yml
|
||||
pre-commit install
|
||||
```
|
||||
|
||||
`ansible.cfg` is intentionally ignored. Copy `ansible.example.cfg` when local configuration is needed.
|
||||
|
||||
## Working rules
|
||||
|
||||
1. Inspect the current branch and worktree before editing. Preserve unrelated user changes.
|
||||
2. Keep changes focused. Avoid drive-by formatting or dependency updates.
|
||||
3. Never add real IP addresses, hostnames, tokens, private keys, kubeconfigs, or inventory secrets.
|
||||
4. Use placeholders in examples and redact sensitive values from logs and issue reports.
|
||||
5. Preserve idempotence. An already-converged host should not report changes without a real state transition.
|
||||
6. Prefer Ansible modules over `ansible.builtin.command` or `ansible.builtin.shell`. When a command is required,
|
||||
define accurate `changed_when` and `failed_when` behavior.
|
||||
7. Use fully qualified collection names, such as `ansible.builtin.copy`.
|
||||
8. Put configurable values in role defaults or inventory variables. Avoid embedding environment-specific values in
|
||||
tasks and templates.
|
||||
9. Maintain compatibility with the operating systems and architectures listed in `README.md`.
|
||||
10. Do not weaken lint rules, tests, or CI checks to make a change pass.
|
||||
|
||||
## Change guidance
|
||||
|
||||
### Ansible tasks and roles
|
||||
|
||||
- Use descriptive task names in sentence case.
|
||||
- Use YAML booleans (`true` and `false`) rather than aliases.
|
||||
- Quote file modes, for example `mode: "0644"`.
|
||||
- Notify handlers only when the managed resource changes.
|
||||
- Use `become: true` only where privilege escalation is needed.
|
||||
- Update role defaults, sample inventory, and the README together when adding or renaming user-facing variables.
|
||||
- Check reset behavior when provisioning introduces persistent services, files, mounts, or network state.
|
||||
|
||||
### Templates and manifests
|
||||
|
||||
- Keep Jinja logic small and readable. Move complicated decisions into task variables where practical.
|
||||
- Render valid YAML after Jinja evaluation.
|
||||
- Preserve explicit handling for optional and undefined variables.
|
||||
- Add or update a focused test under `.github/scripts/` when changing generated Kubernetes manifests or bootstrap
|
||||
behavior.
|
||||
|
||||
### Molecule scenarios
|
||||
|
||||
- Reuse `molecule/resources/` for shared behavior.
|
||||
- Put scenario-specific inputs in `molecule/<scenario>/overrides.yml` and `verify-vars.yml`.
|
||||
- Update `molecule/README.md` when adding, removing, or materially changing a scenario.
|
||||
- Clean up resources created by tests, including failure paths.
|
||||
|
||||
## Validation
|
||||
|
||||
Run the smallest relevant checks while iterating, then run the complete local validation before considering a change
|
||||
ready:
|
||||
|
||||
```bash
|
||||
pre-commit run --all-files
|
||||
```
|
||||
|
||||
For playbook or role changes, also run syntax checks with a non-sensitive inventory:
|
||||
|
||||
```bash
|
||||
ansible-playbook site.yml --syntax-check -i inventory/sample/hosts.ini
|
||||
ansible-playbook reset.yml --syntax-check -i inventory/sample/hosts.ini
|
||||
```
|
||||
|
||||
Run focused regression scripts when their related files change. The mapping is defined in
|
||||
`.pre-commit-config.yaml`.
|
||||
|
||||
Molecule tests require Vagrant, VirtualBox, host-only networking, and substantial local resources. Run the most
|
||||
relevant scenario when that environment is available:
|
||||
|
||||
```bash
|
||||
molecule test --scenario-name <scenario>
|
||||
```
|
||||
|
||||
If a required test can't be run locally, state exactly which check was skipped and why. Never claim a check passed
|
||||
unless it was executed.
|
||||
|
||||
## Documentation and review
|
||||
|
||||
- Keep commands copyable and examples free of secrets.
|
||||
- Update documentation in the same change as user-visible behavior.
|
||||
- Explain behavior changes, compatibility concerns, operational risks, and rollback steps in the pull request.
|
||||
- Use conventional commit messages with a scope, for example `fix(k3s-server): handle an existing token safely`.
|
||||
- Do not commit, push, open a pull request, or modify remote resources unless the user explicitly requests it.
|
||||
|
||||
## Definition of done
|
||||
|
||||
A change is ready for review when it is focused, documented, linted, tested in proportion to its risk, and shown in a
|
||||
clean diff with no secrets or generated artifacts.
|
||||
@@ -0,0 +1,3 @@
|
||||
@AGENTS.md
|
||||
|
||||
`AGENTS.md` is the canonical repository guide. Follow it for all work in this repository.
|
||||
@@ -0,0 +1,94 @@
|
||||
# Contributing to k3s-ansible
|
||||
|
||||
Thank you for improving k3s-ansible. Contributions should be focused, safe to apply to existing clusters, and tested
|
||||
in proportion to their operational impact.
|
||||
|
||||
## Before opening an issue
|
||||
|
||||
- Search existing issues and discussions for the same behavior.
|
||||
- Review the [troubleshooting discussion](https://github.com/timothystewart6/k3s-ansible/discussions/20).
|
||||
- Remove tokens, credentials, public IP addresses, private hostnames, and other sensitive values from logs and
|
||||
configuration.
|
||||
- For support requests, include the k3s-ansible revision, Ansible version, target operating system, architecture,
|
||||
network provider, relevant sanitized variables, and a minimal reproduction.
|
||||
|
||||
Use the bug report template for reproducible defects and the feature request template for proposed behavior.
|
||||
|
||||
## Development environment
|
||||
|
||||
Fork and clone the repository, then create a branch from the latest `master`:
|
||||
|
||||
```bash
|
||||
git switch master
|
||||
git pull --ff-only
|
||||
git switch -c <type>/<short-description>
|
||||
```
|
||||
|
||||
Create a Python environment and install the pinned development dependencies:
|
||||
|
||||
```bash
|
||||
python3 -m venv .env
|
||||
source .env/bin/activate
|
||||
python3 -m pip install -r requirements.txt
|
||||
ansible-galaxy collection install -r collections/requirements.yml
|
||||
pre-commit install
|
||||
```
|
||||
|
||||
Copy `ansible.example.cfg` to the ignored `ansible.cfg` file if local Ansible configuration is needed. Start custom
|
||||
inventories from `inventory/sample/`, keep them out of Git, and never use production credentials in tests.
|
||||
|
||||
## Making changes
|
||||
|
||||
- Keep each pull request focused on one problem or feature.
|
||||
- Follow [AGENTS.md](AGENTS.md) for repository structure, implementation conventions, and safety requirements.
|
||||
- Preserve idempotence and existing-cluster compatibility.
|
||||
- Add or update tests for behavior changes and regressions.
|
||||
- Update role defaults, sample inventory, and documentation when user-facing variables change.
|
||||
- Consider both provisioning and reset behavior for persistent resources.
|
||||
- Avoid unrelated reformatting and generated files.
|
||||
|
||||
## Validation
|
||||
|
||||
Run all pre-commit checks before submitting a pull request:
|
||||
|
||||
```bash
|
||||
pre-commit run --all-files
|
||||
```
|
||||
|
||||
For playbook or role changes, run syntax checks:
|
||||
|
||||
```bash
|
||||
ansible-playbook site.yml --syntax-check -i inventory/sample/hosts.ini
|
||||
ansible-playbook reset.yml --syntax-check -i inventory/sample/hosts.ini
|
||||
```
|
||||
|
||||
Run the most relevant Molecule scenario when Vagrant, VirtualBox, and the required host networking are available:
|
||||
|
||||
```bash
|
||||
molecule test --scenario-name <scenario>
|
||||
```
|
||||
|
||||
See [molecule/README.md](molecule/README.md) for scenario details and local requirements. Pull requests should list
|
||||
every check that was run and clearly identify checks that could not be run locally.
|
||||
|
||||
## Commits and pull requests
|
||||
|
||||
Use a conventional commit subject with a scope:
|
||||
|
||||
```text
|
||||
type(scope): short description
|
||||
```
|
||||
|
||||
Common types are `feat`, `fix`, `refactor`, `test`, `docs`, and `chore`. Write subjects in the imperative mood and
|
||||
keep commits logically focused.
|
||||
|
||||
Pull requests should:
|
||||
|
||||
- Describe the problem and the resulting behavior.
|
||||
- Identify compatibility, security, networking, and upgrade risks.
|
||||
- Include testing evidence without sensitive data.
|
||||
- Call out documentation and sample configuration changes.
|
||||
- Link related issues with `Fixes #<issue>` when applicable.
|
||||
- Avoid checking boxes for tests that were not run.
|
||||
|
||||
Maintainers may ask for a change to be split when unrelated work makes it difficult to review or roll back.
|
||||
@@ -2,51 +2,70 @@
|
||||
|
||||

|
||||
|
||||
This playbook will build an HA Kubernetes cluster with `k3s`, `kube-vip` and MetalLB via `ansible`.
|
||||
This Ansible collection builds a highly available Kubernetes cluster with k3s. It supports kube-vip for the control
|
||||
plane virtual IP, multiple CNI options, and either MetalLB or kube-vip for service load balancing.
|
||||
|
||||
This is based on the work from [this fork](https://github.com/212850a/k3s-ansible) which is based on the work from [k3s-io/k3s-ansible](https://github.com/k3s-io/k3s-ansible). It uses [kube-vip](https://kube-vip.chipzoller.dev/) to create a load balancer for control plane, and [metal-lb](https://metallb.universe.tf/installation/) for its service `LoadBalancer`.
|
||||
This is based on the work from [this fork](https://github.com/212850a/k3s-ansible) which is based on the work from [k3s-io/k3s-ansible](https://github.com/k3s-io/k3s-ansible). It uses [kube-vip](https://kube-vip.io/) to create a load balancer for control plane, and [metal-lb](https://metallb.universe.tf/installation/) for its service `LoadBalancer`.
|
||||
|
||||
If you want more context on how this works, see:
|
||||
For more context on how it works, see:
|
||||
|
||||
📄 [Documentation](https://docs.technotim.live/posts/k3s-etcd-ansible/) (including example commands)
|
||||
📄 [Documentation](https://technotim.com/posts/k3s-etcd-ansible/) (including example commands)
|
||||
|
||||
📺 [Watch the Video](https://www.youtube.com/watch?v=CbkEWcUZ7zM)
|
||||
|
||||
## Project guides
|
||||
|
||||
- [Getting started](#-getting-started)
|
||||
- [Configuration variables](#variables)
|
||||
- [Upgrading an existing cluster](#-upgrading-an-existing-cluster)
|
||||
- [Local Molecule testing](molecule/README.md)
|
||||
- [Contributing guidelines](CONTRIBUTING.md)
|
||||
- [Repository guide for coding agents](AGENTS.md)
|
||||
|
||||
## 📖 k3s Ansible Playbook
|
||||
|
||||
Build a Kubernetes cluster using Ansible with k3s. The goal is easily install a HA Kubernetes cluster on machines running:
|
||||
Build a Kubernetes cluster using Ansible and k3s. The goal is to make a highly available cluster straightforward to
|
||||
install on machines running:
|
||||
|
||||
- [x] Debian (tested on version 11)
|
||||
- [x] Ubuntu (tested on version 22.04)
|
||||
- [x] Rocky (tested on version 9)
|
||||
- [x] Debian (tested on version 13)
|
||||
- [x] Ubuntu (tested on version 26.04 LTS)
|
||||
- [x] Rocky (tested on version 10)
|
||||
|
||||
on processor architecture:
|
||||
Supported processor architectures are:
|
||||
|
||||
- [X] x64
|
||||
- [X] arm64
|
||||
- [X] armhf
|
||||
- [x] x64
|
||||
- [x] arm64
|
||||
- [x] armhf
|
||||
|
||||
## ✅ System requirements
|
||||
|
||||
- Deployment environment must have Ansible 2.4.0+. If you need a quick primer on Ansible [you can check out my docs and setting up Ansible](https://docs.technotim.live/posts/ansible-automation/).
|
||||
- The control node, which runs the Ansible commands, must have Ansible 2.11 or newer. For a quick primer, see
|
||||
[setting up Ansible](https://technotim.com/posts/ansible-automation/).
|
||||
|
||||
- You will also need to install collections that this playbook uses by running `ansible-galaxy collection install -r ./collections/requirements.yml` (important❗)
|
||||
- Install the required collections with
|
||||
`ansible-galaxy collection install -r ./collections/requirements.yml`.
|
||||
|
||||
- [`netaddr` package](https://pypi.org/project/netaddr/) must be available to Ansible. If you have installed Ansible via apt, this is already taken care of. If you have installed Ansible via `pip`, make sure to install `netaddr` into the respective virtual environment.
|
||||
|
||||
- `server` and `agent` nodes should have passwordless SSH access, if not you can supply arguments to provide credentials `--ask-pass --ask-become-pass` to each command.
|
||||
- Server and agent nodes should support passwordless SSH access. Otherwise, pass `--ask-pass --ask-become-pass` to
|
||||
each playbook command.
|
||||
|
||||
- Every node in the cluster must have a **unique hostname**. k3s registers each node keyed by its hostname, so
|
||||
two nodes with the same hostname cannot join the cluster. `site.yml` asserts this up front and fails fast if
|
||||
any duplicate is found.
|
||||
|
||||
## 🚀 Getting Started
|
||||
|
||||
### 🍴 Preparation
|
||||
|
||||
First create a new directory based on the `sample` directory within the `inventory` directory:
|
||||
Create a cluster-specific inventory from the sample. The `inventory/` directory ignores custom inventory content so
|
||||
credentials and environment details aren't committed accidentally.
|
||||
|
||||
```bash
|
||||
cp -R inventory/sample inventory/my-cluster
|
||||
```
|
||||
|
||||
Second, edit `inventory/my-cluster/hosts.ini` to match the system information gathered above
|
||||
Edit `inventory/my-cluster/hosts.ini` to match the target hosts.
|
||||
|
||||
For example:
|
||||
|
||||
@@ -67,7 +86,10 @@ node
|
||||
|
||||
If multiple hosts are in the master group, the playbook will automatically set up k3s in [HA mode with etcd](https://rancher.com/docs/k3s/latest/en/installation/ha-embedded/).
|
||||
|
||||
This requires at least k3s version `1.19.1` however the version is configurable by using the `k3s_version` variable.
|
||||
Copy `ansible.example.cfg` to `ansible.cfg`, then update its inventory path. The local `ansible.cfg` file is ignored by
|
||||
Git.
|
||||
|
||||
The minimum k3s version is `1.19.1`. Select the desired version with the `k3s_version` variable.
|
||||
|
||||
If needed, you can also edit `inventory/my-cluster/group_vars/all.yml` to match your environment.
|
||||
|
||||
@@ -79,7 +101,8 @@ Start provisioning of the cluster using the following command:
|
||||
ansible-playbook site.yml -i inventory/my-cluster/hosts.ini
|
||||
```
|
||||
|
||||
After deployment control plane will be accessible via virtual ip-address which is defined in inventory/group_vars/all.yml as `apiserver_endpoint`
|
||||
After deployment, the control plane is accessible through the virtual IP defined by `apiserver_endpoint` in the
|
||||
inventory variables.
|
||||
|
||||
### 🔥 Remove k3s cluster
|
||||
|
||||
@@ -87,23 +110,166 @@ After deployment control plane will be accessible via virtual ip-address which i
|
||||
ansible-playbook reset.yml -i inventory/my-cluster/hosts.ini
|
||||
```
|
||||
|
||||
>You should also reboot these nodes due to the VIP not being destroyed
|
||||
> Reboot the nodes after reset because the virtual IP may remain configured.
|
||||
|
||||
### ⏻️ Reboot Cluster Nodes
|
||||
|
||||
Reboot all cluster nodes at once or stage the reboot across the cluster.
|
||||
|
||||
```bash
|
||||
ansible-playbook reboot.yml -i inventory/my-cluster/hosts.ini
|
||||
```
|
||||
|
||||
To reboot the nodes in batches, set `concurrent_reboots` to the number of nodes
|
||||
to reboot at a time (or a percentage). Optionally set `wait_seconds_after_reboot`
|
||||
to pause after each batch so pods in the freshly rebooted batch can settle
|
||||
before the next batch reboots.
|
||||
|
||||
```bash
|
||||
ansible-playbook reboot.yml -i inventory/my-cluster/hosts.ini \
|
||||
--extra-vars 'concurrent_reboots=2 wait_seconds_after_reboot=30'
|
||||
```
|
||||
|
||||
## 🔁 Upgrading an existing cluster
|
||||
|
||||
These version variables select the components used for a **fresh** installation.
|
||||
They are not a supported direct in-place upgrade path for an existing cluster.
|
||||
K3s, Calico, and Cilium each require staged upgrades for long-lived clusters.
|
||||
|
||||
- **K3s**: do not jump an embedded-etcd cluster straight to Kubernetes 1.36.
|
||||
Upgrade one Kubernetes minor version at a time. From the sample default
|
||||
(`v1.30.2+k3s2`) the sequence is: the latest supported 1.30 patch, then 1.31,
|
||||
1.32, a 1.33 patch that contains etcd 3.5.26 (for example `v1.33.7+k3s3`),
|
||||
then 1.34, 1.35, and finally 1.36. Upgrade servers one at a time before
|
||||
agents. Take backups and confirm cluster health at each step; this playbook
|
||||
does not automate the upgrade, so those remain manual operational steps. See
|
||||
[K3s manual upgrades](https://docs.k3s.io/upgrades/manual) and the
|
||||
[v1.34 release notes](https://docs.k3s.io/release-notes/v1.34.X).
|
||||
- **Cilium**: upstream supports only consecutive minor upgrades. Update to the
|
||||
latest patch of the current minor, then upgrade 1.17, 1.18, 1.19, and 1.20 in
|
||||
order, reading each version's upgrade notes and running preflight checks.
|
||||
Do not attempt a direct upgrade from an old Cilium to 1.20.
|
||||
- **Calico**: starting with 3.28 the v3 resource UID behavior changed. If you
|
||||
have operators with OwnerReferences pointing to `projectcalico.org/v3`
|
||||
resources, remove and recreate those references around an in-place upgrade.
|
||||
- **MetalLB**: this project installs application tag `v0.16.0`. A newer
|
||||
chart-only tag such as `metallb-chart-0.16.1` is not an application or image
|
||||
release and must not be used as the controller or speaker image tag.
|
||||
|
||||
## ⚙️ Kube Config
|
||||
|
||||
To copy your `kube config` locally so that you can access your **Kubernetes** cluster run:
|
||||
|
||||
```bash
|
||||
scp debian@master_ip:~/.kube/config ~/.kube/config
|
||||
scp debian@master_ip:/etc/rancher/k3s/k3s.yaml ~/.kube/config
|
||||
```
|
||||
If the copy fails with a permission error, grant the SSH user temporary read access using the least permissive method
|
||||
available for the target system. Restore the original ownership and permissions immediately after copying. Avoid
|
||||
world-writable permissions on the kubeconfig because it contains cluster credentials.
|
||||
|
||||
For example, copy the file to a temporary user-readable path from the control node:
|
||||
|
||||
```bash
|
||||
ssh debian@master_ip 'sudo install -o "$(id -un)" -m 0600 /etc/rancher/k3s/k3s.yaml /tmp/k3s.yaml'
|
||||
```
|
||||
|
||||
Copy `/tmp/k3s.yaml`, then remove the temporary remote copy:
|
||||
|
||||
```bash
|
||||
scp debian@master_ip:/tmp/k3s.yaml ~/.kube/config
|
||||
ssh debian@master_ip rm -f /tmp/k3s.yaml
|
||||
```
|
||||
|
||||
You'll then want to modify the config to point to master IP by running:
|
||||
```bash
|
||||
sudo nano ~/.kube/config
|
||||
```
|
||||
Then change `server: https://127.0.0.1:6443` to match your master IP: `server: https://192.168.1.222:6443`
|
||||
|
||||
### 🔨 Testing your cluster
|
||||
|
||||
See the commands [here](https://docs.technotim.live/posts/k3s-etcd-ansible/#testing-your-cluster).
|
||||
See the commands [here](https://technotim.com/posts/k3s-etcd-ansible/#testing-your-cluster).
|
||||
|
||||
### Variables
|
||||
|
||||
| Role(s) | Variable | Type | Default | Required | Description |
|
||||
|---|---|---|---|---|---|
|
||||
| `download` | `k3s_version` | string | ❌ | Required | K3s binaries version |
|
||||
| `k3s_agent`, `k3s_server`, `k3s_server_post` | `apiserver_endpoint` | string | ❌ | Required | Virtual ip-address configured on each master |
|
||||
| `k3s_agent` | `extra_agent_args` | string | `null` | Not required | Extra arguments for agents nodes |
|
||||
| `k3s_agent`, `k3s_server` | `group_name_master` | string | `null` | Not required | Name of the master group |
|
||||
| `k3s_agent` | `k3s_token` | string | `null` | Not required | Token used to communicate between masters |
|
||||
| `k3s_agent`, `k3s_server` | `proxy_env` | dict | `null` | Not required | Internet proxy configurations |
|
||||
| `k3s_agent`, `k3s_server` | `proxy_env.HTTP_PROXY` | string | ❌ | Required | HTTP internet proxy |
|
||||
| `k3s_agent`, `k3s_server` | `proxy_env.HTTPS_PROXY` | string | ❌ | Required | HTTP internet proxy |
|
||||
| `k3s_agent`, `k3s_server` | `proxy_env.NO_PROXY` | string | ❌ | Required | Addresses that will not use the proxies |
|
||||
| `k3s_agent`, `k3s_server`, `reset` | `systemd_dir` | string | `/etc/systemd/system` | Not required | Path to systemd services |
|
||||
| `k3s_custom_registries` | `custom_registries_yaml` | string | ❌ | Required | YAML block defining custom registries. The following is an example that pulls all images used in this playbook through your private registries. It also allows you to pull your own images from your private registry, without having to use imagePullSecrets in your deployments. If all you need is your own images and you don't care about caching the docker/quay/ghcr.io images, you can just remove those from the mirrors: section. |
|
||||
| `k3s_server`, `k3s_server_post` | `cilium_bgp` | bool | `~` | Not required | Enable cilium BGP control plane for LB services and pod cidrs. Disables the use of MetalLB. |
|
||||
| `k3s_server`, `k3s_server_post` | `cilium_iface` | string | ❌ | Not required | The network interface used for when Cilium is enabled |
|
||||
| `k3s_server` | `extra_server_args` | string | `""` | Not required | Extra arguments for server nodes |
|
||||
| `k3s_server` | `k3s_create_kubectl_symlink` | bool | `false` | Not required | Create the kubectl -> k3s symlink |
|
||||
| `k3s_server` | `k3s_create_crictl_symlink` | bool | `true` | Not required | Create the crictl -> k3s symlink |
|
||||
| `k3s_server` | `kube_vip_arp` | bool | `true` | Not required | Enables kube-vip ARP broadcasts |
|
||||
| `k3s_server` | `kube_vip_bgp` | bool | `false` | Not required | Enables kube-vip BGP peering |
|
||||
| `k3s_server` | `kube_vip_bgp_routerid` | string | `"127.0.0.1"` | Not required | Defines the router ID for the kube-vip BGP server |
|
||||
| `k3s_server` | `kube_vip_bgp_as` | string | `"64513"` | Not required | Defines the AS for the kube-vip BGP server |
|
||||
| `k3s_server` | `kube_vip_bgp_peeraddress` | string | `"192.168.30.1"` | Not required | Defines the address for the kube-vip BGP peer |
|
||||
| `k3s_server` | `kube_vip_bgp_peeras` | string | `"64512"` | Not required | Defines the AS for the kube-vip BGP peer |
|
||||
| `k3s_server` | `kube_vip_bgp_peers` | list | `[]` | Not required | List of BGP peer ASN & address pairs |
|
||||
| `k3s_server` | `kube_vip_bgp_peers_groups` | list | `['k3s_master']` | Not required | Inventory group in which to search for additional `kube_vip_bgp_peers` parameters to merge. |
|
||||
| `k3s_server` | `kube_vip_iface` | string | `~` | Not required | Explicitly define an interface that ALL control nodes should use to propagate the VIP, define it here. Otherwise, kube-vip will determine the right interface automatically at runtime. |
|
||||
| `k3s_server` | `kube_vip_endpoint` | string | `~` | Not required | Overrides the internal address kube-vip binds/listens on, which can differ from the announced apiserver_endpoint for complex routing/tunnels. Defaults to apiserver_endpoint. |
|
||||
| `k3s_server` | `kube_vip_tag_version` | string | `v1.2.2` | Not required | Image tag for kube-vip |
|
||||
| `k3s_server` | `kube_vip_cloud_provider_tag_version` | string | `v0.0.12` | Not required | Tag for kube-vip-cloud-provider manifest when enable |
|
||||
| `k3s_server`, `k3_server_post` | `kube_vip_lb_ip_range` | string | `~` | Not required | IP range for kube-vip load balancer |
|
||||
| `k3s_server`, `k3s_server_post` | `metal_lb_controller_tag_version` | string | `v0.16.0` | Not required | Image tag for MetalLB |
|
||||
| `k3s_server` | `metal_lb_speaker_tag_version` | string | `v0.16.0` | Not required | Image tag for MetalLB |
|
||||
| `k3s_server` | `metal_lb_type` | string | `native` | Not required | Use FRR mode or native. Valid values are `frr` and `native` |
|
||||
| `k3s_server` | `retry_count` | int | `20` | Not required | Amount of retries when verifying that nodes joined |
|
||||
| `k3s_server` | `server_init_args` | string | ❌ | Not required | Arguments for server nodes |
|
||||
| `k3s_server_post` | `bpf_lb_algorithm` | string | `maglev` | Not required | BPF lb algorithm |
|
||||
| `k3s_server_post` | `bpf_lb_mode` | string | `hybrid` | Not required | BPF lb mode |
|
||||
| `k3s_server_post` | `calico_blocksize` | int | `26` | Not required | IP pool block size |
|
||||
| `k3s_server_post` | `calico_ebpf` | bool | `false` | Not required | Use eBPF dataplane instead of iptables |
|
||||
| `k3s_server_post` | `calico_encapsulation` | string | `VXLANCrossSubnet` | Not required | IP pool encapsulation |
|
||||
| `k3s_server_post` | `calico_natOutgoing` | string | `Enabled` | Not required | IP pool NAT outgoing |
|
||||
| `k3s_server_post` | `calico_nodeSelector` | string | `all()` | Not required | IP pool node selector |
|
||||
| `k3s_server_post` | `calico_iface` | string | `~` | Not required | The network interface used for when Calico is enabled |
|
||||
| `k3s_server_post` | `calico_tag` | string | `v3.32.1` | Not required | Calico version tag |
|
||||
| `k3s_server_post` | `cilium_bgp_my_asn` | int | `64513` | Not required | Local ASN for BGP peer |
|
||||
| `k3s_server_post` | `cilium_bgp_peer_asn` | int | `64512` | Not required | BGP peer ASN |
|
||||
| `k3s_server_post` | `cilium_bgp_peer_address` | string | `~` | Not required | BGP peer address |
|
||||
| `k3s_server_post` | `cilium_bgp_neighbors` | list | `[]` | Not required | List of BGP peer ASN & address pairs |
|
||||
| `k3s_server_post` | `cilium_bgp_neighbors_groups` | list | `['k3s_all']` | Not required | Inventory group in which to search for additional `cilium_bgp_neighbors` parameters to merge. |
|
||||
| `k3s_server_post` | `cilium_bgp_lb_cidr` | string | `192.168.31.0/24` | Not required | BGP load balancer IP range |
|
||||
| `k3s_server_post` | `cilium_exportPodCIDR` | bool | `true` | Not required | Export pod CIDR |
|
||||
| `k3s_server_post` | `cilium_hubble` | bool | `true` | Not required | Enable Cilium Hubble |
|
||||
| `k3s_server_post` | `cilium_mode` | string | `native` | Not required | Inner-node communication mode (choices are `native` and `tunnel`; `routed` is a deprecated alias for `tunnel`) |
|
||||
| `k3s_server_post` | `cilium_tag` | string | `v1.20.0` | Not required | Cilium version tag |
|
||||
| `k3s_server_post` | `cilium_cli_tag` | string | `v0.19.7` | Not required | Cilium CLI version tag |
|
||||
| `k3s_server_post` | `cluster_cidr` | string | `10.52.0.0/16` | Not required | Inner-cluster IP range |
|
||||
| `k3s_server_post` | `enable_bpf_masquerade` | bool | `true` | Not required | Use IP masquerading |
|
||||
| `k3s_server_post` | `kube_proxy_replacement` | bool | `true` | Not required | Replace the native kube-proxy with Cilium |
|
||||
| `k3s_server_post` | `metal_lb_available_timeout` | string | `240s` | Not required | Wait for MetalLB resources |
|
||||
| `k3s_server_post` | `metal_lb_ip_range` | string | `192.168.30.80-192.168.30.90` | Not required | MetalLB ip range for load balancer |
|
||||
| `k3s_server_post` | `metal_lb_controller_tag_version` | string | `v0.16.0` | Not required | Image tag for MetalLB |
|
||||
| `k3s_server_post` | `metal_lb_mode` | string | `layer2` | Not required | Metallb mode (choices are `bgp` and `layer2`) |
|
||||
| `k3s_server_post` | `metal_lb_bgp_my_asn` | string | `~` | Not required | BGP ASN configurations |
|
||||
| `k3s_server_post` | `metal_lb_bgp_peer_asn` | string | `~` | Not required | BGP peer ASN configurations |
|
||||
| `k3s_server_post` | `metal_lb_bgp_peer_address` | string | `~` | Not required | BGP peer address |
|
||||
| `lxc` | `custom_reboot_command` | string | `~` | Not required | Command to run on reboot |
|
||||
| `reboot` (playbook) | `concurrent_reboots` | int/string | `100%` | Not required | Number (or percentage) of nodes to reboot at a time for a staggered reboot |
|
||||
| `reboot` (playbook) | `wait_seconds_after_reboot` | int | `0` | Not required | Pause in seconds between staggered reboot batches |
|
||||
| `prereq` | `system_timezone` | string | `null` | Not required | Timezone to be set on all nodes |
|
||||
| `prereq` | `disable_swap` | bool | `true` | Not required | Disable swap on all cluster nodes (swapoff + comment out /etc/fstab swap entries), all-or-nothing |
|
||||
| `proxmox_lxc`, `reset_proxmox_lxc` | `proxmox_lxc_ct_ids` | list | ❌ | Required | Proxmox container ID list |
|
||||
| `raspberrypi` | `state` | string | `present` | Not required | Indicates whether the k3s prerequisites for Raspberry Pi should be set up (possible values are `present` and `absent`) |
|
||||
|
||||
|
||||
### Troubleshooting
|
||||
|
||||
Be sure to see [this post](https://github.com/techno-tim/k3s-ansible/discussions/20) on how to troubleshoot common problems
|
||||
Be sure to see [this post](https://github.com/timothystewart6/k3s-ansible/discussions/20) on how to troubleshoot common problems
|
||||
|
||||
### Testing the playbook using molecule
|
||||
|
||||
@@ -112,9 +278,33 @@ It is run automatically in CI, but you can also run the tests locally.
|
||||
This might be helpful for quick feedback in a few cases.
|
||||
You can find more information about it [here](molecule/README.md).
|
||||
|
||||
### Pre-commit Hooks
|
||||
### Pre-commit hooks
|
||||
|
||||
This repo uses `pre-commit` and `pre-commit-hooks` to lint and fix common style and syntax errors. Be sure to install python packages and then run `pre-commit install`. For more information, see [pre-commit](https://pre-commit.com/)
|
||||
This repository uses `pre-commit` to check style, syntax, Ansible content, and shell scripts. Install the Python
|
||||
dependencies, run `pre-commit install` once, and run `pre-commit run --all-files` before submitting a change. See
|
||||
[CONTRIBUTING.md](CONTRIBUTING.md) for the complete development workflow.
|
||||
|
||||
## 🌌 Ansible Galaxy
|
||||
|
||||
This collection can now be used in larger ansible projects.
|
||||
|
||||
Instructions:
|
||||
|
||||
- create or modify a file `collections/requirements.yml` in your project
|
||||
|
||||
```yml
|
||||
collections:
|
||||
- name: ansible.utils
|
||||
- name: community.general
|
||||
- name: ansible.posix
|
||||
- name: kubernetes.core
|
||||
- name: https://github.com/timothystewart6/k3s-ansible.git
|
||||
type: git
|
||||
version: master
|
||||
```
|
||||
|
||||
- install via `ansible-galaxy collection install -r ./collections/requirements.yml`
|
||||
- every role is now available via the prefix `techno_tim.k3s_ansible.` e.g. `techno_tim.k3s_ansible.lxc`
|
||||
|
||||
## Thanks 🤝
|
||||
|
||||
|
||||
-23
@@ -1,23 +0,0 @@
|
||||
[defaults]
|
||||
nocows = True
|
||||
roles_path = ./roles
|
||||
inventory = ./hosts.ini
|
||||
stdout_callback = yaml
|
||||
|
||||
remote_tmp = $HOME/.ansible/tmp
|
||||
local_tmp = $HOME/.ansible/tmp
|
||||
timeout = 60
|
||||
host_key_checking = False
|
||||
deprecation_warnings = False
|
||||
callbacks_enabled = profile_tasks
|
||||
log_path = ./ansible.log
|
||||
|
||||
[privilege_escalation]
|
||||
become = True
|
||||
|
||||
[ssh_connection]
|
||||
scp_if_ssh = smart
|
||||
retries = 3
|
||||
ssh_args = -o ControlMaster=auto -o ControlPersist=30m -o Compression=yes -o ServerAliveInterval=15s
|
||||
pipelining = True
|
||||
control_path = %(directory)s/%%h-%%r
|
||||
@@ -0,0 +1,2 @@
|
||||
[defaults]
|
||||
inventory = inventory/my-cluster/hosts.ini ; Adapt this to the path to your inventory file
|
||||
@@ -1,3 +1,3 @@
|
||||
#!/bin/bash
|
||||
|
||||
ansible-playbook site.yml -i inventory/my-cluster/hosts.ini
|
||||
ansible-playbook site.yml
|
||||
|
||||
+81
@@ -0,0 +1,81 @@
|
||||
### REQUIRED
|
||||
# The namespace of the collection. This can be a company/brand/organization or product namespace under which all
|
||||
# content lives. May only contain alphanumeric lowercase characters and underscores. Namespaces cannot start with
|
||||
# underscores or numbers and cannot contain consecutive underscores
|
||||
namespace: techno_tim
|
||||
|
||||
# The name of the collection. Has the same character restrictions as 'namespace'
|
||||
name: k3s_ansible
|
||||
|
||||
# The version of the collection. Must be compatible with semantic versioning
|
||||
version: 1.0.0
|
||||
|
||||
# The path to the Markdown (.md) readme file. This path is relative to the root of the collection
|
||||
readme: README.md
|
||||
|
||||
# A list of the collection's content authors. Can be just the name or in the format 'Full Name <email> (url)
|
||||
# @nicks:irc/im.site#channel'
|
||||
authors:
|
||||
- your name <example@domain.com>
|
||||
|
||||
|
||||
### OPTIONAL but strongly recommended
|
||||
# A short summary description of the collection
|
||||
description: >
|
||||
The easiest way to bootstrap a self-hosted High Availability Kubernetes
|
||||
cluster. A fully automated HA k3s etcd install with kube-vip, MetalLB,
|
||||
and more.
|
||||
|
||||
# Either a single license or a list of licenses for content inside of a collection. Ansible Galaxy currently only
|
||||
# accepts L(SPDX,https://spdx.org/licenses/) licenses. This key is mutually exclusive with 'license_file'
|
||||
license:
|
||||
- Apache-2.0
|
||||
|
||||
|
||||
# A list of tags you want to associate with the collection for indexing/searching. A tag name has the same character
|
||||
# requirements as 'namespace' and 'name'
|
||||
tags:
|
||||
- etcd
|
||||
- high-availability
|
||||
- k8s
|
||||
- k3s
|
||||
- k3s-cluster
|
||||
- kube-vip
|
||||
- kubernetes
|
||||
- metallb
|
||||
- rancher
|
||||
|
||||
# Collections that this collection requires to be installed for it to be usable. The key of the dict is the
|
||||
# collection label 'namespace.name'. The value is a version range
|
||||
# L(specifiers,https://python-semanticversion.readthedocs.io/en/latest/#requirement-specification). Multiple version
|
||||
# range specifiers can be set and are separated by ','
|
||||
dependencies:
|
||||
ansible.utils: '*'
|
||||
ansible.posix: '*'
|
||||
community.general: '*'
|
||||
kubernetes.core: '*'
|
||||
|
||||
# The URL of the originating SCM repository
|
||||
repository: https://github.com/timothystewart6/k3s-ansible
|
||||
|
||||
# The URL to any online docs
|
||||
documentation: https://github.com/timothystewart6/k3s-ansible
|
||||
|
||||
# The URL to the homepage of the collection/project
|
||||
homepage: https://www.youtube.com/watch?v=CbkEWcUZ7zM
|
||||
|
||||
# The URL to the collection issue tracker
|
||||
issues: https://github.com/timothystewart6/k3s-ansible/issues
|
||||
|
||||
# A list of file glob-like patterns used to filter any files or directories that should not be included in the build
|
||||
# artifact. A pattern is matched from the relative path of the file or directory of the collection directory. This
|
||||
# uses 'fnmatch' to match the files or directories. Some directories and files like 'galaxy.yml', '*.pyc', '*.retry',
|
||||
# and '.git' are always filtered. Mutually exclusive with 'manifest'
|
||||
build_ignore: []
|
||||
|
||||
# A dict controlling use of manifest directives used in building the collection artifact. The key 'directives' is a
|
||||
# list of MANIFEST.in style
|
||||
# L(directives,https://packaging.python.org/en/latest/guides/using-manifest-in/#manifest-in-commands). The key
|
||||
# 'omit_default_directives' is a boolean that controls whether the default directives are used. Mutually exclusive
|
||||
# with 'build_ignore'
|
||||
# manifest: null
|
||||
@@ -1,51 +1,207 @@
|
||||
---
|
||||
k3s_version: v1.24.8+k3s1
|
||||
k3s_version: v1.36.2+k3s1
|
||||
# this is the user that has ssh access to these machines
|
||||
ansible_user: ansibleuser
|
||||
systemd_dir: /etc/systemd/system
|
||||
|
||||
# Set your timezone
|
||||
system_timezone: "Your/Timezone"
|
||||
system_timezone: Your/Timezone
|
||||
|
||||
# k3s recommends swap be disabled on every cluster node. Applied uniformly to all
|
||||
# nodes (all-or-nothing) in the prereq role. Set to false to leave swap enabled.
|
||||
disable_swap: true
|
||||
|
||||
# interface which will be used for flannel
|
||||
flannel_iface: "eth0"
|
||||
# Defaults to each host's default IPv4 interface (e.g. eth0, enp1s0, ens3)
|
||||
# so KVM/cloud hosts without eth0 work out of the box. Override per-host if needed.
|
||||
flannel_iface: "{{ ansible_facts.default_ipv4.interface }}"
|
||||
|
||||
# apiserver_endpoint is virtual ip-address which will be configured on each master
|
||||
apiserver_endpoint: "192.168.30.222"
|
||||
# uncomment calico_iface to use tigera operator/calico cni instead of flannel https://docs.tigera.io/calico/latest/about
|
||||
# calico_iface: "{{ ansible_facts.default_ipv4.interface }}"
|
||||
calico_ebpf: false # use eBPF dataplane instead of iptables
|
||||
calico_tag: v3.32.1 # calico version tag
|
||||
|
||||
# uncomment cilium_iface to use cilium cni instead of flannel or calico
|
||||
# ensure v4.19.57, v5.1.16, v5.2.0 or more recent kernel
|
||||
# cilium_iface: "{{ ansible_facts.default_ipv4.interface }}"
|
||||
cilium_mode: native # native when nodes are on the same subnet or use BGP, otherwise set tunnel
|
||||
cilium_tag: v1.20.0 # cilium version tag
|
||||
cilium_cli_tag: v0.19.7 # cilium cli version tag
|
||||
cilium_hubble: true # enable hubble observability relay and ui
|
||||
cilium_envoy: true # enable the Envoy proxy for Cilium L7 policies
|
||||
|
||||
# disable cilium_envoy to skip the Envoy proxy entirely (e.g. no L7 policies)
|
||||
# cilium_envoy: false
|
||||
|
||||
# if using calico or cilium, you may specify the cluster pod cidr pool
|
||||
cluster_cidr: 10.52.0.0/16
|
||||
|
||||
# enable cilium bgp control plane for lb services and pod cidrs. disables metallb.
|
||||
cilium_bgp: false
|
||||
|
||||
# bgp parameters for cilium cni. only active when cilium_iface is defined and cilium_bgp is true.
|
||||
cilium_bgp_my_asn: "64513"
|
||||
cilium_bgp_peer_asn: "64512"
|
||||
cilium_bgp_peer_address: 192.168.30.1
|
||||
cilium_bgp_lb_cidr: 192.168.31.0/24 # cidr for cilium loadbalancer ipam
|
||||
|
||||
# enable kube-vip ARP broadcasts
|
||||
kube_vip_arp: true
|
||||
|
||||
# (optional) overrides the address kube-vip binds/listens on internally, which
|
||||
# can differ from the announced apiserver_endpoint for complex routing/tunnels.
|
||||
# Defaults to apiserver_endpoint. Also used to derive the kube-vip subnet.
|
||||
# kube_vip_endpoint: 10.66.1.5
|
||||
|
||||
# enable kube-vip BGP peering
|
||||
kube_vip_bgp: false
|
||||
|
||||
# bgp parameters for kube-vip
|
||||
kube_vip_bgp_routerid: "127.0.0.1" # Defines the router ID for the BGP server
|
||||
kube_vip_bgp_as: "64513" # Defines the AS for the BGP server
|
||||
kube_vip_bgp_peeraddress: "192.168.30.1" # Defines the address for the BGP peer
|
||||
kube_vip_bgp_peeras: "64512" # Defines the AS for the BGP peer
|
||||
|
||||
# apiserver_endpoint is virtual ip-address which will be configured on each master.
|
||||
# This must be a free, routable IP on your network (not already assigned to a host
|
||||
# or service), and is used by kube-vip / MetalLB to expose the Kubernetes API.
|
||||
apiserver_endpoint: 192.168.30.222
|
||||
|
||||
# k3s_token is required masters can talk together securely
|
||||
# this token should be alpha numeric only
|
||||
k3s_token: "some-SUPER-DEDEUPER-secret-password"
|
||||
k3s_token: some-SUPER-DEDEUPER-secret-password
|
||||
|
||||
# The IP on which the node is reachable in the cluster.
|
||||
# Here, a sensible default is provided, you can still override
|
||||
# it for each of your hosts, though.
|
||||
k3s_node_ip: '{{ ansible_facts[flannel_iface]["ipv4"]["address"] }}'
|
||||
k3s_node_ip: "{{ ansible_facts[(cilium_iface | default(calico_iface | default(flannel_iface)))]['ipv4']['address'] }}"
|
||||
|
||||
# Disable the taint manually by setting: k3s_master_taint = false
|
||||
k3s_master_taint: "{{ true if groups['node'] | default([]) | length >= 1 else false }}"
|
||||
|
||||
# these arguments are recommended for servers as well as agents:
|
||||
extra_args: >-
|
||||
--flannel-iface={{ flannel_iface }}
|
||||
{{ '--flannel-iface=' + flannel_iface if calico_iface is not defined and cilium_iface is not defined else '' }}
|
||||
--node-ip={{ k3s_node_ip }}
|
||||
|
||||
# change these to your liking, the only required are: --disable servicelb, --tls-san {{ apiserver_endpoint }}
|
||||
# the contents of the if block is also required if using calico or cilium
|
||||
extra_server_args: >-
|
||||
{{ extra_args }}
|
||||
{{ '--node-taint node-role.kubernetes.io/master=true:NoSchedule' if k3s_master_taint else '' }}
|
||||
{% if calico_iface is defined or cilium_iface is defined %}
|
||||
--flannel-backend=none
|
||||
--disable-network-policy
|
||||
--cluster-cidr={{ cluster_cidr | default('10.52.0.0/16') }}
|
||||
{% endif %}
|
||||
--tls-san {{ apiserver_endpoint }}
|
||||
--disable servicelb
|
||||
--disable traefik
|
||||
|
||||
extra_agent_args: >-
|
||||
{{ extra_args }}
|
||||
|
||||
# image tag for kube-vip
|
||||
kube_vip_tag_version: "v0.5.7"
|
||||
kube_vip_tag_version: v1.2.2
|
||||
|
||||
# tag for kube-vip-cloud-provider manifest
|
||||
# kube_vip_cloud_provider_tag_version: "v0.0.12"
|
||||
|
||||
# kube-vip ip range for load balancer
|
||||
# (uncomment to use kube-vip for services instead of MetalLB)
|
||||
# kube_vip_lb_ip_range: "192.168.30.80-192.168.30.90"
|
||||
|
||||
# metallb type frr or native
|
||||
metal_lb_type: native
|
||||
|
||||
# metallb mode layer2 or bgp
|
||||
metal_lb_mode: layer2
|
||||
|
||||
# bgp options
|
||||
# metal_lb_bgp_my_asn: "64513"
|
||||
# metal_lb_bgp_peer_asn: "64512"
|
||||
# metal_lb_bgp_peer_address: "192.168.30.1"
|
||||
|
||||
# image tag for metal lb
|
||||
metal_lb_speaker_tag_version: "v0.13.7"
|
||||
metal_lb_controller_tag_version: "v0.13.7"
|
||||
metal_lb_speaker_tag_version: v0.16.0
|
||||
metal_lb_controller_tag_version: v0.16.0
|
||||
|
||||
# metallb ip range for load balancer
|
||||
metal_lb_ip_range: "192.168.30.80-192.168.30.90"
|
||||
metal_lb_ip_range: 192.168.30.80-192.168.30.90
|
||||
|
||||
# (optional) limit MetalLB layer2 announcements to specific network interfaces.
|
||||
# Leave empty (default) to announce on all interfaces.
|
||||
# metal_lb_interfaces:
|
||||
# - eth1
|
||||
# - eth2
|
||||
|
||||
# Only enable if your nodes are proxmox LXC nodes, make sure to configure your proxmox nodes
|
||||
# in your hosts.ini file.
|
||||
# Please read https://gist.github.com/triangletodd/02f595cd4c0dc9aac5f7763ca2264185 before using this.
|
||||
# Most notably, your containers must be privileged, and must not have nesting set to true.
|
||||
# Please note this script disables most of the security of lxc containers, with the trade off being that lxc
|
||||
# containers are significantly more resource efficient compared to full VMs.
|
||||
# Mixing and matching VMs and lxc containers is not supported, ymmv if you want to do this.
|
||||
# I would only really recommend using this if you have particularly low powered proxmox nodes where the overhead of
|
||||
# VMs would use a significant portion of your available resources.
|
||||
proxmox_lxc_configure: false
|
||||
# the user that you would use to ssh into the host, for example if you run ssh some-user@my-proxmox-host,
|
||||
# set this value to some-user
|
||||
proxmox_lxc_ssh_user: root
|
||||
# the unique proxmox ids for all of the containers in the cluster, both worker and master nodes
|
||||
proxmox_lxc_ct_ids:
|
||||
- 200
|
||||
- 201
|
||||
- 202
|
||||
- 203
|
||||
- 204
|
||||
|
||||
# Only enable this if you have set up your own container registry to act as a mirror / pull-through cache
|
||||
# (harbor / nexus / docker's official registry / etc).
|
||||
# Can be beneficial for larger dev/test environments (for example if you're getting rate limited by docker hub),
|
||||
# or air-gapped environments where your nodes don't have internet access after the initial setup
|
||||
# (which is still needed for downloading the k3s binary and such).
|
||||
# k3s's documentation about private registries here: https://docs.k3s.io/installation/private-registry
|
||||
custom_registries: false
|
||||
# The registries can be authenticated or anonymous, depending on your registry server configuration.
|
||||
# If they allow anonymous access, simply remove the following bit from custom_registries_yaml
|
||||
# configs:
|
||||
# "registry.domain.com":
|
||||
# auth:
|
||||
# username: yourusername
|
||||
# password: yourpassword
|
||||
# The following is an example that pulls all images used in this playbook through your private registries.
|
||||
# It also allows you to pull your own images from your private registry, without having to use imagePullSecrets
|
||||
# in your deployments.
|
||||
# If all you need is your own images and you don't care about caching the docker/quay/ghcr.io images,
|
||||
# you can just remove those from the mirrors: section.
|
||||
custom_registries_yaml: |
|
||||
mirrors:
|
||||
docker.io:
|
||||
endpoint:
|
||||
- "https://registry.domain.com/v2/dockerhub"
|
||||
quay.io:
|
||||
endpoint:
|
||||
- "https://registry.domain.com/v2/quayio"
|
||||
ghcr.io:
|
||||
endpoint:
|
||||
- "https://registry.domain.com/v2/ghcrio"
|
||||
registry.domain.com:
|
||||
endpoint:
|
||||
- "https://registry.domain.com"
|
||||
|
||||
configs:
|
||||
"registry.domain.com":
|
||||
auth:
|
||||
username: yourusername
|
||||
password: yourpassword
|
||||
|
||||
# On some distros like Diet Pi, there is no dbus installed. dbus required by the default reboot command.
|
||||
# Uncomment if you need a custom reboot command
|
||||
# custom_reboot_command: /usr/sbin/shutdown -r now
|
||||
|
||||
# Only enable and configure these if you access the internet through a proxy
|
||||
# proxy_env:
|
||||
# HTTP_PROXY: "http://proxy.domain.local:3128"
|
||||
# HTTPS_PROXY: "http://proxy.domain.local:3128"
|
||||
# NO_PROXY: "*.domain.local,127.0.0.0/8,10.0.0.0/8,172.16.0.0/12,192.168.0.0/16"
|
||||
|
||||
@@ -0,0 +1,2 @@
|
||||
---
|
||||
ansible_user: "{{ proxmox_lxc_ssh_user }}"
|
||||
@@ -7,6 +7,11 @@
|
||||
192.168.30.41
|
||||
192.168.30.42
|
||||
|
||||
# only required if proxmox_lxc_configure: true
|
||||
# must contain all proxmox instances that have a master or worker node
|
||||
# [proxmox]
|
||||
# 192.168.30.43
|
||||
|
||||
[k3s_cluster:children]
|
||||
master
|
||||
node
|
||||
|
||||
@@ -13,6 +13,12 @@ We have these scenarios:
|
||||
To save a bit of test time, this cluster is _not_ highly available, it consists of only one control and one worker node.
|
||||
- **single_node**:
|
||||
Very similar to the default scenario, but uses only a single node for all cluster functionality.
|
||||
- **calico**:
|
||||
The same as single node, but uses calico cni instead of flannel.
|
||||
- **cilium**:
|
||||
The same as single node, but uses cilium cni instead of flannel.
|
||||
- **kube-vip**
|
||||
The same as single node, but uses kube-vip as service loadbalancer instead of MetalLB
|
||||
|
||||
## How to execute
|
||||
|
||||
|
||||
@@ -0,0 +1,44 @@
|
||||
---
|
||||
dependency:
|
||||
name: galaxy
|
||||
driver:
|
||||
name: vagrant
|
||||
platforms:
|
||||
- name: control1
|
||||
box: bento/ubuntu-26.04
|
||||
memory: 4096
|
||||
cpus: 4
|
||||
groups:
|
||||
- k3s_cluster
|
||||
- master
|
||||
interfaces:
|
||||
- network_name: private_network
|
||||
ip: 192.168.30.62
|
||||
provisioner:
|
||||
name: ansible
|
||||
env:
|
||||
ANSIBLE_VERBOSITY: 1
|
||||
playbooks:
|
||||
converge: ../resources/converge.yml
|
||||
side_effect: ../resources/reset.yml
|
||||
verify: ../resources/verify.yml
|
||||
inventory:
|
||||
links:
|
||||
group_vars: ../../inventory/sample/group_vars
|
||||
scenario:
|
||||
test_sequence:
|
||||
- dependency
|
||||
- cleanup
|
||||
- destroy
|
||||
- syntax
|
||||
- create
|
||||
- prepare
|
||||
- converge
|
||||
# idempotence is not possible with the playbook in its current form.
|
||||
- verify
|
||||
# We are repurposing side_effect here to test the reset playbook.
|
||||
# This is why we do not run it before verify (which tests the cluster),
|
||||
# but after the verify step.
|
||||
- side_effect
|
||||
- cleanup
|
||||
- destroy
|
||||
@@ -0,0 +1,18 @@
|
||||
---
|
||||
- name: Apply overrides
|
||||
hosts: all
|
||||
serial: 1
|
||||
tasks:
|
||||
- name: Override host variables
|
||||
ansible.builtin.set_fact:
|
||||
# See:
|
||||
# https://github.com/flannel-io/flannel/blob/67d603aaf45ef80f5dd39f43714fc5e6f8a637eb/Documentation/troubleshooting.md#Vagrant
|
||||
calico_iface: "{{ 'eth1' if 'eth1' in ansible_facts.interfaces else 'enp0s8' }}"
|
||||
kube_vip_iface: "{{ 'eth1' if 'eth1' in ansible_facts.interfaces else 'enp0s8' }}"
|
||||
|
||||
# The test VMs might be a bit slow, so we give them more time to join the cluster:
|
||||
retry_count: 45
|
||||
|
||||
# Make sure that our IP ranges do not collide with those of the other scenarios
|
||||
apiserver_endpoint: 192.168.30.224
|
||||
metal_lb_ip_range: 192.168.30.100-192.168.30.109
|
||||
@@ -0,0 +1,6 @@
|
||||
---
|
||||
# Durable verify inputs for the calico (Calico CNI + MetalLB) scenario.
|
||||
verify_cni: calico
|
||||
verify_lb: metallb
|
||||
verify_lb_ip_range:
|
||||
- 192.168.30.100-192.168.30.109
|
||||
@@ -0,0 +1,44 @@
|
||||
---
|
||||
dependency:
|
||||
name: galaxy
|
||||
driver:
|
||||
name: vagrant
|
||||
platforms:
|
||||
- name: control1
|
||||
box: bento/ubuntu-26.04
|
||||
memory: 4096
|
||||
cpus: 4
|
||||
groups:
|
||||
- k3s_cluster
|
||||
- master
|
||||
interfaces:
|
||||
- network_name: private_network
|
||||
ip: 192.168.30.63
|
||||
provisioner:
|
||||
name: ansible
|
||||
env:
|
||||
ANSIBLE_VERBOSITY: 1
|
||||
playbooks:
|
||||
converge: ../resources/converge.yml
|
||||
side_effect: ../resources/reset.yml
|
||||
verify: ../resources/verify.yml
|
||||
inventory:
|
||||
links:
|
||||
group_vars: ../../inventory/sample/group_vars
|
||||
scenario:
|
||||
test_sequence:
|
||||
- dependency
|
||||
- cleanup
|
||||
- destroy
|
||||
- syntax
|
||||
- create
|
||||
- prepare
|
||||
- converge
|
||||
# idempotence is not possible with the playbook in its current form.
|
||||
- verify
|
||||
# We are repurposing side_effect here to test the reset playbook.
|
||||
# This is why we do not run it before verify (which tests the cluster),
|
||||
# but after the verify step.
|
||||
- side_effect
|
||||
- cleanup
|
||||
- destroy
|
||||
@@ -0,0 +1,18 @@
|
||||
---
|
||||
- name: Apply overrides
|
||||
hosts: all
|
||||
serial: 1
|
||||
tasks:
|
||||
- name: Override host variables
|
||||
ansible.builtin.set_fact:
|
||||
# See:
|
||||
# https://github.com/flannel-io/flannel/blob/67d603aaf45ef80f5dd39f43714fc5e6f8a637eb/Documentation/troubleshooting.md#Vagrant
|
||||
cilium_iface: "{{ 'eth1' if 'eth1' in ansible_facts.interfaces else 'enp0s8' }}"
|
||||
kube_vip_iface: "{{ 'eth1' if 'eth1' in ansible_facts.interfaces else 'enp0s8' }}"
|
||||
|
||||
# The test VMs might be a bit slow, so we give them more time to join the cluster:
|
||||
retry_count: 45
|
||||
|
||||
# Make sure that our IP ranges do not collide with those of the other scenarios
|
||||
apiserver_endpoint: 192.168.30.225
|
||||
metal_lb_ip_range: 192.168.30.110-192.168.30.119
|
||||
@@ -0,0 +1,6 @@
|
||||
---
|
||||
# Durable verify inputs for the cilium (Cilium CNI + MetalLB) scenario.
|
||||
verify_cni: cilium
|
||||
verify_lb: metallb
|
||||
verify_lb_ip_range:
|
||||
- 192.168.30.110-192.168.30.119
|
||||
@@ -0,0 +1,80 @@
|
||||
---
|
||||
- name: Create
|
||||
hosts: localhost
|
||||
connection: local
|
||||
gather_facts: false
|
||||
no_log: "{{ molecule_no_log }}"
|
||||
vars:
|
||||
create_batches:
|
||||
- [control1, control2]
|
||||
- [control3, node1]
|
||||
- [node2]
|
||||
tasks:
|
||||
- name: Verify that bounded batches cover the configured platforms exactly once
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- create_batches | flatten | sort == molecule_yml.platforms | map(attribute='name') | sort
|
||||
- create_batches | flatten | length == create_batches | flatten | unique | length
|
||||
- create_batches | map('length') | max <= 2
|
||||
fail_msg: Bounded create batches do not match the configured default platforms.
|
||||
|
||||
- name: Generate the complete Vagrant configuration # noqa fqcn[action]
|
||||
vagrant:
|
||||
instances: "{{ molecule_yml.platforms }}"
|
||||
default_box: "{{ molecule_yml.driver.default_box | default('generic/alpine316') }}"
|
||||
provider_name: "{{ molecule_yml.driver.provider.name | default(omit, true) }}"
|
||||
provision: "{{ molecule_yml.driver.provision | default(omit) }}"
|
||||
cachier: "{{ molecule_yml.driver.cachier | default(omit) }}"
|
||||
parallel: false
|
||||
state: halt
|
||||
changed_when: false
|
||||
|
||||
- name: Start clean Vagrant guests in bounded batches
|
||||
ansible.builtin.command:
|
||||
argv: >-
|
||||
{{
|
||||
[playbook_dir + '/../../.github/scripts/vagrant-up-timed.sh',
|
||||
molecule_ephemeral_directory] + item
|
||||
}}
|
||||
loop: "{{ create_batches }}"
|
||||
loop_control:
|
||||
label: "{{ item | join(', ') }}"
|
||||
changed_when: true
|
||||
|
||||
- name: Reconcile all instances and collect their connection configuration # noqa fqcn[action]
|
||||
vagrant:
|
||||
instances: "{{ molecule_yml.platforms }}"
|
||||
default_box: "{{ molecule_yml.driver.default_box | default('generic/alpine316') }}"
|
||||
provider_name: "{{ molecule_yml.driver.provider.name | default(omit, true) }}"
|
||||
provision: "{{ molecule_yml.driver.provision | default(omit) }}"
|
||||
cachier: "{{ molecule_yml.driver.cachier | default(omit) }}"
|
||||
parallel: false
|
||||
state: up
|
||||
register: server
|
||||
no_log: false
|
||||
|
||||
- name: Populate instance configuration dictionaries
|
||||
ansible.builtin.set_fact:
|
||||
instance_conf_dict:
|
||||
instance: "{{ item.Host }}"
|
||||
address: "{{ item.HostName }}"
|
||||
user: "{{ item.User }}"
|
||||
port: "{{ item.Port }}"
|
||||
identity_file: "{{ item.IdentityFile }}"
|
||||
loop: "{{ server.results }}"
|
||||
register: instance_config_dict
|
||||
|
||||
- name: Convert instance configuration dictionaries to a list
|
||||
ansible.builtin.set_fact:
|
||||
instance_conf: >-
|
||||
{{
|
||||
instance_config_dict.results
|
||||
| map(attribute='ansible_facts.instance_conf_dict')
|
||||
| list
|
||||
}}
|
||||
|
||||
- name: Write Molecule instance configuration
|
||||
ansible.builtin.copy:
|
||||
content: "{{ instance_conf | to_json | from_json | to_yaml }}"
|
||||
dest: "{{ molecule_instance_config }}"
|
||||
mode: "0600"
|
||||
@@ -3,76 +3,83 @@ dependency:
|
||||
name: galaxy
|
||||
driver:
|
||||
name: vagrant
|
||||
# The Vagrant driver warns that parallel VirtualBox creation can cause
|
||||
# platform issues. Keep this five-node, mixed-distribution scenario serial.
|
||||
parallel: false
|
||||
platforms:
|
||||
|
||||
- name: control1
|
||||
box: generic/ubuntu2204
|
||||
memory: 2048
|
||||
box: bento/ubuntu-26.04
|
||||
memory: 1024
|
||||
cpus: 2
|
||||
groups:
|
||||
- k3s_cluster
|
||||
- master
|
||||
# Keep adapter 2 stable across linked-clone rebuilds so stale host-only
|
||||
# neighbor state still identifies the current scenario guest.
|
||||
provider_raw_config_args:
|
||||
- "customize ['modifyvm', :id, '--mac-address2', '080027A13038']"
|
||||
interfaces:
|
||||
- network_name: private_network
|
||||
ip: 192.168.30.38
|
||||
config_options:
|
||||
# We currently can not use public-key based authentication on Ubuntu 22.04,
|
||||
# see: https://github.com/chef/bento/issues/1405
|
||||
ssh.username: "vagrant"
|
||||
ssh.password: "vagrant"
|
||||
|
||||
- name: control2
|
||||
box: generic/debian11
|
||||
memory: 2048
|
||||
box: bento/debian-13
|
||||
memory: 1024
|
||||
cpus: 2
|
||||
groups:
|
||||
- k3s_cluster
|
||||
- master
|
||||
provider_raw_config_args:
|
||||
- "customize ['modifyvm', :id, '--mac-address2', '080027A13039']"
|
||||
interfaces:
|
||||
- network_name: private_network
|
||||
ip: 192.168.30.39
|
||||
|
||||
- name: control3
|
||||
box: generic/rocky9
|
||||
memory: 2048
|
||||
box: bento/rockylinux-10.1
|
||||
memory: 1024
|
||||
cpus: 2
|
||||
groups:
|
||||
- k3s_cluster
|
||||
- master
|
||||
provider_raw_config_args:
|
||||
- "customize ['modifyvm', :id, '--mac-address2', '080027A13040']"
|
||||
interfaces:
|
||||
- network_name: private_network
|
||||
ip: 192.168.30.40
|
||||
|
||||
- name: node1
|
||||
box: generic/ubuntu2204
|
||||
memory: 2048
|
||||
box: bento/ubuntu-26.04
|
||||
memory: 1024
|
||||
cpus: 2
|
||||
groups:
|
||||
- k3s_cluster
|
||||
- node
|
||||
provider_raw_config_args:
|
||||
- "customize ['modifyvm', :id, '--mac-address2', '080027A13041']"
|
||||
interfaces:
|
||||
- network_name: private_network
|
||||
ip: 192.168.30.41
|
||||
config_options:
|
||||
# We currently can not use public-key based authentication on Ubuntu 22.04,
|
||||
# see: https://github.com/chef/bento/issues/1405
|
||||
ssh.username: "vagrant"
|
||||
ssh.password: "vagrant"
|
||||
|
||||
- name: node2
|
||||
box: generic/rocky9
|
||||
memory: 2048
|
||||
box: bento/rockylinux-10.1
|
||||
memory: 1024
|
||||
cpus: 2
|
||||
groups:
|
||||
- k3s_cluster
|
||||
- node
|
||||
provider_raw_config_args:
|
||||
- "customize ['modifyvm', :id, '--mac-address2', '080027A13042']"
|
||||
interfaces:
|
||||
- network_name: private_network
|
||||
ip: 192.168.30.42
|
||||
|
||||
provisioner:
|
||||
name: ansible
|
||||
env:
|
||||
ANSIBLE_VERBOSITY: 1
|
||||
playbooks:
|
||||
create: create.yml
|
||||
converge: ../resources/converge.yml
|
||||
side_effect: ../resources/reset.yml
|
||||
verify: ../resources/verify.yml
|
||||
@@ -82,7 +89,6 @@ provisioner:
|
||||
scenario:
|
||||
test_sequence:
|
||||
- dependency
|
||||
- lint
|
||||
- cleanup
|
||||
- destroy
|
||||
- syntax
|
||||
|
||||
@@ -1,11 +1,16 @@
|
||||
---
|
||||
- name: Apply overrides
|
||||
hosts: all
|
||||
serial: 1
|
||||
tasks:
|
||||
- name: Override host variables
|
||||
ansible.builtin.set_fact:
|
||||
# See: https://github.com/flannel-io/flannel/blob/67d603aaf45ef80f5dd39f43714fc5e6f8a637eb/Documentation/troubleshooting.md#Vagrant # noqa yaml[line-length]
|
||||
flannel_iface: eth1
|
||||
# See:
|
||||
# https://github.com/flannel-io/flannel/blob/67d603aaf45ef80f5dd39f43714fc5e6f8a637eb/Documentation/troubleshooting.md#Vagrant
|
||||
flannel_iface: "{{ 'eth1' if 'eth1' in ansible_facts.interfaces else 'enp0s8' }}"
|
||||
# kube-vip cannot infer the cluster interface in these multi-NIC
|
||||
# Vagrant guests because the default route is on eth0.
|
||||
kube_vip_iface: "{{ 'eth1' if 'eth1' in ansible_facts.interfaces else 'enp0s8' }}"
|
||||
|
||||
# The test VMs might be a bit slow, so we give them more time to join the cluster:
|
||||
retry_count: 45
|
||||
|
||||
@@ -5,18 +5,137 @@
|
||||
|
||||
- name: Network setup
|
||||
hosts: all
|
||||
vars:
|
||||
primary_master: "{{ groups[group_name_master | default('master')][0] }}"
|
||||
primary_cluster_ip: >-
|
||||
{{ hostvars[primary_master].k3s_node_ip | split(',') | first }}
|
||||
cluster_interface: >-
|
||||
{{ cilium_iface | default(calico_iface | default(flannel_iface)) }}
|
||||
primary_cluster_interface: >-
|
||||
{{ hostvars[primary_master].cilium_iface
|
||||
| default(hostvars[primary_master].calico_iface
|
||||
| default(hostvars[primary_master].flannel_iface)) }}
|
||||
primary_cluster_mac: >-
|
||||
{{ hostvars[primary_master].ansible_facts[primary_cluster_interface].macaddress }}
|
||||
tasks:
|
||||
- name: Disable firewalld
|
||||
when: ansible_distribution == "Rocky"
|
||||
# Rocky Linux comes with firewalld enabled. It blocks some of the network
|
||||
# connections needed for our k3s cluster. For our test setup, we just disable
|
||||
# it since the VM host's firewall is still active for connections to and from
|
||||
# the Internet.
|
||||
- name: Gather service facts
|
||||
ansible.builtin.service_facts:
|
||||
|
||||
- name: Disable guest firewall services
|
||||
# The disposable test guests use an isolated VirtualBox network. A distro
|
||||
# firewall can allow ICMP while silently blocking the inter-node Kubernetes
|
||||
# API connection, so disable the known guest firewalls consistently.
|
||||
# When building your own cluster, please DO NOT blindly copy this. Instead,
|
||||
# please create a custom firewall configuration that fits your network design
|
||||
# and security needs.
|
||||
ansible.builtin.systemd:
|
||||
name: firewalld
|
||||
enabled: no
|
||||
name: "{{ item }}"
|
||||
enabled: false
|
||||
state: stopped
|
||||
become: true
|
||||
loop:
|
||||
- firewalld.service
|
||||
- nftables.service
|
||||
- ufw.service
|
||||
when: item in ansible_facts.services
|
||||
|
||||
- name: Verify the private cluster interface
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- flannel_iface in ansible_facts
|
||||
- ansible_facts[flannel_iface].ipv4 is defined
|
||||
- ansible_facts[flannel_iface].ipv4.address is defined
|
||||
fail_msg: >-
|
||||
The Vagrant private interface {{ flannel_iface }} does not have an
|
||||
IPv4 address on {{ inventory_hostname }}.
|
||||
|
||||
- name: Pin disposable cluster peer neighbor entries
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- ip
|
||||
- neigh
|
||||
- replace
|
||||
- "{{ peer_cluster_ip }}"
|
||||
- lladdr
|
||||
- "{{ peer_cluster_mac }}"
|
||||
- nud
|
||||
- permanent
|
||||
- dev
|
||||
- "{{ cluster_interface }}"
|
||||
become: true
|
||||
changed_when: false
|
||||
loop: "{{ groups['k3s_cluster'] }}"
|
||||
loop_control:
|
||||
label: "{{ inventory_hostname }} -> {{ item }}"
|
||||
vars:
|
||||
peer_cluster_interface: >-
|
||||
{{ hostvars[item].cilium_iface
|
||||
| default(hostvars[item].calico_iface
|
||||
| default(hostvars[item].flannel_iface)) }}
|
||||
peer_cluster_ip: >-
|
||||
{{ hostvars[item].k3s_node_ip | split(',') | first }}
|
||||
peer_cluster_mac: >-
|
||||
{{ hostvars[item].ansible_facts[peer_cluster_interface].macaddress }}
|
||||
when: item != inventory_hostname
|
||||
|
||||
- name: Verify guest-to-guest cluster network reachability
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- ping
|
||||
- -c
|
||||
- "1"
|
||||
- -W
|
||||
- "1"
|
||||
- "{{ primary_cluster_ip }}"
|
||||
register: primary_cluster_ping
|
||||
until: primary_cluster_ping.rc == 0
|
||||
retries: 6
|
||||
delay: 2
|
||||
changed_when: false
|
||||
|
||||
- name: Read the primary neighbor entry
|
||||
ansible.builtin.command:
|
||||
argv:
|
||||
- ip
|
||||
- neigh
|
||||
- show
|
||||
- to
|
||||
- "{{ primary_cluster_ip }}"
|
||||
- dev
|
||||
- "{{ cluster_interface }}"
|
||||
register: primary_cluster_neighbor
|
||||
changed_when: false
|
||||
when: inventory_hostname != primary_master
|
||||
|
||||
- name: Verify the primary neighbor identity
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- (primary_cluster_mac | lower) in (primary_cluster_neighbor.stdout | lower)
|
||||
fail_msg: >-
|
||||
{{ inventory_hostname }} resolved primary {{ primary_cluster_ip }} to
|
||||
an unexpected MAC on {{ cluster_interface }}. Expected
|
||||
{{ primary_cluster_mac }}, got: {{ primary_cluster_neighbor.stdout }}
|
||||
when: inventory_hostname != primary_master
|
||||
|
||||
- name: Verify GitHub release host DNS
|
||||
ansible.builtin.getent:
|
||||
database: hosts
|
||||
key: github.com
|
||||
register: github_dns
|
||||
retries: 6
|
||||
delay: 5
|
||||
until: github_dns is succeeded
|
||||
|
||||
- name: Verify k3s checksum URL is reachable
|
||||
ansible.builtin.uri:
|
||||
url: >-
|
||||
https://github.com/k3s-io/k3s/releases/download/{{ k3s_version
|
||||
}}/sha256sum-amd64.txt
|
||||
method: HEAD
|
||||
follow_redirects: safe
|
||||
status_code: [200, 302]
|
||||
timeout: 15
|
||||
register: k3s_checksum_request
|
||||
retries: 3
|
||||
delay: 5
|
||||
until: k3s_checksum_request.status in [200, 302]
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
---
|
||||
# Durable verify inputs for the default (flannel + MetalLB) scenario.
|
||||
# These are plain inventory vars linked into the shared Molecule inventory so
|
||||
# the verify play can see them even though the converge play's set_fact values
|
||||
# are not persisted between the two Ansible processes.
|
||||
verify_cni: flannel
|
||||
verify_lb: metallb
|
||||
verify_lb_ip_range:
|
||||
- 192.168.30.80-192.168.30.90
|
||||
@@ -0,0 +1,3 @@
|
||||
---
|
||||
node_ipv4: 192.168.123.12
|
||||
node_ipv6: fdad:bad:ba55::de:12
|
||||
+17
-16
@@ -4,10 +4,9 @@ dependency:
|
||||
driver:
|
||||
name: vagrant
|
||||
platforms:
|
||||
|
||||
- name: control1
|
||||
box: generic/ubuntu2204
|
||||
memory: 2048
|
||||
box: bento/ubuntu-26.04
|
||||
memory: 1024
|
||||
cpus: 2
|
||||
groups:
|
||||
- k3s_cluster
|
||||
@@ -15,15 +14,21 @@ platforms:
|
||||
interfaces:
|
||||
- network_name: private_network
|
||||
ip: fdad:bad:ba55::de:11
|
||||
config_options:
|
||||
# We currently can not use public-key based authentication on Ubuntu 22.04,
|
||||
# see: https://github.com/chef/bento/issues/1405
|
||||
ssh.username: "vagrant"
|
||||
ssh.password: "vagrant"
|
||||
|
||||
- name: control2
|
||||
box: bento/ubuntu-26.04
|
||||
memory: 1024
|
||||
cpus: 2
|
||||
groups:
|
||||
- k3s_cluster
|
||||
- master
|
||||
interfaces:
|
||||
- network_name: private_network
|
||||
ip: fdad:bad:ba55::de:12
|
||||
|
||||
- name: node1
|
||||
box: generic/ubuntu2204
|
||||
memory: 2048
|
||||
box: bento/ubuntu-26.04
|
||||
memory: 1024
|
||||
cpus: 2
|
||||
groups:
|
||||
- k3s_cluster
|
||||
@@ -31,13 +36,10 @@ platforms:
|
||||
interfaces:
|
||||
- network_name: private_network
|
||||
ip: fdad:bad:ba55::de:21
|
||||
config_options:
|
||||
# We currently can not use public-key based authentication on Ubuntu 22.04,
|
||||
# see: https://github.com/chef/bento/issues/1405
|
||||
ssh.username: "vagrant"
|
||||
ssh.password: "vagrant"
|
||||
provisioner:
|
||||
name: ansible
|
||||
env:
|
||||
ANSIBLE_VERBOSITY: 1
|
||||
playbooks:
|
||||
converge: ../resources/converge.yml
|
||||
side_effect: ../resources/reset.yml
|
||||
@@ -48,7 +50,6 @@ provisioner:
|
||||
scenario:
|
||||
test_sequence:
|
||||
- dependency
|
||||
- lint
|
||||
- cleanup
|
||||
- destroy
|
||||
- syntax
|
||||
|
||||
@@ -1,11 +1,18 @@
|
||||
---
|
||||
- name: Apply overrides
|
||||
hosts: all
|
||||
serial: 1
|
||||
tasks:
|
||||
- name: Override host variables (1/2)
|
||||
ansible.builtin.set_fact:
|
||||
# See: https://github.com/flannel-io/flannel/blob/67d603aaf45ef80f5dd39f43714fc5e6f8a637eb/Documentation/troubleshooting.md#Vagrant # noqa yaml[line-length]
|
||||
flannel_iface: eth1
|
||||
# See:
|
||||
# https://github.com/flannel-io/flannel/blob/67d603aaf45ef80f5dd39f43714fc5e6f8a637eb/Documentation/troubleshooting.md#Vagrant
|
||||
flannel_iface: "{{ 'eth1' if 'eth1' in ansible_facts.interfaces else 'enp0s8' }}"
|
||||
|
||||
# In this scenario, we have multiple interfaces that the VIP could be
|
||||
# broadcasted on. Since we have assigned a dedicated private network
|
||||
# here, let's make sure that it is used.
|
||||
kube_vip_iface: "{{ 'eth1' if 'eth1' in ansible_facts.interfaces else 'enp0s8' }}"
|
||||
|
||||
# The test VMs might be a bit slow, so we give them more time to join the cluster:
|
||||
retry_count: 45
|
||||
|
||||
@@ -38,7 +38,7 @@
|
||||
dest: /etc/netplan/55-flannel-ipv4.yaml
|
||||
owner: root
|
||||
group: root
|
||||
mode: 0644
|
||||
mode: "0644"
|
||||
register: netplan_template
|
||||
|
||||
- name: Apply netplan configuration
|
||||
|
||||
@@ -0,0 +1,7 @@
|
||||
---
|
||||
# Durable verify inputs for the ipv6 (flannel CNI + MetalLB) scenario.
|
||||
verify_cni: flannel
|
||||
verify_lb: metallb
|
||||
verify_lb_ip_range:
|
||||
- fdad:bad:ba55::1b:0/112
|
||||
- 192.168.123.80-192.168.123.90
|
||||
@@ -0,0 +1,44 @@
|
||||
---
|
||||
dependency:
|
||||
name: galaxy
|
||||
driver:
|
||||
name: vagrant
|
||||
platforms:
|
||||
- name: control1
|
||||
box: bento/ubuntu-26.04
|
||||
memory: 4096
|
||||
cpus: 4
|
||||
groups:
|
||||
- k3s_cluster
|
||||
- master
|
||||
interfaces:
|
||||
- network_name: private_network
|
||||
ip: 192.168.30.62
|
||||
provisioner:
|
||||
name: ansible
|
||||
env:
|
||||
ANSIBLE_VERBOSITY: 1
|
||||
playbooks:
|
||||
converge: ../resources/converge.yml
|
||||
side_effect: ../resources/reset.yml
|
||||
verify: ../resources/verify.yml
|
||||
inventory:
|
||||
links:
|
||||
group_vars: ../../inventory/sample/group_vars
|
||||
scenario:
|
||||
test_sequence:
|
||||
- dependency
|
||||
- cleanup
|
||||
- destroy
|
||||
- syntax
|
||||
- create
|
||||
- prepare
|
||||
- converge
|
||||
# idempotence is not possible with the playbook in its current form.
|
||||
- verify
|
||||
# We are repurposing side_effect here to test the reset playbook.
|
||||
# This is why we do not run it before verify (which tests the cluster),
|
||||
# but after the verify step.
|
||||
- side_effect
|
||||
- cleanup
|
||||
- destroy
|
||||
@@ -0,0 +1,19 @@
|
||||
---
|
||||
- name: Apply overrides
|
||||
hosts: all
|
||||
serial: 1
|
||||
tasks:
|
||||
- name: Override host variables
|
||||
ansible.builtin.set_fact:
|
||||
# See:
|
||||
# https://github.com/flannel-io/flannel/blob/67d603aaf45ef80f5dd39f43714fc5e6f8a637eb/Documentation/troubleshooting.md#Vagrant
|
||||
flannel_iface: "{{ 'eth1' if 'eth1' in ansible_facts.interfaces else 'enp0s8' }}"
|
||||
kube_vip_iface: "{{ 'eth1' if 'eth1' in ansible_facts.interfaces else 'enp0s8' }}"
|
||||
|
||||
# The test VMs might be a bit slow, so we give them more time to join the cluster:
|
||||
retry_count: 45
|
||||
|
||||
# Make sure that our IP ranges do not collide with those of the other scenarios
|
||||
apiserver_endpoint: 192.168.30.225
|
||||
# Use kube-vip instead of MetalLB
|
||||
kube_vip_lb_ip_range: 192.168.30.110-192.168.30.119
|
||||
@@ -0,0 +1,9 @@
|
||||
---
|
||||
# Durable verify inputs for the kube-vip (flannel CNI + kube-vip LB) scenario.
|
||||
verify_cni: flannel
|
||||
verify_lb: kube-vip
|
||||
# The kube-vip cloud provider tag is not defined in the linked sample group
|
||||
# vars (its sample entry is commented out), so it is supplied here.
|
||||
verify_kube_vip_cloud_provider_tag: v0.0.12
|
||||
verify_lb_ip_range:
|
||||
- 192.168.30.110-192.168.30.119
|
||||
@@ -1,5 +1,8 @@
|
||||
---
|
||||
- name: Verify
|
||||
hosts: all
|
||||
vars_files:
|
||||
- >-
|
||||
{{ lookup("ansible.builtin.env", "MOLECULE_SCENARIO_DIRECTORY") }}/verify-vars.yml
|
||||
roles:
|
||||
- verify/from_outside
|
||||
- verify_from_outside
|
||||
|
||||
@@ -1,58 +0,0 @@
|
||||
---
|
||||
- name: Deploy example
|
||||
block:
|
||||
- name: "Create namespace: {{ testing_namespace }}"
|
||||
kubernetes.core.k8s:
|
||||
api_version: v1
|
||||
kind: Namespace
|
||||
name: "{{ testing_namespace }}"
|
||||
state: present
|
||||
wait: true
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
|
||||
- name: Apply example manifests
|
||||
kubernetes.core.k8s:
|
||||
src: "{{ example_manifests_path }}/{{ item }}"
|
||||
namespace: "{{ testing_namespace }}"
|
||||
state: present
|
||||
wait: true
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
with_items:
|
||||
- deployment.yml
|
||||
- service.yml
|
||||
|
||||
- name: Get info about nginx service
|
||||
kubernetes.core.k8s_info:
|
||||
kind: service
|
||||
name: nginx
|
||||
namespace: "{{ testing_namespace }}"
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
vars: &load_balancer_metadata
|
||||
metallb_ip: status.loadBalancer.ingress[0].ip
|
||||
metallb_port: spec.ports[0].port
|
||||
register: nginx_services
|
||||
|
||||
- name: Assert that the nginx welcome page is available
|
||||
ansible.builtin.uri:
|
||||
url: http://{{ ip | ansible.utils.ipwrap }}:{{ port }}/
|
||||
return_content: yes
|
||||
register: result
|
||||
failed_when: "'Welcome to nginx!' not in result.content"
|
||||
vars:
|
||||
ip: >-
|
||||
{{ nginx_services.resources[0].status.loadBalancer.ingress[0].ip }}
|
||||
port: >-
|
||||
{{ nginx_services.resources[0].spec.ports[0].port }}
|
||||
# Deactivated linter rules:
|
||||
# - jinja[invalid]: As of version 6.6.0, ansible-lint complains that the input to ipwrap
|
||||
# would be undefined. This will not be the case during playbook execution.
|
||||
# noqa jinja[invalid]
|
||||
|
||||
always:
|
||||
- name: "Remove namespace: {{ testing_namespace }}"
|
||||
kubernetes.core.k8s:
|
||||
api_version: v1
|
||||
kind: Namespace
|
||||
name: "{{ testing_namespace }}"
|
||||
state: absent
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
+1
-1
@@ -6,4 +6,4 @@ outside_host: localhost
|
||||
testing_namespace: molecule-verify-from-outside
|
||||
|
||||
# The directory in which the example manifests reside
|
||||
example_manifests_path: ../../../../example
|
||||
example_manifests_path: ../../../example
|
||||
+2
@@ -7,6 +7,8 @@
|
||||
ansible.builtin.import_tasks: kubecfg-fetch.yml
|
||||
- name: "TEST CASE: Get nodes"
|
||||
ansible.builtin.include_tasks: test/get-nodes.yml
|
||||
- name: "TEST CASE: Verify components"
|
||||
ansible.builtin.include_tasks: test/verify-components.yml
|
||||
- name: "TEST CASE: Deploy example"
|
||||
ansible.builtin.include_tasks: test/deploy-example.yml
|
||||
always:
|
||||
@@ -0,0 +1,127 @@
|
||||
---
|
||||
- name: Deploy example
|
||||
block:
|
||||
- name: "Create namespace: {{ testing_namespace }}"
|
||||
kubernetes.core.k8s:
|
||||
api_version: v1
|
||||
kind: Namespace
|
||||
name: "{{ testing_namespace }}"
|
||||
state: present
|
||||
wait: true
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
|
||||
- name: Apply example manifests
|
||||
kubernetes.core.k8s:
|
||||
src: "{{ example_manifests_path }}/{{ item }}"
|
||||
namespace: "{{ testing_namespace }}"
|
||||
state: present
|
||||
wait: true
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
with_items:
|
||||
- deployment.yml
|
||||
- service.yml
|
||||
|
||||
- name: Get info about nginx service
|
||||
kubernetes.core.k8s_info:
|
||||
kind: service
|
||||
name: nginx
|
||||
namespace: "{{ testing_namespace }}"
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
vars:
|
||||
metallb_ip: status.loadBalancer.ingress[0].ip
|
||||
metallb_port: spec.ports[0].port
|
||||
register: nginx_services
|
||||
|
||||
- name: Wait for the load balancer address to be assigned
|
||||
ansible.builtin.set_fact:
|
||||
nginx_lb_ip: >-
|
||||
{{
|
||||
nginx_services.resources[0].status.loadBalancer.ingress[0].ip
|
||||
if (nginx_services.resources | length > 0) and
|
||||
(nginx_services.resources[0].status.loadBalancer.ingress is defined) and
|
||||
(nginx_services.resources[0].status.loadBalancer.ingress | length > 0)
|
||||
else ''
|
||||
}}
|
||||
|
||||
- name: Retry until the load balancer service has an external IP
|
||||
block:
|
||||
- name: Refresh nginx service until it has an assigned address
|
||||
kubernetes.core.k8s_info:
|
||||
kind: service
|
||||
name: nginx
|
||||
namespace: "{{ testing_namespace }}"
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: nginx_lb_wait
|
||||
until: >-
|
||||
(nginx_lb_wait.resources | length > 0) and
|
||||
(nginx_lb_wait.resources[0].status.loadBalancer.ingress is defined) and
|
||||
(nginx_lb_wait.resources[0].status.loadBalancer.ingress | length > 0)
|
||||
retries: 30
|
||||
delay: 5
|
||||
|
||||
- name: Record the assigned load balancer address
|
||||
ansible.builtin.set_fact:
|
||||
nginx_lb_ip: >-
|
||||
{{ nginx_lb_wait.resources[0].status.loadBalancer.ingress[0].ip }}
|
||||
|
||||
- name: Assert that the nginx welcome page is available
|
||||
ansible.builtin.uri:
|
||||
url: http://{{ nginx_lb_ip | ansible.utils.ipwrap }}:{{ port_ }}/
|
||||
return_content: true
|
||||
register: result
|
||||
failed_when: "'Welcome to nginx!' not in result.content"
|
||||
vars:
|
||||
port_: >-
|
||||
{{ nginx_services.resources[0].spec.ports[0].port }}
|
||||
|
||||
- name: Initialize load balancer address range check
|
||||
ansible.builtin.set_fact:
|
||||
lb_addr_in_range: false
|
||||
lb_ip_value: "{{ nginx_lb_ip }}"
|
||||
|
||||
- name: Check load balancer address against start-end pools
|
||||
ansible.builtin.set_fact:
|
||||
lb_addr_in_range: true
|
||||
loop: "{{ verify_lb_ip_range }}"
|
||||
loop_control:
|
||||
label: "{{ item }}"
|
||||
when:
|
||||
- "'-' in item"
|
||||
- "'/' not in item"
|
||||
- >-
|
||||
(lb_ip_value | ansible.utils.ipaddr('int') | int) >=
|
||||
(item.split('-')[0] | ansible.utils.ipaddr('int') | int)
|
||||
- >-
|
||||
(lb_ip_value | ansible.utils.ipaddr('int') | int) <=
|
||||
(item.split('-')[1] | ansible.utils.ipaddr('int') | int)
|
||||
|
||||
- name: Check load balancer address against CIDR pools
|
||||
ansible.builtin.set_fact:
|
||||
lb_addr_in_range: true
|
||||
loop: "{{ verify_lb_ip_range }}"
|
||||
loop_control:
|
||||
label: "{{ item }}"
|
||||
when:
|
||||
- "'/' in item"
|
||||
- (lb_ip_value | ansible.utils.ipaddr(item)) is string
|
||||
|
||||
- name: Assert that the load balancer address is within a configured pool
|
||||
ansible.builtin.assert:
|
||||
that: lb_addr_in_range
|
||||
success_msg: "LoadBalancer address {{ lb_ip_value }} is in a configured range"
|
||||
fail_msg: >-
|
||||
LoadBalancer address {{ lb_ip_value }} is not in a configured
|
||||
range {{ verify_lb_ip_range }}
|
||||
# Deactivated linter rules:
|
||||
# - jinja[invalid]: As of version 6.6.0, ansible-lint complains that the input to ipwrap
|
||||
# would be undefined. This will not be the case during playbook execution.
|
||||
# noqa jinja[invalid]
|
||||
|
||||
always:
|
||||
- name: "Remove namespace: {{ testing_namespace }}"
|
||||
kubernetes.core.k8s:
|
||||
api_version: v1
|
||||
kind: Namespace
|
||||
name: "{{ testing_namespace }}"
|
||||
state: absent
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
+2
-2
@@ -9,7 +9,7 @@
|
||||
ansible.builtin.assert:
|
||||
that: found_nodes == expected_nodes
|
||||
success_msg: "Found nodes as expected: {{ found_nodes }}"
|
||||
fail_msg: "Expected nodes {{ expected_nodes }}, but found nodes {{ found_nodes }}"
|
||||
fail_msg: Expected nodes {{ expected_nodes }}, but found nodes {{ found_nodes }}
|
||||
vars:
|
||||
found_nodes: >-
|
||||
{{ cluster_nodes | json_query('resources[*].metadata.name') | unique | sort }}
|
||||
@@ -22,7 +22,7 @@
|
||||
| unique
|
||||
| sort
|
||||
}}
|
||||
# Deactivated linter rules:
|
||||
# Deactivated linter rules:
|
||||
# - jinja[invalid]: As of version 6.6.0, ansible-lint complains that the input to ipwrap
|
||||
# would be undefined. This will not be the case during playbook execution.
|
||||
# noqa jinja[invalid]
|
||||
@@ -0,0 +1,354 @@
|
||||
---
|
||||
# Scenario-aware verification of cluster components and their live image tags.
|
||||
# Scenario identity (verify_cni / verify_lb) and expected address range come
|
||||
# from each scenario's verify-vars.yml, which is plain inventory data available
|
||||
# to the verify play. Converge-time set_fact values are not persisted between
|
||||
# the two Ansible processes, so they are never used here.
|
||||
- name: Verify cluster components report expected versions
|
||||
block:
|
||||
- name: Get all nodes with their kubelet versions
|
||||
kubernetes.core.k8s_info:
|
||||
kind: node
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: verify_nodes
|
||||
|
||||
- name: Assert each node reports the expected Kubernetes version
|
||||
ansible.builtin.assert:
|
||||
that: item.status.nodeInfo.kubeletVersion == k3s_version
|
||||
success_msg: "{{ item.metadata.name }} reports {{ k3s_version }}"
|
||||
fail_msg: >-
|
||||
{{ item.metadata.name }} reports
|
||||
{{ item.status.nodeInfo.kubeletVersion }},
|
||||
expected {{ k3s_version }}
|
||||
loop: "{{ verify_nodes.resources }}"
|
||||
loop_control:
|
||||
label: "{{ item.metadata.name }}"
|
||||
|
||||
- name: Verify Flannel is the active CNI
|
||||
when: verify_cni == 'flannel'
|
||||
block:
|
||||
- name: Assert every node reports Ready
|
||||
ansible.builtin.assert:
|
||||
that: item.status.conditions
|
||||
| selectattr('type', 'equalto', 'Ready')
|
||||
| map(attribute='status') | first | default('') == 'True'
|
||||
success_msg: "{{ item.metadata.name }} is Ready"
|
||||
fail_msg: "{{ item.metadata.name }} is not Ready"
|
||||
loop: "{{ verify_nodes.resources }}"
|
||||
loop_control:
|
||||
label: "{{ item.metadata.name }} ready"
|
||||
|
||||
- name: Assert every node registered a node IP from its interface
|
||||
# Each k3s node is launched with --node-ip derived from flannel_iface.
|
||||
# Confirm every node carries a real InternalIP (not a loopback), which
|
||||
# proves k3s bound to the cluster interface rather than defaulting to 127.0.0.1.
|
||||
ansible.builtin.assert:
|
||||
that: >-
|
||||
(node_internal_ips | length) >= 1 and
|
||||
(node_internal_ips | reject('eq', '127.0.0.1') | list | length) == node_internal_ips | length
|
||||
success_msg: "{{ item.metadata.name }} is bound to {{ node_internal_ips | join(', ') }}"
|
||||
fail_msg: >-
|
||||
{{ item.metadata.name }} has no non-loopback InternalIP
|
||||
(got: {{ node_internal_ips | join(', ') }})
|
||||
vars:
|
||||
node_internal_ips: >-
|
||||
{{
|
||||
(item.status.addresses | default([]))
|
||||
| selectattr('type', 'equalto', 'InternalIP')
|
||||
| map(attribute='address')
|
||||
| list
|
||||
}}
|
||||
loop: "{{ verify_nodes.resources }}"
|
||||
loop_control:
|
||||
label: "{{ item.metadata.name }} InternalIP"
|
||||
|
||||
- name: Get any Calico namespaces with Flannel enabled
|
||||
kubernetes.core.k8s_info:
|
||||
kind: Namespace
|
||||
name: calico-system
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: flannel_calico_absent
|
||||
|
||||
- name: Assert there is no Calico system namespace
|
||||
ansible.builtin.assert:
|
||||
that: flannel_calico_absent.resources | length == 0
|
||||
success_msg: "No Calico present with Flannel"
|
||||
fail_msg: "A Calico namespace exists alongside Flannel"
|
||||
|
||||
- name: Get the Cilium namespace with Flannel enabled
|
||||
kubernetes.core.k8s_info:
|
||||
kind: Namespace
|
||||
name: cilium
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: flannel_cilium
|
||||
|
||||
- name: Assert the Cilium namespace is absent
|
||||
ansible.builtin.assert:
|
||||
that: flannel_cilium.resources | length == 0
|
||||
success_msg: "No Cilium present with Flannel"
|
||||
fail_msg: "A Cilium namespace exists alongside Flannel"
|
||||
|
||||
- name: Verify Calico is the active CNI
|
||||
when: verify_cni == 'calico'
|
||||
block:
|
||||
- name: Get the Calico node DaemonSet image
|
||||
kubernetes.core.k8s_info:
|
||||
kind: DaemonSet
|
||||
name: calico-node
|
||||
namespace: calico-system
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: calico_node_ds
|
||||
|
||||
- name: Assert the Calico node image uses the expected tag
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- calico_node_ds.resources | length == 1
|
||||
- calico_node_image | regex_search(':' ~ calico_tag) is not none
|
||||
success_msg: "Calico node image uses tag {{ calico_tag }}"
|
||||
fail_msg: >-
|
||||
Calico node image {{ calico_node_image }},
|
||||
expected {{ calico_tag }}
|
||||
vars:
|
||||
calico_node_image: "{{ calico_node_ds.resources[0].spec.template.spec.containers[0].image }}"
|
||||
|
||||
- name: Get Calico TigeraStatus for calico and apiserver
|
||||
kubernetes.core.k8s_info:
|
||||
api_version: operator.tigera.io/v1
|
||||
kind: TigeraStatus
|
||||
name: "{{ item }}"
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: calico_tigerastatus
|
||||
loop:
|
||||
- calico
|
||||
- apiserver
|
||||
loop_control:
|
||||
label: "Tigerastatus/{{ item }}"
|
||||
|
||||
- name: Assert Calico TigeraStatus reports Available
|
||||
ansible.builtin.assert:
|
||||
that: >-
|
||||
item.resources | length == 1 and
|
||||
(item.resources[0].status.conditions
|
||||
| selectattr('type', 'equalto', 'Available')
|
||||
| map(attribute='status') | first | default('')) == 'True'
|
||||
success_msg: "Tigerastatus {{ item.resources[0].metadata.name }} is Available"
|
||||
fail_msg: "Tigerastatus is not Available"
|
||||
loop: "{{ calico_tigerastatus.results }}"
|
||||
loop_control:
|
||||
label: "Tigerastatus Available"
|
||||
|
||||
- name: Get any Flannel DaemonSets with Calico enabled
|
||||
kubernetes.core.k8s_info:
|
||||
kind: DaemonSet
|
||||
namespace: kube-flannel
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: no_flannel_ds
|
||||
|
||||
- name: Assert there are no Flannel DaemonSets
|
||||
ansible.builtin.assert:
|
||||
that: no_flannel_ds.resources | length == 0
|
||||
success_msg: "No Flannel DaemonSet present with Calico"
|
||||
fail_msg: "A Flannel DaemonSet exists alongside Calico"
|
||||
|
||||
- name: Verify Cilium is the active CNI
|
||||
when: verify_cni == 'cilium'
|
||||
block:
|
||||
- name: Get the Cilium agent and operator images
|
||||
kubernetes.core.k8s_info:
|
||||
kind: "{{ item.kind }}"
|
||||
name: "{{ item.name }}"
|
||||
namespace: kube-system
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: cilium_info
|
||||
loop:
|
||||
- { kind: DaemonSet, name: cilium }
|
||||
- { kind: Deployment, name: cilium-operator }
|
||||
loop_control:
|
||||
label: "{{ item.kind }}/{{ item.name }}"
|
||||
|
||||
- name: Assert Cilium agent and operator use the expected image tag
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- cilium_agent_image | regex_search(':' ~ cilium_tag) is not none
|
||||
- cilium_operator_image | regex_search(':' ~ cilium_tag) is not none
|
||||
success_msg: "Cilium agent and operator use {{ cilium_tag }}"
|
||||
fail_msg: >-
|
||||
Cilium agent {{ cilium_agent_image }},
|
||||
operator {{ cilium_operator_image }},
|
||||
expected {{ cilium_tag }}
|
||||
vars:
|
||||
cilium_agent_image: >-
|
||||
{{ (cilium_info.results
|
||||
| selectattr('resources', 'defined')
|
||||
| map(attribute='resources')
|
||||
| list
|
||||
| map(attribute='0')
|
||||
| selectattr('kind', 'equalto', 'DaemonSet')
|
||||
| list)[0].spec.template.spec.containers[0].image }}
|
||||
cilium_operator_image: >-
|
||||
{{ (cilium_info.results
|
||||
| selectattr('resources', 'defined')
|
||||
| map(attribute='resources')
|
||||
| list
|
||||
| map(attribute='0')
|
||||
| selectattr('kind', 'equalto', 'Deployment')
|
||||
| list)[0].spec.template.spec.containers[0].image }}
|
||||
|
||||
- name: Get Hubble relay and UI deployments when enabled
|
||||
kubernetes.core.k8s_info:
|
||||
kind: Deployment
|
||||
name: "{{ item }}"
|
||||
namespace: kube-system
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: hubble_info
|
||||
loop:
|
||||
- hubble-relay
|
||||
- hubble-ui
|
||||
loop_control:
|
||||
label: "Deployment/{{ item }}"
|
||||
when: cilium_hubble | bool
|
||||
|
||||
- name: Assert Hubble components are Ready when enabled
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- item.resources | length == 1
|
||||
- item.resources[0].status.readyReplicas | default(0) >= 1
|
||||
success_msg: "Hubble deployment {{ item.resources[0].metadata.name }} is Ready"
|
||||
fail_msg: "Hubble deployment is not Ready"
|
||||
loop: "{{ hubble_info.results }}"
|
||||
loop_control:
|
||||
label: "Hubble deployment"
|
||||
when: cilium_hubble | bool
|
||||
|
||||
- name: Get any Flannel DaemonSets with Cilium enabled
|
||||
kubernetes.core.k8s_info:
|
||||
kind: DaemonSet
|
||||
namespace: kube-flannel
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: no_flannel_ds_cilium
|
||||
|
||||
- name: Assert there are no Flannel DaemonSets
|
||||
ansible.builtin.assert:
|
||||
that: no_flannel_ds_cilium.resources | length == 0
|
||||
success_msg: "No Flannel DaemonSet present with Cilium"
|
||||
fail_msg: "A Flannel DaemonSet exists alongside Cilium"
|
||||
|
||||
- name: Verify MetalLB is the active load balancer
|
||||
when: verify_lb == 'metallb'
|
||||
block:
|
||||
- name: Get the MetalLB controller and speaker images
|
||||
kubernetes.core.k8s_info:
|
||||
kind: "{{ item.kind }}"
|
||||
name: "{{ item.name }}"
|
||||
namespace: metallb-system
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: metallb_info
|
||||
until: metallb_info.resources | length > 0
|
||||
retries: 15
|
||||
delay: 10
|
||||
loop:
|
||||
- { kind: Deployment, name: controller }
|
||||
- { kind: DaemonSet, name: speaker }
|
||||
loop_control:
|
||||
label: "{{ item.kind }}/{{ item.name }}"
|
||||
|
||||
- name: Fail with a clear message if MetalLB resources are missing
|
||||
ansible.builtin.fail:
|
||||
msg: >-
|
||||
Did not find {{ item.kind | lower }} {{ item.name }} in
|
||||
metallb-system. Expected MetalLB to be deployed in this
|
||||
scenario (verify_lb: {{ verify_lb }}).
|
||||
when: item.resources | length == 0
|
||||
loop: "{{ metallb_info.results }}"
|
||||
loop_control:
|
||||
label: "{{ item.item.kind }}/{{ item.item.name }}"
|
||||
|
||||
- name: Assert MetalLB controller and speaker use the expected image tags
|
||||
ansible.builtin.assert:
|
||||
# regex_search returns a string or none; check for a match with `is not
|
||||
# none` so the assertion is a real boolean (ansible-core 2.19 rejects
|
||||
# string conditionals and `| bool` deprecates string coercion).
|
||||
that:
|
||||
- controller_image | regex_search(metal_lb_controller_tag_version) is not none
|
||||
- speaker_image | regex_search(metal_lb_speaker_tag_version) is not none
|
||||
success_msg: >-
|
||||
MetalLB controller {{ metal_lb_controller_tag_version }},
|
||||
speaker {{ metal_lb_speaker_tag_version }}
|
||||
fail_msg: >-
|
||||
MetalLB controller {{ controller_image }},
|
||||
speaker {{ speaker_image }}
|
||||
vars:
|
||||
controller_image: >-
|
||||
{{ (metallb_info.results
|
||||
| selectattr('resources', 'defined')
|
||||
| map(attribute='resources')
|
||||
| list
|
||||
| map(attribute='0')
|
||||
| selectattr('kind', 'equalto', 'Deployment')
|
||||
| list)[0].spec.template.spec.containers[0].image }}
|
||||
speaker_image: >-
|
||||
{{ (metallb_info.results
|
||||
| selectattr('resources', 'defined')
|
||||
| map(attribute='resources')
|
||||
| list
|
||||
| map(attribute='0')
|
||||
| selectattr('kind', 'equalto', 'DaemonSet')
|
||||
| list)[0].spec.template.spec.containers[0].image }}
|
||||
|
||||
- name: Verify kube-vip is the active load balancer
|
||||
when: verify_lb == 'kube-vip'
|
||||
block:
|
||||
- name: Get the kube-vip and cloud provider images
|
||||
kubernetes.core.k8s_info:
|
||||
kind: "{{ item.kind }}"
|
||||
name: "{{ item.name }}"
|
||||
namespace: kube-system
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: kubevip_info
|
||||
loop:
|
||||
- { kind: DaemonSet, name: kube-vip-ds }
|
||||
- { kind: Deployment, name: kube-vip-cloud-provider }
|
||||
loop_control:
|
||||
label: "{{ item.kind }}/{{ item.name }}"
|
||||
|
||||
- name: Assert the kube-vip and cloud provider image tags
|
||||
ansible.builtin.assert:
|
||||
that:
|
||||
- kubevip_image | regex_search(':' ~ kube_vip_tag_version) is not none
|
||||
- cloud_provider_image | regex_search(verify_kube_vip_cloud_provider_tag) is not none
|
||||
success_msg: >-
|
||||
kube-vip {{ kube_vip_tag_version }},
|
||||
cloud provider {{ verify_kube_vip_cloud_provider_tag }}
|
||||
fail_msg: >-
|
||||
kube-vip {{ kubevip_image }},
|
||||
cloud provider {{ cloud_provider_image }}
|
||||
vars:
|
||||
kubevip_image: >-
|
||||
{{ (kubevip_info.results
|
||||
| selectattr('resources', 'defined')
|
||||
| map(attribute='resources')
|
||||
| list
|
||||
| map(attribute='0')
|
||||
| selectattr('kind', 'equalto', 'DaemonSet')
|
||||
| list)[0].spec.template.spec.containers[0].image }}
|
||||
cloud_provider_image: >-
|
||||
{{ (kubevip_info.results
|
||||
| selectattr('resources', 'defined')
|
||||
| map(attribute='resources')
|
||||
| list
|
||||
| map(attribute='0')
|
||||
| selectattr('kind', 'equalto', 'Deployment')
|
||||
| list)[0].spec.template.spec.containers[0].image }}
|
||||
|
||||
- name: Get the MetalLB namespace with kube-vip enabled
|
||||
kubernetes.core.k8s_info:
|
||||
kind: Namespace
|
||||
name: metallb-system
|
||||
kubeconfig: "{{ kubecfg_path }}"
|
||||
register: metallb_absent
|
||||
|
||||
- name: Assert the MetalLB namespace does not exist
|
||||
ansible.builtin.assert:
|
||||
that: metallb_absent.resources | length == 0
|
||||
success_msg: "MetalLB is not installed with kube-vip"
|
||||
fail_msg: "MetalLB namespace exists alongside kube-vip"
|
||||
@@ -5,14 +5,9 @@ driver:
|
||||
name: vagrant
|
||||
platforms:
|
||||
- name: control1
|
||||
box: generic/ubuntu2204
|
||||
box: bento/ubuntu-26.04
|
||||
memory: 4096
|
||||
cpus: 4
|
||||
config_options:
|
||||
# We currently can not use public-key based authentication on Ubuntu 22.04,
|
||||
# see: https://github.com/chef/bento/issues/1405
|
||||
ssh.username: "vagrant"
|
||||
ssh.password: "vagrant"
|
||||
groups:
|
||||
- k3s_cluster
|
||||
- master
|
||||
@@ -21,6 +16,8 @@ platforms:
|
||||
ip: 192.168.30.50
|
||||
provisioner:
|
||||
name: ansible
|
||||
env:
|
||||
ANSIBLE_VERBOSITY: 1
|
||||
playbooks:
|
||||
converge: ../resources/converge.yml
|
||||
side_effect: ../resources/reset.yml
|
||||
@@ -31,7 +28,6 @@ provisioner:
|
||||
scenario:
|
||||
test_sequence:
|
||||
- dependency
|
||||
- lint
|
||||
- cleanup
|
||||
- destroy
|
||||
- syntax
|
||||
|
||||
@@ -1,15 +1,18 @@
|
||||
---
|
||||
- name: Apply overrides
|
||||
hosts: all
|
||||
serial: 1
|
||||
tasks:
|
||||
- name: Override host variables
|
||||
ansible.builtin.set_fact:
|
||||
# See: https://github.com/flannel-io/flannel/blob/67d603aaf45ef80f5dd39f43714fc5e6f8a637eb/Documentation/troubleshooting.md#Vagrant # noqa yaml[line-length]
|
||||
flannel_iface: eth1
|
||||
# See:
|
||||
# https://github.com/flannel-io/flannel/blob/67d603aaf45ef80f5dd39f43714fc5e6f8a637eb/Documentation/troubleshooting.md#Vagrant
|
||||
flannel_iface: "{{ 'eth1' if 'eth1' in ansible_facts.interfaces else 'enp0s8' }}"
|
||||
kube_vip_iface: "{{ 'eth1' if 'eth1' in ansible_facts.interfaces else 'enp0s8' }}"
|
||||
|
||||
# The test VMs might be a bit slow, so we give them more time to join the cluster:
|
||||
retry_count: 45
|
||||
|
||||
# Make sure that our IP ranges do not collide with those of the default scenario
|
||||
apiserver_endpoint: "192.168.30.223"
|
||||
metal_lb_ip_range: "192.168.30.91-192.168.30.99"
|
||||
apiserver_endpoint: 192.168.30.223
|
||||
metal_lb_ip_range: 192.168.30.91-192.168.30.99
|
||||
|
||||
@@ -0,0 +1,6 @@
|
||||
---
|
||||
# Durable verify inputs for the single_node (flannel + MetalLB) scenario.
|
||||
verify_cni: flannel
|
||||
verify_lb: metallb
|
||||
verify_lb_ip_range:
|
||||
- 192.168.30.91-192.168.30.99
|
||||
@@ -1,3 +1,3 @@
|
||||
#!/bin/bash
|
||||
|
||||
ansible-playbook reboot.yml -i inventory/my-cluster/hosts.ini
|
||||
ansible-playbook reboot.yml
|
||||
|
||||
+25
-4
@@ -1,9 +1,30 @@
|
||||
---
|
||||
- name: Reboot k3s_cluster
|
||||
hosts: k3s_cluster
|
||||
gather_facts: yes
|
||||
become: yes
|
||||
gather_facts: true
|
||||
|
||||
# Stagger the reboot across the cluster when concurrent_reboots is set.
|
||||
# Defaults to '100%' so the whole cluster reboots at once (backward compatible).
|
||||
serial: "{{ concurrent_reboots | default('100%') }}"
|
||||
|
||||
tasks:
|
||||
- name: Reboot the nodes (and Wait upto 5 mins max)
|
||||
reboot:
|
||||
- name: >-
|
||||
{{
|
||||
'Reboot all nodes at once'
|
||||
if (concurrent_reboots is not defined)
|
||||
else 'Reboot nodes with concurrency of ' ~ concurrent_reboots
|
||||
}}
|
||||
become: true
|
||||
ansible.builtin.reboot:
|
||||
reboot_command: "{{ custom_reboot_command | default(omit) }}"
|
||||
reboot_timeout: 300
|
||||
test_command: >-
|
||||
{{ 'kubectl get nodes' if 'master' in group_names else 'whoami' }}
|
||||
|
||||
- name: Optional wait before rebooting the next batch of nodes
|
||||
ansible.builtin.pause:
|
||||
seconds: "{{ wait_seconds_after_reboot | int }}"
|
||||
when: >-
|
||||
concurrent_reboots is defined and
|
||||
wait_seconds_after_reboot is defined and
|
||||
wait_seconds_after_reboot | int > 0
|
||||
|
||||
+10
-12
@@ -1,12 +1,10 @@
|
||||
ansible-core>=2.13.5
|
||||
ansible-lint>=6.8.6
|
||||
jmespath>=1.0.1
|
||||
jsonpatch>=1.32
|
||||
kubernetes>=25.3.0
|
||||
molecule-vagrant>=1.0.0
|
||||
molecule>=4.0.3
|
||||
netaddr>=0.8.0
|
||||
pre-commit>=2.20.0
|
||||
pre-commit-hooks>=1.3.1
|
||||
pyyaml>=6.0
|
||||
yamllint>=1.28.0
|
||||
ansible-core>=2.19.11
|
||||
jmespath>=1.1.0
|
||||
jsonpatch>=1.33
|
||||
kubernetes>=29.0.0
|
||||
molecule-plugins[vagrant]
|
||||
molecule>=6.0.3
|
||||
netaddr>=0.10.1
|
||||
pre-commit>=3.6.0
|
||||
pre-commit-hooks>=4.5.0
|
||||
pyyaml>=6.0.1
|
||||
|
||||
+81
-124
@@ -1,212 +1,169 @@
|
||||
#
|
||||
# This file is autogenerated by pip-compile with python 3.8
|
||||
# To update, run:
|
||||
# This file is autogenerated by pip-compile with Python 3.11
|
||||
# by the following command:
|
||||
#
|
||||
# pip-compile requirements.in
|
||||
# pip-compile --output-file=requirements.txt requirements.in
|
||||
#
|
||||
ansible-compat==2.2.4
|
||||
# via
|
||||
# ansible-lint
|
||||
# molecule
|
||||
ansible-core==2.13.5
|
||||
ansible-compat==4.1.11
|
||||
# via molecule
|
||||
ansible-core==2.19.11
|
||||
# via
|
||||
# -r requirements.in
|
||||
# ansible-lint
|
||||
ansible-lint==6.8.6
|
||||
# via -r requirements.in
|
||||
arrow==1.2.3
|
||||
# via jinja2-time
|
||||
attrs==22.1.0
|
||||
# via jsonschema
|
||||
binaryornot==0.4.4
|
||||
# via cookiecutter
|
||||
black==22.10.0
|
||||
# via ansible-lint
|
||||
bracex==2.3.post1
|
||||
# ansible-compat
|
||||
# molecule
|
||||
attrs==23.2.0
|
||||
# via
|
||||
# jsonschema
|
||||
# referencing
|
||||
bracex==2.4
|
||||
# via wcmatch
|
||||
cachetools==5.2.0
|
||||
cachetools==5.3.2
|
||||
# via google-auth
|
||||
certifi==2022.9.24
|
||||
certifi==2023.11.17
|
||||
# via
|
||||
# kubernetes
|
||||
# requests
|
||||
cffi==1.15.1
|
||||
cffi==1.16.0
|
||||
# via cryptography
|
||||
cfgv==3.3.1
|
||||
cfgv==3.4.0
|
||||
# via pre-commit
|
||||
chardet==5.0.0
|
||||
# via binaryornot
|
||||
charset-normalizer==2.1.1
|
||||
charset-normalizer==3.3.2
|
||||
# via requests
|
||||
click==8.1.3
|
||||
click==8.1.7
|
||||
# via
|
||||
# black
|
||||
# click-help-colors
|
||||
# cookiecutter
|
||||
# molecule
|
||||
click-help-colors==0.9.1
|
||||
click-help-colors==0.9.4
|
||||
# via molecule
|
||||
commonmark==0.9.1
|
||||
# via rich
|
||||
cookiecutter==2.1.1
|
||||
# via molecule
|
||||
cryptography==38.0.3
|
||||
cryptography==41.0.7
|
||||
# via ansible-core
|
||||
distlib==0.3.6
|
||||
distlib==0.3.8
|
||||
# via virtualenv
|
||||
distro==1.8.0
|
||||
# via selinux
|
||||
enrich==1.2.7
|
||||
# via molecule
|
||||
filelock==3.8.0
|
||||
# via
|
||||
# ansible-lint
|
||||
# virtualenv
|
||||
google-auth==2.14.0
|
||||
filelock==3.13.1
|
||||
# via virtualenv
|
||||
google-auth==2.26.2
|
||||
# via kubernetes
|
||||
identify==2.5.8
|
||||
identify==2.5.33
|
||||
# via pre-commit
|
||||
idna==3.4
|
||||
idna==3.6
|
||||
# via requests
|
||||
jinja2==3.1.2
|
||||
jinja2==3.1.3
|
||||
# via
|
||||
# ansible-core
|
||||
# cookiecutter
|
||||
# jinja2-time
|
||||
# molecule
|
||||
# molecule-vagrant
|
||||
jinja2-time==0.2.0
|
||||
# via cookiecutter
|
||||
jmespath==1.0.1
|
||||
jmespath==1.1.0
|
||||
# via -r requirements.in
|
||||
jsonpatch==1.32
|
||||
jsonpatch==1.33
|
||||
# via -r requirements.in
|
||||
jsonpointer==2.3
|
||||
jsonpointer==2.4
|
||||
# via jsonpatch
|
||||
jsonschema==4.17.0
|
||||
jsonschema==4.21.1
|
||||
# via
|
||||
# ansible-compat
|
||||
# ansible-lint
|
||||
# molecule
|
||||
kubernetes==25.3.0
|
||||
jsonschema-specifications==2023.12.1
|
||||
# via jsonschema
|
||||
kubernetes==29.0.0
|
||||
# via -r requirements.in
|
||||
markupsafe==2.1.1
|
||||
markdown-it-py==3.0.0
|
||||
# via rich
|
||||
markupsafe==2.1.4
|
||||
# via jinja2
|
||||
molecule==4.0.3
|
||||
mdurl==0.1.2
|
||||
# via markdown-it-py
|
||||
molecule==6.0.3
|
||||
# via
|
||||
# -r requirements.in
|
||||
# molecule-vagrant
|
||||
molecule-vagrant==1.0.0
|
||||
# molecule-plugins
|
||||
molecule-plugins[vagrant]==23.6.0
|
||||
# via -r requirements.in
|
||||
mypy-extensions==0.4.3
|
||||
# via black
|
||||
netaddr==0.8.0
|
||||
netaddr==0.10.1
|
||||
# via -r requirements.in
|
||||
nodeenv==1.7.0
|
||||
nodeenv==1.8.0
|
||||
# via pre-commit
|
||||
oauthlib==3.2.2
|
||||
# via requests-oauthlib
|
||||
packaging==21.3
|
||||
# via
|
||||
# kubernetes
|
||||
# requests-oauthlib
|
||||
packaging==23.2
|
||||
# via
|
||||
# ansible-compat
|
||||
# ansible-core
|
||||
# ansible-lint
|
||||
# molecule
|
||||
pathspec==0.10.1
|
||||
# via
|
||||
# black
|
||||
# yamllint
|
||||
platformdirs==2.5.2
|
||||
# via
|
||||
# black
|
||||
# virtualenv
|
||||
pluggy==1.0.0
|
||||
platformdirs==4.1.0
|
||||
# via virtualenv
|
||||
pluggy==1.3.0
|
||||
# via molecule
|
||||
pre-commit==2.20.0
|
||||
pre-commit==3.8.0
|
||||
# via -r requirements.in
|
||||
pre-commit-hooks==4.4.0
|
||||
pre-commit-hooks==4.6.0
|
||||
# via -r requirements.in
|
||||
pyasn1==0.4.8
|
||||
pyasn1==0.5.1
|
||||
# via
|
||||
# pyasn1-modules
|
||||
# rsa
|
||||
pyasn1-modules==0.2.8
|
||||
pyasn1-modules==0.3.0
|
||||
# via google-auth
|
||||
pycparser==2.21
|
||||
# via cffi
|
||||
pygments==2.13.0
|
||||
pygments==2.17.2
|
||||
# via rich
|
||||
pyparsing==3.0.9
|
||||
# via packaging
|
||||
pyrsistent==0.19.2
|
||||
# via jsonschema
|
||||
python-dateutil==2.8.2
|
||||
# via
|
||||
# arrow
|
||||
# kubernetes
|
||||
python-slugify==6.1.2
|
||||
# via cookiecutter
|
||||
# via kubernetes
|
||||
python-vagrant==1.0.0
|
||||
# via molecule-vagrant
|
||||
pyyaml==6.0
|
||||
# via molecule-plugins
|
||||
pyyaml==6.0.2
|
||||
# via
|
||||
# -r requirements.in
|
||||
# ansible-compat
|
||||
# ansible-core
|
||||
# ansible-lint
|
||||
# cookiecutter
|
||||
# kubernetes
|
||||
# molecule
|
||||
# molecule-vagrant
|
||||
# pre-commit
|
||||
# yamllint
|
||||
requests==2.28.1
|
||||
referencing==0.32.1
|
||||
# via
|
||||
# jsonschema
|
||||
# jsonschema-specifications
|
||||
requests==2.31.0
|
||||
# via
|
||||
# cookiecutter
|
||||
# kubernetes
|
||||
# requests-oauthlib
|
||||
requests-oauthlib==1.3.1
|
||||
# via kubernetes
|
||||
resolvelib==0.8.1
|
||||
resolvelib==1.0.1
|
||||
# via ansible-core
|
||||
rich==12.6.0
|
||||
rich==13.7.0
|
||||
# via
|
||||
# ansible-lint
|
||||
# enrich
|
||||
# molecule
|
||||
rpds-py==0.17.1
|
||||
# via
|
||||
# jsonschema
|
||||
# referencing
|
||||
rsa==4.9
|
||||
# via google-auth
|
||||
ruamel-yaml==0.17.21
|
||||
# via
|
||||
# ansible-lint
|
||||
# pre-commit-hooks
|
||||
selinux==0.2.1
|
||||
# via molecule-vagrant
|
||||
ruamel-yaml==0.18.5
|
||||
# via pre-commit-hooks
|
||||
ruamel-yaml-clib==0.2.15
|
||||
# via ruamel-yaml
|
||||
six==1.16.0
|
||||
# via
|
||||
# google-auth
|
||||
# kubernetes
|
||||
# python-dateutil
|
||||
subprocess-tee==0.3.5
|
||||
subprocess-tee==0.4.1
|
||||
# via ansible-compat
|
||||
text-unidecode==1.3
|
||||
# via python-slugify
|
||||
toml==0.10.2
|
||||
# via pre-commit
|
||||
urllib3==1.26.12
|
||||
urllib3==2.1.0
|
||||
# via
|
||||
# kubernetes
|
||||
# requests
|
||||
virtualenv==20.16.6
|
||||
virtualenv==20.25.0
|
||||
# via pre-commit
|
||||
wcmatch==8.4.1
|
||||
# via ansible-lint
|
||||
websocket-client==1.4.2
|
||||
wcmatch==8.5
|
||||
# via molecule
|
||||
websocket-client==1.7.0
|
||||
# via kubernetes
|
||||
yamllint==1.28.0
|
||||
# via
|
||||
# -r requirements.in
|
||||
# ansible-lint
|
||||
|
||||
# The following packages are considered to be unsafe in a requirements file:
|
||||
# setuptools
|
||||
|
||||
@@ -1,3 +1,3 @@
|
||||
#!/bin/bash
|
||||
|
||||
ansible-playbook reset.yml -i inventory/my-cluster/hosts.ini
|
||||
ansible-playbook reset.yml
|
||||
|
||||
@@ -1,13 +1,25 @@
|
||||
---
|
||||
|
||||
- hosts: k3s_cluster
|
||||
gather_facts: yes
|
||||
become: yes
|
||||
- name: Reset k3s cluster
|
||||
hosts: k3s_cluster
|
||||
gather_facts: true
|
||||
roles:
|
||||
- role: reset
|
||||
become: true
|
||||
- role: raspberrypi
|
||||
vars: {state: absent}
|
||||
become: true
|
||||
vars: { state: absent }
|
||||
post_tasks:
|
||||
- name: Reboot and wait for node to come back up
|
||||
reboot:
|
||||
become: true
|
||||
ansible.builtin.reboot:
|
||||
reboot_command: "{{ custom_reboot_command | default(omit) }}"
|
||||
reboot_timeout: 3600
|
||||
|
||||
- name: Revert changes to Proxmox cluster
|
||||
hosts: proxmox
|
||||
gather_facts: true
|
||||
become: true
|
||||
remote_user: "{{ proxmox_lxc_ssh_user }}"
|
||||
roles:
|
||||
- role: reset_proxmox_lxc
|
||||
when: proxmox_lxc_configure
|
||||
|
||||
@@ -0,0 +1,8 @@
|
||||
---
|
||||
argument_specs:
|
||||
main:
|
||||
short_description: Manage the downloading of K3S binaries
|
||||
options:
|
||||
k3s_version:
|
||||
description: The desired version of K3S
|
||||
required: true
|
||||
@@ -1,36 +1,46 @@
|
||||
---
|
||||
|
||||
- name: Download k3s binary x64
|
||||
get_url:
|
||||
ansible.builtin.get_url:
|
||||
url: https://github.com/k3s-io/k3s/releases/download/{{ k3s_version }}/k3s
|
||||
checksum: sha256:https://github.com/k3s-io/k3s/releases/download/{{ k3s_version }}/sha256sum-amd64.txt
|
||||
dest: /usr/local/bin/k3s
|
||||
owner: root
|
||||
group: root
|
||||
mode: 0755
|
||||
mode: "0755"
|
||||
register: k3s_download_x64
|
||||
retries: 5
|
||||
delay: 10
|
||||
until: k3s_download_x64 is succeeded
|
||||
when: ansible_facts.architecture == "x86_64"
|
||||
|
||||
- name: Download k3s binary arm64
|
||||
get_url:
|
||||
ansible.builtin.get_url:
|
||||
url: https://github.com/k3s-io/k3s/releases/download/{{ k3s_version }}/k3s-arm64
|
||||
checksum: sha256:https://github.com/k3s-io/k3s/releases/download/{{ k3s_version }}/sha256sum-arm64.txt
|
||||
dest: /usr/local/bin/k3s
|
||||
owner: root
|
||||
group: root
|
||||
mode: 0755
|
||||
mode: "0755"
|
||||
register: k3s_download_arm64
|
||||
retries: 5
|
||||
delay: 10
|
||||
until: k3s_download_arm64 is succeeded
|
||||
when:
|
||||
- ( ansible_facts.architecture is search("arm") and
|
||||
ansible_facts.userspace_bits == "64" ) or
|
||||
ansible_facts.architecture is search("aarch64")
|
||||
- ( ansible_facts.architecture is search("arm") and ansible_facts.userspace_bits == "64" )
|
||||
or ansible_facts.architecture is search("aarch64")
|
||||
|
||||
- name: Download k3s binary armhf
|
||||
get_url:
|
||||
ansible.builtin.get_url:
|
||||
url: https://github.com/k3s-io/k3s/releases/download/{{ k3s_version }}/k3s-armhf
|
||||
checksum: sha256:https://github.com/k3s-io/k3s/releases/download/{{ k3s_version }}/sha256sum-arm.txt
|
||||
dest: /usr/local/bin/k3s
|
||||
owner: root
|
||||
group: root
|
||||
mode: 0755
|
||||
mode: "0755"
|
||||
register: k3s_download_armhf
|
||||
retries: 5
|
||||
delay: 10
|
||||
until: k3s_download_armhf is succeeded
|
||||
when:
|
||||
- ansible_facts.architecture is search("arm")
|
||||
- ansible_facts.userspace_bits == "32"
|
||||
|
||||
@@ -1,12 +0,0 @@
|
||||
---
|
||||
ansible_user: root
|
||||
server_init_args: >-
|
||||
{% if groups['master'] | length > 1 %}
|
||||
{% if ansible_hostname == hostvars[groups['master'][0]]['ansible_hostname'] %}
|
||||
--cluster-init
|
||||
{% else %}
|
||||
--server https://{{ hostvars[groups['master'][0]].k3s_node_ip }}:6443
|
||||
{% endif %}
|
||||
--token {{ k3s_token }}
|
||||
{% endif %}
|
||||
{{ extra_server_args | default('') }}
|
||||
@@ -1,28 +0,0 @@
|
||||
---
|
||||
# Download logs of k3s-init.service from the nodes to localhost.
|
||||
# Note that log_destination must be set.
|
||||
|
||||
- name: Fetch k3s-init.service logs
|
||||
ansible.builtin.command:
|
||||
cmd: journalctl --all --unit=k3s-init.service
|
||||
changed_when: false
|
||||
register: k3s_init_log
|
||||
|
||||
- name: Create {{ log_destination }}
|
||||
delegate_to: localhost
|
||||
run_once: true
|
||||
become: false
|
||||
ansible.builtin.file:
|
||||
path: "{{ log_destination }}"
|
||||
state: directory
|
||||
mode: "0755"
|
||||
|
||||
- name: Store logs to {{ log_destination }}
|
||||
delegate_to: localhost
|
||||
become: false
|
||||
ansible.builtin.template:
|
||||
src: content.j2
|
||||
dest: "{{ log_destination }}/k3s-init@{{ ansible_hostname }}.log"
|
||||
mode: 0644
|
||||
vars:
|
||||
content: "{{ k3s_init_log.stdout }}"
|
||||
@@ -1,199 +0,0 @@
|
||||
---
|
||||
|
||||
- name: Clean previous runs of k3s-init
|
||||
systemd:
|
||||
name: k3s-init
|
||||
state: stopped
|
||||
failed_when: false
|
||||
|
||||
- name: Clean previous runs of k3s-init
|
||||
command: systemctl reset-failed k3s-init
|
||||
failed_when: false
|
||||
changed_when: false
|
||||
args:
|
||||
warn: false # The ansible systemd module does not support reset-failed
|
||||
|
||||
- name: Create manifests directory on first master
|
||||
file:
|
||||
path: /var/lib/rancher/k3s/server/manifests
|
||||
state: directory
|
||||
owner: root
|
||||
group: root
|
||||
mode: 0644
|
||||
when: ansible_hostname == hostvars[groups['master'][0]]['ansible_hostname']
|
||||
|
||||
- name: Copy vip rbac manifest to first master
|
||||
template:
|
||||
src: "vip.rbac.yaml.j2"
|
||||
dest: "/var/lib/rancher/k3s/server/manifests/vip-rbac.yaml"
|
||||
owner: root
|
||||
group: root
|
||||
mode: 0644
|
||||
when: ansible_hostname == hostvars[groups['master'][0]]['ansible_hostname']
|
||||
|
||||
- name: Copy vip manifest to first master
|
||||
template:
|
||||
src: "vip.yaml.j2"
|
||||
dest: "/var/lib/rancher/k3s/server/manifests/vip.yaml"
|
||||
owner: root
|
||||
group: root
|
||||
mode: 0644
|
||||
when: ansible_hostname == hostvars[groups['master'][0]]['ansible_hostname']
|
||||
|
||||
# these will be copied and installed now, then tested later and apply config
|
||||
- name: Copy metallb namespace to first master
|
||||
template:
|
||||
src: "metallb.namespace.j2"
|
||||
dest: "/var/lib/rancher/k3s/server/manifests/metallb-namespace.yaml"
|
||||
owner: root
|
||||
group: root
|
||||
mode: 0644
|
||||
when: ansible_hostname == hostvars[groups['master'][0]]['ansible_hostname']
|
||||
|
||||
- name: Copy metallb namespace to first master
|
||||
template:
|
||||
src: "metallb.crds.j2"
|
||||
dest: "/var/lib/rancher/k3s/server/manifests/metallb-crds.yaml"
|
||||
owner: root
|
||||
group: root
|
||||
mode: 0644
|
||||
when: ansible_hostname == hostvars[groups['master'][0]]['ansible_hostname']
|
||||
|
||||
- name: Init cluster inside the transient k3s-init service
|
||||
command:
|
||||
cmd: "systemd-run -p RestartSec=2 \
|
||||
-p Restart=on-failure \
|
||||
--unit=k3s-init \
|
||||
k3s server {{ server_init_args }}"
|
||||
creates: "{{ systemd_dir }}/k3s.service"
|
||||
|
||||
- name: Verification
|
||||
block:
|
||||
- name: Verify that all nodes actually joined (check k3s-init.service if this fails)
|
||||
command:
|
||||
cmd: k3s kubectl get nodes -l "node-role.kubernetes.io/master=true" -o=jsonpath="{.items[*].metadata.name}"
|
||||
register: nodes
|
||||
until: nodes.rc == 0 and (nodes.stdout.split() | length) == (groups['master'] | length)
|
||||
retries: "{{ retry_count | default(20) }}"
|
||||
delay: 10
|
||||
changed_when: false
|
||||
always:
|
||||
- name: Save logs of k3s-init.service
|
||||
include_tasks: fetch_k3s_init_logs.yml
|
||||
when: log_destination
|
||||
vars:
|
||||
log_destination: >-
|
||||
{{ lookup('ansible.builtin.env', 'ANSIBLE_K3S_LOG_DIR', default=False) }}
|
||||
- name: Kill the temporary service used for initialization
|
||||
systemd:
|
||||
name: k3s-init
|
||||
state: stopped
|
||||
failed_when: false
|
||||
when: not ansible_check_mode
|
||||
|
||||
- name: Copy K3s service file
|
||||
register: k3s_service
|
||||
template:
|
||||
src: "k3s.service.j2"
|
||||
dest: "{{ systemd_dir }}/k3s.service"
|
||||
owner: root
|
||||
group: root
|
||||
mode: 0644
|
||||
|
||||
- name: Enable and check K3s service
|
||||
systemd:
|
||||
name: k3s
|
||||
daemon_reload: yes
|
||||
state: restarted
|
||||
enabled: yes
|
||||
|
||||
- name: Wait for node-token
|
||||
wait_for:
|
||||
path: /var/lib/rancher/k3s/server/node-token
|
||||
|
||||
- name: Register node-token file access mode
|
||||
stat:
|
||||
path: /var/lib/rancher/k3s/server
|
||||
register: p
|
||||
|
||||
- name: Change file access node-token
|
||||
file:
|
||||
path: /var/lib/rancher/k3s/server
|
||||
mode: "g+rx,o+rx"
|
||||
|
||||
- name: Read node-token from master
|
||||
slurp:
|
||||
src: /var/lib/rancher/k3s/server/node-token
|
||||
register: node_token
|
||||
|
||||
- name: Store Master node-token
|
||||
set_fact:
|
||||
token: "{{ node_token.content | b64decode | regex_replace('\n', '') }}"
|
||||
|
||||
- name: Restore node-token file access
|
||||
file:
|
||||
path: /var/lib/rancher/k3s/server
|
||||
mode: "{{ p.stat.mode }}"
|
||||
|
||||
- name: Create directory .kube
|
||||
file:
|
||||
path: ~{{ ansible_user }}/.kube
|
||||
state: directory
|
||||
owner: "{{ ansible_user }}"
|
||||
mode: "u=rwx,g=rx,o="
|
||||
|
||||
- name: Copy config file to user home directory
|
||||
copy:
|
||||
src: /etc/rancher/k3s/k3s.yaml
|
||||
dest: ~{{ ansible_user }}/.kube/config
|
||||
remote_src: yes
|
||||
owner: "{{ ansible_user }}"
|
||||
mode: "u=rw,g=,o="
|
||||
|
||||
- name: Configure kubectl cluster to {{ endpoint_url }}
|
||||
command: >-
|
||||
k3s kubectl config set-cluster default
|
||||
--server={{ endpoint_url }}
|
||||
--kubeconfig ~{{ ansible_user }}/.kube/config
|
||||
changed_when: true
|
||||
vars:
|
||||
endpoint_url: >-
|
||||
https://{{ apiserver_endpoint | ansible.utils.ipwrap }}:6443
|
||||
# Deactivated linter rules:
|
||||
# - jinja[invalid]: As of version 6.6.0, ansible-lint complains that the input to ipwrap
|
||||
# would be undefined. This will not be the case during playbook execution.
|
||||
# noqa jinja[invalid]
|
||||
|
||||
- name: Create kubectl symlink
|
||||
file:
|
||||
src: /usr/local/bin/k3s
|
||||
dest: /usr/local/bin/kubectl
|
||||
state: link
|
||||
|
||||
- name: Create crictl symlink
|
||||
file:
|
||||
src: /usr/local/bin/k3s
|
||||
dest: /usr/local/bin/crictl
|
||||
state: link
|
||||
|
||||
- name: Get contents of manifests folder
|
||||
find:
|
||||
paths: /var/lib/rancher/k3s/server/manifests
|
||||
file_type: file
|
||||
register: k3s_server_manifests
|
||||
|
||||
- name: Get sub dirs of manifests folder
|
||||
find:
|
||||
paths: /var/lib/rancher/k3s/server/manifests
|
||||
file_type: directory
|
||||
register: k3s_server_manifests_directories
|
||||
|
||||
- name: Remove manifests and folders that are only needed for bootstrapping cluster so k3s doesn't auto apply on start
|
||||
file:
|
||||
path: "{{ item.path }}"
|
||||
state: absent
|
||||
with_items:
|
||||
- "{{ k3s_server_manifests.files }}"
|
||||
- "{{ k3s_server_manifests_directories.files }}"
|
||||
loop_control:
|
||||
label: "{{ item.path }}"
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,6 +0,0 @@
|
||||
apiVersion: v1
|
||||
kind: Namespace
|
||||
metadata:
|
||||
name: metallb-system
|
||||
labels:
|
||||
app: metallb
|
||||
@@ -1,32 +0,0 @@
|
||||
apiVersion: v1
|
||||
kind: ServiceAccount
|
||||
metadata:
|
||||
name: kube-vip
|
||||
namespace: kube-system
|
||||
---
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
kind: ClusterRole
|
||||
metadata:
|
||||
annotations:
|
||||
rbac.authorization.kubernetes.io/autoupdate: "true"
|
||||
name: system:kube-vip-role
|
||||
rules:
|
||||
- apiGroups: [""]
|
||||
resources: ["services", "services/status", "nodes", "endpoints"]
|
||||
verbs: ["list","get","watch", "update"]
|
||||
- apiGroups: ["coordination.k8s.io"]
|
||||
resources: ["leases"]
|
||||
verbs: ["list", "get", "watch", "update", "create"]
|
||||
---
|
||||
kind: ClusterRoleBinding
|
||||
apiVersion: rbac.authorization.k8s.io/v1
|
||||
metadata:
|
||||
name: system:kube-vip-binding
|
||||
roleRef:
|
||||
apiGroup: rbac.authorization.k8s.io
|
||||
kind: ClusterRole
|
||||
name: system:kube-vip-role
|
||||
subjects:
|
||||
- kind: ServiceAccount
|
||||
name: kube-vip
|
||||
namespace: kube-system
|
||||
@@ -0,0 +1,3 @@
|
||||
---
|
||||
# Name of the master group
|
||||
group_name_master: master
|
||||
@@ -1,16 +0,0 @@
|
||||
---
|
||||
|
||||
- name: Copy K3s service file
|
||||
template:
|
||||
src: "k3s.service.j2"
|
||||
dest: "{{ systemd_dir }}/k3s-node.service"
|
||||
owner: root
|
||||
group: root
|
||||
mode: 0755
|
||||
|
||||
- name: Enable and check K3s service
|
||||
systemd:
|
||||
name: k3s-node
|
||||
daemon_reload: yes
|
||||
state: restarted
|
||||
enabled: yes
|
||||
@@ -1,3 +0,0 @@
|
||||
---
|
||||
# Timeout to wait for MetalLB services to come up
|
||||
metal_lb_available_timeout: 120s
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user