From 78ac139a7611d2cdd4369ac3191c95610cee7f0a Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Mon, 3 Aug 2026 15:29:41 +0300 Subject: [PATCH 01/30] feature: fsmount --- .gitignore | 1 + kubemarine/core/resources.py | 2 + kubemarine/fsmount.py | 168 ++++++++++++++++++ kubemarine/patches/__init__.py | 30 +++- kubemarine/procedures/install.py | 14 +- .../resources/configurations/defaults.yaml | 12 ++ .../schemas/definitions/services.json | 3 + .../schemas/definitions/services/fsmount.json | 69 +++++++ kubemarine/resources/scripts/zram.sh | 5 + kubemarine/system.py | 11 +- kubemarine/templates/zram-setup.service.j2 | 15 ++ 11 files changed, 327 insertions(+), 3 deletions(-) create mode 100644 kubemarine/fsmount.py create mode 100644 kubemarine/resources/schemas/definitions/services/fsmount.json create mode 100644 kubemarine/resources/scripts/zram.sh create mode 100644 kubemarine/templates/zram-setup.service.j2 diff --git a/.gitignore b/.gitignore index 5584798f9..45c8125eb 100644 --- a/.gitignore +++ b/.gitignore @@ -83,3 +83,4 @@ Icon Network Trash Folder Temporary Items .apdisk +.claude/* diff --git a/kubemarine/core/resources.py b/kubemarine/core/resources.py index 175fb12fd..1b8761a5b 100644 --- a/kubemarine/core/resources.py +++ b/kubemarine/core/resources.py @@ -33,6 +33,7 @@ import kubemarine.keepalived import kubemarine.kubernetes import kubemarine.kubernetes_accounts +import kubemarine.fsmount import kubemarine.modprobe import kubemarine.packages import kubemarine.plugins @@ -485,6 +486,7 @@ def enrichment_functions(self) -> List[c.EnrichmentFunction]: kubemarine.system.verify_inventory, kubemarine.system.enrich_etc_hosts, kubemarine.modprobe.enrich_kernel_modules, + kubemarine.fsmount.enrich_inventory, # Calculate some differences between previous and new inventory # Depends on kubemarine.packages.enrich_inventory diff --git a/kubemarine/fsmount.py b/kubemarine/fsmount.py new file mode 100644 index 000000000..92eea9a83 --- /dev/null +++ b/kubemarine/fsmount.py @@ -0,0 +1,168 @@ +# Copyright 2021-2022 NetCracker Technology Corporation +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +import io +import os +from typing import List, Union + +from jinja2 import Template + +from kubemarine.core import utils +from kubemarine.core.cluster import KubernetesCluster, EnrichmentStage, enrichment +from kubemarine.core.group import NodeGroup + + +@enrichment(EnrichmentStage.FULL) +def enrich_inventory(cluster: KubernetesCluster) -> None: + fsmount_list: List[dict] = cluster.inventory.get('services', {}).get('fsmount', []) + # JSON schema + for i, item in enumerate(fsmount_list): + path = ['services', 'fsmount', i] + if not isinstance(item, dict): + raise Exception(f"fsmount item at {utils.pretty_path(path)} must be a mapping") + + for required in ('name', 'device', 'path'): + if not item.get(required): + raise Exception(f"{required!r} is required for fsmount item at {utils.pretty_path(path)}") + + if not item.get('template', {}).get('source'): + raise Exception(f"'template.source' is required for fsmount item at {utils.pretty_path(path)}") + if not item.get('template', {}).get('destination'): + raise Exception(f"'template.destination' is required for fsmount item at {utils.pretty_path(path)}") + + ## TODO: SKIP + #if item.get('groups') is None and item.get('nodes') is None: + # item['groups'] = ['control-plane', 'worker', 'balancer'] + + preparation_script = item.get('preparation_script') + if preparation_script is not None: + ext_path = utils.get_external_resource_path(preparation_script) + if not os.path.isfile(ext_path) and not os.path.isfile(utils.get_internal_resource_path(preparation_script)): + raise Exception( + f"'preparation_script' file {preparation_script!r} not found " + f"for fsmount item at {utils.pretty_path(path)}") + + if item.get('nodes') is not None: + all_nodes_names = cluster.nodes['all'].get_nodes_names() + unknown_nodes = set(item['nodes']) - set(all_nodes_names) + if unknown_nodes: + cluster.log.warning( + f"Unknown node names {', '.join(map(repr, unknown_nodes))} " + f"provided for fsmount item {item['name']!r}.") + + +def _get_applicable_items(cluster: KubernetesCluster, node: NodeGroup) -> List[dict]: + fsmount_list: List[dict] = cluster.inventory.get('services', {}).get('fsmount', []) + applicable = [] + for item in fsmount_list: + groups: Union[List[str], None] = item.get('groups') + nodes: Union[List[str], None] = item.get('nodes') + group = cluster.create_group_from_groups_nodes_names(groups or [], nodes or []) + if group.has_node(node.get_node_name()): + applicable.append(item) + return applicable + + +def _render_unit(item: dict) -> str: + template_source = item['template']['source'] + ext_path = utils.get_external_resource_path(template_source) + if os.path.isfile(ext_path): + template_content = utils.read_external(template_source) + else: + template_content = utils.read_internal(template_source) + + return Template(template_content).render( + name=item['name'], + device=item['device'], + path=item['path'], + size=item.get('size', ''), + type=item.get('type', ''), + ) + + +def is_mounted(group: NodeGroup) -> bool: + cluster: KubernetesCluster = group.cluster + results = group.sudo("cat /proc/mounts") + + for node in group.get_ordered_members_list(): + applicable = _get_applicable_items(cluster, node) + if not applicable: + continue + host = node.get_host() + mounts_output = results[host].stdout + for item in applicable: + if item['path'].rstrip('/') not in mounts_output: + cluster.log.debug(f"Mount path {item['path']!r} not found in /proc/mounts on {host}") + return False + + return True + + +def setup_fsmount(group: NodeGroup) -> bool: + cluster: KubernetesCluster = group.cluster + logger = cluster.log + + if is_mounted(group): + # ??? + logger.debug("Skipped - all required filesystems are already mounted") + return False + + changed = False + for node in group.get_ordered_members_list(): + applicable = _get_applicable_items(cluster, node) + if not applicable: + continue + + host = node.get_host() + mounts_output = node.sudo("cat /proc/mounts")[host].stdout + + for item in applicable: + if item['path'].rstrip('/') in mounts_output: + logger.debug(f"Skipping fsmount item {item['name']!r} on {node.get_node_name()}: already mounted") + continue + + preparation_script = item.get('preparation_script') + if preparation_script: + logger.debug(f"Running preparation script for fsmount item {item['name']!r} on {node.get_node_name()}") + ext_path = utils.get_external_resource_path(preparation_script) + if os.path.isfile(ext_path): + script_content = utils.read_external(preparation_script) + else: + script_content = utils.read_internal(preparation_script) + remote_path = f"/tmp/fsmount_{item['name']}_prep.sh" + node.put(io.StringIO(script_content), remote_path, sudo=True) + node.sudo(f"chmod +x {remote_path}") + prep_result = node.sudo(f"bash {remote_path}", warn=True) + node.sudo(f"rm -f {remote_path}") + if prep_result[host].return_code != 0: + logger.warning( + f"Preparation script for fsmount item {item['name']!r} " + f"failed on {node.get_node_name()}, skipping." + f"The output is: {prep_result[host]}") + continue + + unit_content = _render_unit(item) + unit_destination = item['template']['destination'] + unit_name = unit_destination.rsplit('/', 1)[-1] + unit_dir = unit_destination.rsplit('/', 1)[0] + + logger.debug(f"Setting up fsmount item {item['name']!r} on {node.get_node_name()}") + node.sudo(f"mkdir -p {unit_dir}") + node.put(io.StringIO(unit_content), unit_destination, backup=True, sudo=True) + utils.dump_file(cluster, unit_content, f'fsmount/{item["name"]}_{node.get_node_name()}.service') + node.sudo("systemctl daemon-reload") + node.sudo(f"systemctl enable --now {unit_name}") + changed = True + + return changed diff --git a/kubemarine/patches/__init__.py b/kubemarine/patches/__init__.py index 7958d24dd..25bccfd62 100644 --- a/kubemarine/patches/__init__.py +++ b/kubemarine/patches/__init__.py @@ -21,9 +21,37 @@ from typing import List -from kubemarine.core.patch import Patch +from kubemarine import fsmount +from kubemarine.core.action import Action +from kubemarine.core.patch import Patch, RegularPatch +from kubemarine.core.resources import DynamicResources + + +class _FsmountPatchAction(Action): + def __init__(self) -> None: + super().__init__('fsmount') + + def run(self, res: DynamicResources) -> None: + cluster = res.cluster() + group = cluster.nodes['all'] + group.call(fsmount.setup_fsmount) + + +class _FsmountPatch(RegularPatch): + def __init__(self) -> None: + super().__init__('fsmount') + + @property + def action(self) -> Action: + return _FsmountPatchAction() + + @property + def description(self) -> str: + return "Sets up fsmount items (e.g. zram) with default settings on all existing cluster nodes." + patches: List[Patch] = [ + _FsmountPatch(), ] """ List of patches that is sorted according to the Patch.priority() before execution. diff --git a/kubemarine/procedures/install.py b/kubemarine/procedures/install.py index 7e0d34cc7..c7a6f265b 100755 --- a/kubemarine/procedures/install.py +++ b/kubemarine/procedures/install.py @@ -22,7 +22,7 @@ from kubemarine.core.errors import KME from kubemarine import ( system, sysctl, haproxy, keepalived, kubernetes, plugins, - kubernetes_accounts, selinux, thirdparties, audit, coredns, cri, packages, apparmor, modprobe + kubernetes_accounts, selinux, thirdparties, audit, coredns, cri, packages, apparmor, modprobe, fsmount ) from kubemarine.core import flow, utils, summary from kubemarine.core.group import NodeGroup, RunnersGroupResult, CollectorCallback @@ -140,6 +140,15 @@ def system_prepare_system_sysctl(group: NodeGroup) -> None: group.call(system.verify_sysctl) +@_applicable_for_new_nodes_with_roles('all') +def system_prepare_system_fsmount(group: NodeGroup) -> None: + cluster: KubernetesCluster = group.cluster + if not cluster.inventory.get('services', {}).get('fsmount'): + cluster.log.debug("Skipped - no fsmount items defined in config file") + return + group.call(fsmount.setup_fsmount) + + @_applicable_for_new_nodes_with_roles('all') def system_prepare_system_setup_selinux(group: NodeGroup) -> None: system.configure_sensitive_service(group, selinux.setup_selinux) @@ -535,6 +544,7 @@ def overview(cluster: KubernetesCluster) -> None: "disable_swap": system_prepare_system_disable_swap, "modprobe": system_prepare_system_modprobe, "sysctl": system_prepare_system_sysctl, + "fsmount": system_prepare_system_fsmount, "audit": { "install": system_install_audit, "configure": system_prepare_audit, @@ -577,9 +587,11 @@ def overview(cluster: KubernetesCluster) -> None: # This is done before `prepare.system.audit`. system.reboot_nodes: [ "prepare.system.modprobe", + "prepare.system.fsmount", "prepare.system.audit" ], system.verify_system: [ + "prepare.system.fsmount", "prepare.system.audit" ], # Some checks can be done only at the end when the necessary services are configured. diff --git a/kubemarine/resources/configurations/defaults.yaml b/kubemarine/resources/configurations/defaults.yaml index 3367ffc64..ce96a32b6 100644 --- a/kubemarine/resources/configurations/defaults.yaml +++ b/kubemarine/resources/configurations/defaults.yaml @@ -163,6 +163,18 @@ services: debian: *modprobe-default-modules ubuntu26.04: *modprobe-default-modules + fsmount: + - name: zram + size: 1G + type: ext4 + device: /dev/zram0 + path: /var/log/pods + template: + source: templates/zram-setup.service.j2 + destination: /etc/systemd/system/zram-setup.service + preparation_script: resources/scripts/zram.sh + groups: [control-plane, worker] + sysctl: net.ipv4.ip_nonlocal_bind: value: 1 diff --git a/kubemarine/resources/schemas/definitions/services.json b/kubemarine/resources/schemas/definitions/services.json index cb616a7e6..b44e6410b 100644 --- a/kubemarine/resources/schemas/definitions/services.json +++ b/kubemarine/resources/schemas/definitions/services.json @@ -42,6 +42,9 @@ "modprobe": { "$ref": "services/modprobe.json" }, + "fsmount": { + "$ref": "services/fsmount.json" + }, "sysctl": { "$ref": "services/sysctl.json" }, diff --git a/kubemarine/resources/schemas/definitions/services/fsmount.json b/kubemarine/resources/schemas/definitions/services/fsmount.json new file mode 100644 index 000000000..440d8c1c9 --- /dev/null +++ b/kubemarine/resources/schemas/definitions/services/fsmount.json @@ -0,0 +1,69 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema", + "type": "array", + "description": "List of filesystem mount configurations to be set up on cluster nodes via systemd units.", + "items": { + "oneOf": [ + {"$ref": "#/definitions/FsmountItem"}, + {"$ref": "../common/utils.json#/definitions/ListMergingSymbol"} + ] + }, + "definitions": { + "FsmountItem": { + "type": "object", + "properties": { + "name": { + "type": "string", + "description": "Name of the fsmount item, used for identification" + }, + "device": { + "type": "string", + "description": "The device file to mount (e.g. /dev/sdd1, /dev/zram0)" + }, + "path": { + "type": "string", + "description": "The mount point path on the node" + }, + "size": { + "type": "string", + "description": "Size of the filesystem (optional, e.g. 1G)" + }, + "type": { + "type": "string", + "description": "Filesystem type (e.g. ext4, tmpfs)" + }, + "template": { + "type": "object", + "description": "Systemd unit template configuration", + "properties": { + "source": { + "type": "string", + "description": "Path to the Jinja2 template for the systemd unit" + }, + "destination": { + "type": "string", + "description": "Absolute path on the node where the rendered unit file is placed" + } + }, + "required": ["source", "destination"], + "additionalProperties": false + }, + "preparation_script": { + "type": "string", + "description": "Path to a shell script run before creating the filesystem. If it fails, this item is skipped." + }, + "groups": { + "$ref": "../common/node_ref.json#/definitions/Roles", + "default": ["worker", "control-plane", "balancer"], + "description": "The list of node roles where this mount should be applied" + }, + "nodes": { + "$ref": "../common/node_ref.json#/definitions/Names", + "description": "The list of node names where this mount should be applied" + } + }, + "required": ["name", "device", "path", "template"], + "additionalProperties": false + } + } +} diff --git a/kubemarine/resources/scripts/zram.sh b/kubemarine/resources/scripts/zram.sh new file mode 100644 index 000000000..3c4ddebd4 --- /dev/null +++ b/kubemarine/resources/scripts/zram.sh @@ -0,0 +1,5 @@ +#!/bin/bash + +apt install -yq linux-modules-extra-$(uname -r) +modprobe zram +mkdir -p /var/log/pods diff --git a/kubemarine/system.py b/kubemarine/system.py index 238f666e3..0fb8b44aa 100644 --- a/kubemarine/system.py +++ b/kubemarine/system.py @@ -23,7 +23,7 @@ from dateutil.parser import parse from ordered_set import OrderedSet -from kubemarine import selinux, apparmor, sysctl, modprobe +from kubemarine import selinux, apparmor, sysctl, modprobe, fsmount from kubemarine.core import utils, static from kubemarine.core.cluster import KubernetesCluster, EnrichmentStage, enrichment from kubemarine.core.executor import RunnersResult, Token, GenericResult, Callback, RawExecutor @@ -571,6 +571,15 @@ def verify_system(cluster: KubernetesCluster) -> None: else: log.debug('Kernel parameters verification skipped - origin setup task was not completed') + if cluster.is_task_completed('prepare.system.fsmount'): + log.debug("Verifying fsmount...") + fsmount_ok = fsmount.is_mounted(group) + if not fsmount_ok: + raise Exception("Required filesystem mounts are not configured") + log.debug("Required filesystem mounts are configured") + else: + log.debug('Fsmount verification skipped - origin setup task was not completed') + def verify_sysctl(group: NodeGroup) -> None: cluster: KubernetesCluster = group.cluster diff --git a/kubemarine/templates/zram-setup.service.j2 b/kubemarine/templates/zram-setup.service.j2 new file mode 100644 index 000000000..60a7fa541 --- /dev/null +++ b/kubemarine/templates/zram-setup.service.j2 @@ -0,0 +1,15 @@ +[Unit] +Description=Create zram device +DefaultDependencies=no +Before=local-fs.target + +[Service] +Type=oneshot +RemainAfterExit=yes +ExecStart=/usr/sbin/modprobe zram +ExecStart=/usr/sbin/zramctl {{ device }} --size {{ size }} --algorithm zstd +ExecStart=/usr/sbin/mkfs.{{ type }} -F {{ device }} +ExecStart=/usr/bin/mount {{ device }} {{ path }} + +[Install] +WantedBy=local-fs.target From 9e4e6425ff178fe1e932220cbe5ec0ca5c43c1e3 Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Tue, 4 Aug 2026 09:56:01 +0300 Subject: [PATCH 02/30] feature: fsmount --- kubemarine/fsmount.py | 21 ++++----------------- 1 file changed, 4 insertions(+), 17 deletions(-) diff --git a/kubemarine/fsmount.py b/kubemarine/fsmount.py index 92eea9a83..92579a535 100644 --- a/kubemarine/fsmount.py +++ b/kubemarine/fsmount.py @@ -26,24 +26,11 @@ @enrichment(EnrichmentStage.FULL) def enrich_inventory(cluster: KubernetesCluster) -> None: fsmount_list: List[dict] = cluster.inventory.get('services', {}).get('fsmount', []) - # JSON schema for i, item in enumerate(fsmount_list): - path = ['services', 'fsmount', i] - if not isinstance(item, dict): - raise Exception(f"fsmount item at {utils.pretty_path(path)} must be a mapping") - - for required in ('name', 'device', 'path'): - if not item.get(required): - raise Exception(f"{required!r} is required for fsmount item at {utils.pretty_path(path)}") - - if not item.get('template', {}).get('source'): - raise Exception(f"'template.source' is required for fsmount item at {utils.pretty_path(path)}") - if not item.get('template', {}).get('destination'): - raise Exception(f"'template.destination' is required for fsmount item at {utils.pretty_path(path)}") - - ## TODO: SKIP - #if item.get('groups') is None and item.get('nodes') is None: - # item['groups'] = ['control-plane', 'worker', 'balancer'] + path: List[Union[str, int]] = ['services', 'fsmount', i] + + if item.get('groups') is None and item.get('nodes') is None: + continue preparation_script = item.get('preparation_script') if preparation_script is not None: From 646d9d2d30734c6759c9dd212b304a1dded05554 Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Wed, 5 Aug 2026 11:29:49 +0300 Subject: [PATCH 03/30] feature: migration --- kubemarine/fsmount.py | 1 - kubemarine/patches/__init__.py | 36 +++++++++++++++++++++++++--- kubemarine/resources/scripts/zram.sh | 9 ++++++- 3 files changed, 41 insertions(+), 5 deletions(-) diff --git a/kubemarine/fsmount.py b/kubemarine/fsmount.py index 92579a535..36978ae24 100644 --- a/kubemarine/fsmount.py +++ b/kubemarine/fsmount.py @@ -101,7 +101,6 @@ def setup_fsmount(group: NodeGroup) -> bool: logger = cluster.log if is_mounted(group): - # ??? logger.debug("Skipped - all required filesystems are already mounted") return False diff --git a/kubemarine/patches/__init__.py b/kubemarine/patches/__init__.py index 25bccfd62..562032ece 100644 --- a/kubemarine/patches/__init__.py +++ b/kubemarine/patches/__init__.py @@ -21,7 +21,7 @@ from typing import List -from kubemarine import fsmount +from kubemarine import fsmount, kubernetes, system from kubemarine.core.action import Action from kubemarine.core.patch import Patch, RegularPatch from kubemarine.core.resources import DynamicResources @@ -33,8 +33,38 @@ def __init__(self) -> None: def run(self, res: DynamicResources) -> None: cluster = res.cluster() - group = cluster.nodes['all'] - group.call(fsmount.setup_fsmount) + first_control_plane = cluster.nodes['control-plane'].get_first_member() + timeout_config = cluster.inventory['globals']['expect']['pods']['kubernetes'] + + for node in cluster.nodes['all'].get_ordered_members_list(): + node_name = node.get_node_name() + node_config = node.get_config() + is_k8s_node = 'control-plane' in node_config['roles'] or 'worker' in node_config['roles'] + + if is_k8s_node: + cluster.log.debug(f"Draining node {node_name!r} before fsmount setup") + first_control_plane.sudo( + kubernetes.prepare_drain_command(cluster, node_name, disable_eviction=False), + warn=True, pty=True) + + applicable = fsmount._get_applicable_items(cluster, node) + for item in applicable: + mount_path = item['path'].rstrip('/') + cluster.log.debug(f"Removing files in {mount_path!r} on {node_name!r}") + node.sudo(f"rm -rf {mount_path}/*") + + node.call(fsmount.setup_fsmount) + + cluster.log.debug(f"Rebooting node {node_name!r} after fsmount setup") + system.perform_group_reboot(node) + + if is_k8s_node: + cluster.log.debug(f"Uncordoning node {node_name!r} after reboot") + first_control_plane.wait_command_successful( + f"kubectl uncordon {node_name}", + hide=False, pty=True, + timeout=timeout_config['timeout'], + retries=timeout_config['retries']) class _FsmountPatch(RegularPatch): diff --git a/kubemarine/resources/scripts/zram.sh b/kubemarine/resources/scripts/zram.sh index 3c4ddebd4..b9149cfb2 100644 --- a/kubemarine/resources/scripts/zram.sh +++ b/kubemarine/resources/scripts/zram.sh @@ -1,5 +1,12 @@ #!/bin/bash -apt install -yq linux-modules-extra-$(uname -r) +if command -v apt-get &>/dev/null; then + apt-get install -yq linux-modules-extra-$(uname -r) +elif command -v dnf &>/dev/null; then + dnf install -y kernel-modules-extra +elif command -v yum &>/dev/null; then + yum install -y kernel-modules-extra +fi + modprobe zram mkdir -p /var/log/pods From cda2073e3d9251bc31c31f9ff33a06e60fdd4b0d Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Wed, 5 Aug 2026 13:40:57 +0300 Subject: [PATCH 04/30] feature: check_paas --- kubemarine/fsmount.py | 39 +++++++++++++++++++++++++++-- kubemarine/procedures/check_paas.py | 19 ++++++++++++-- 2 files changed, 54 insertions(+), 4 deletions(-) diff --git a/kubemarine/fsmount.py b/kubemarine/fsmount.py index 36978ae24..558b0fa13 100644 --- a/kubemarine/fsmount.py +++ b/kubemarine/fsmount.py @@ -78,6 +78,16 @@ def _render_unit(item: dict) -> str: ) +def _parse_mounts(mounts_output: str) -> dict: + """Parse /proc/mounts into {mountpoint: fstype}.""" + result = {} + for line in mounts_output.splitlines(): + parts = line.split() + if len(parts) >= 3: + result[parts[1]] = parts[2] + return result + + def is_mounted(group: NodeGroup) -> bool: cluster: KubernetesCluster = group.cluster results = group.sudo("cat /proc/mounts") @@ -87,15 +97,40 @@ def is_mounted(group: NodeGroup) -> bool: if not applicable: continue host = node.get_host() - mounts_output = results[host].stdout + mounts = _parse_mounts(results[host].stdout) for item in applicable: - if item['path'].rstrip('/') not in mounts_output: + if item['path'].rstrip('/') not in mounts: cluster.log.debug(f"Mount path {item['path']!r} not found in /proc/mounts on {host}") return False return True +def check_mounts(group: NodeGroup) -> List[str]: + """Return a list of human-readable error strings for missing or wrong-type mounts.""" + cluster: KubernetesCluster = group.cluster + results = group.sudo("cat /proc/mounts") + errors = [] + + for node in group.get_ordered_members_list(): + applicable = _get_applicable_items(cluster, node) + if not applicable: + continue + host = node.get_host() + node_name = node.get_node_name() + mounts = _parse_mounts(results[host].stdout) + for item in applicable: + mount_path = item['path'].rstrip('/') + expected_type = item.get('type', '') + if mount_path not in mounts: + errors.append(f"{node_name}: {mount_path!r} is not mounted") + elif expected_type and mounts[mount_path] != expected_type: + errors.append( + f"{node_name}: {mount_path!r} has fstype {mounts[mount_path]!r}, expected {expected_type!r}") + + return errors + + def setup_fsmount(group: NodeGroup) -> bool: cluster: KubernetesCluster = group.cluster logger = cluster.log diff --git a/kubemarine/procedures/check_paas.py b/kubemarine/procedures/check_paas.py index 9dd6b3148..ee059e083 100755 --- a/kubemarine/procedures/check_paas.py +++ b/kubemarine/procedures/check_paas.py @@ -29,7 +29,7 @@ from kubemarine import ( packages as pckgs, system, selinux, etcd, thirdparties, apparmor, kubernetes, sysctl, audit, - plugins, modprobe, admission + plugins, modprobe, admission, fsmount ) from kubemarine.core.cluster import KubernetesCluster from kubemarine.core.group import NodeGroup, CollectorCallback, GroupResultException @@ -1033,6 +1033,18 @@ def verify_modprobe_rules(cluster: KubernetesCluster) -> None: f"the differences manually and make changes on the appropriate nodes.") +def verify_fsmount(cluster: KubernetesCluster) -> None: + with TestCase(cluster, '236', "System", "Filesystem mounts") as tc: + group = cluster.make_group_from_roles(['control-plane', 'worker']) + errors = fsmount.check_mounts(group) + if not errors: + tc.success(results='mounted') + else: + raise TestFailure('invalid', + hint="Filesystem mount issues found:\n" + "\n".join(f" - {e}" for e in errors) + + "\nRun the fsmount task in the installation procedure to set them up.") + + def verify_sysctl_config(cluster: KubernetesCluster) -> None: """ This test compares the kernel parameters on the nodes @@ -1593,7 +1605,7 @@ def verify_kubernetes_version(cluster: KubernetesCluster) -> None: """ The method checks if used kubernetes version is deprecated in kubemarine """ - with TestCase(cluster, '225', "Kubernetes", "Version") as tc: + with TestCase(cluster, '235', "Kubernetes", "Version") as tc: target_version = cluster.inventory['services']['kubeadm']['kubernetesVersion'] if not kubernetes.verify_supported_version(target_version, cluster.log): raise TestWarn(f"Kubernetes version {target_version} is deprecated", @@ -1763,6 +1775,9 @@ def verify_apparmor_config(cluster: KubernetesCluster) -> None: 'modprobe': { 'rules': verify_modprobe_rules }, + 'fsmount': { + 'mounts': verify_fsmount + }, 'sysctl': { 'config': verify_sysctl_config }, From 20ba9ba27693c1234b4eff2729f324768db09fec Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Wed, 5 Aug 2026 14:15:47 +0300 Subject: [PATCH 05/30] feature: migration --- kubemarine/fsmount.py | 8 ++++---- kubemarine/patches/__init__.py | 2 +- 2 files changed, 5 insertions(+), 5 deletions(-) diff --git a/kubemarine/fsmount.py b/kubemarine/fsmount.py index 558b0fa13..79058499a 100644 --- a/kubemarine/fsmount.py +++ b/kubemarine/fsmount.py @@ -49,7 +49,7 @@ def enrich_inventory(cluster: KubernetesCluster) -> None: f"provided for fsmount item {item['name']!r}.") -def _get_applicable_items(cluster: KubernetesCluster, node: NodeGroup) -> List[dict]: +def get_applicable_items(cluster: KubernetesCluster, node: NodeGroup) -> List[dict]: fsmount_list: List[dict] = cluster.inventory.get('services', {}).get('fsmount', []) applicable = [] for item in fsmount_list: @@ -93,7 +93,7 @@ def is_mounted(group: NodeGroup) -> bool: results = group.sudo("cat /proc/mounts") for node in group.get_ordered_members_list(): - applicable = _get_applicable_items(cluster, node) + applicable = get_applicable_items(cluster, node) if not applicable: continue host = node.get_host() @@ -113,7 +113,7 @@ def check_mounts(group: NodeGroup) -> List[str]: errors = [] for node in group.get_ordered_members_list(): - applicable = _get_applicable_items(cluster, node) + applicable = get_applicable_items(cluster, node) if not applicable: continue host = node.get_host() @@ -141,7 +141,7 @@ def setup_fsmount(group: NodeGroup) -> bool: changed = False for node in group.get_ordered_members_list(): - applicable = _get_applicable_items(cluster, node) + applicable = get_applicable_items(cluster, node) if not applicable: continue diff --git a/kubemarine/patches/__init__.py b/kubemarine/patches/__init__.py index 562032ece..386c9e7b0 100644 --- a/kubemarine/patches/__init__.py +++ b/kubemarine/patches/__init__.py @@ -47,7 +47,7 @@ def run(self, res: DynamicResources) -> None: kubernetes.prepare_drain_command(cluster, node_name, disable_eviction=False), warn=True, pty=True) - applicable = fsmount._get_applicable_items(cluster, node) + applicable = fsmount.get_applicable_items(cluster, node) for item in applicable: mount_path = item['path'].rstrip('/') cluster.log.debug(f"Removing files in {mount_path!r} on {node_name!r}") From 53a9abfeaf50cd20a0fcd510776787807868be87 Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Wed, 5 Aug 2026 15:15:24 +0300 Subject: [PATCH 06/30] feature: fsmount --- kubemarine/fsmount.py | 18 +++++++++++++++--- kubemarine/templates/zram-setup.service.j2 | 2 +- 2 files changed, 16 insertions(+), 4 deletions(-) diff --git a/kubemarine/fsmount.py b/kubemarine/fsmount.py index 79058499a..5b71f842e 100644 --- a/kubemarine/fsmount.py +++ b/kubemarine/fsmount.py @@ -20,7 +20,7 @@ from kubemarine.core import utils from kubemarine.core.cluster import KubernetesCluster, EnrichmentStage, enrichment -from kubemarine.core.group import NodeGroup +from kubemarine.core.group import NodeGroup, CollectorCallback @enrichment(EnrichmentStage.FULL) @@ -109,7 +109,14 @@ def is_mounted(group: NodeGroup) -> bool: def check_mounts(group: NodeGroup) -> List[str]: """Return a list of human-readable error strings for missing or wrong-type mounts.""" cluster: KubernetesCluster = group.cluster - results = group.sudo("cat /proc/mounts") + + mounts_collector = CollectorCallback(cluster) + zramctl_collector = CollectorCallback(cluster) + defer = group.new_defer() + defer.sudo("cat /proc/mounts", callback=mounts_collector) + defer.sudo("zramctl --output-all", warn=True, callback=zramctl_collector) + defer.flush() + errors = [] for node in group.get_ordered_members_list(): @@ -118,7 +125,9 @@ def check_mounts(group: NodeGroup) -> List[str]: continue host = node.get_host() node_name = node.get_node_name() - mounts = _parse_mounts(results[host].stdout) + mounts = _parse_mounts(mounts_collector.result[host].stdout) + zramctl_output = zramctl_collector.result[host].stdout + for item in applicable: mount_path = item['path'].rstrip('/') expected_type = item.get('type', '') @@ -128,6 +137,9 @@ def check_mounts(group: NodeGroup) -> List[str]: errors.append( f"{node_name}: {mount_path!r} has fstype {mounts[mount_path]!r}, expected {expected_type!r}") + if item['device'].startswith('/dev/zram') and mount_path not in zramctl_output: + errors.append(f"{node_name}: {mount_path!r} not found in zramctl output") + return errors diff --git a/kubemarine/templates/zram-setup.service.j2 b/kubemarine/templates/zram-setup.service.j2 index 60a7fa541..5f02066ee 100644 --- a/kubemarine/templates/zram-setup.service.j2 +++ b/kubemarine/templates/zram-setup.service.j2 @@ -8,7 +8,7 @@ Type=oneshot RemainAfterExit=yes ExecStart=/usr/sbin/modprobe zram ExecStart=/usr/sbin/zramctl {{ device }} --size {{ size }} --algorithm zstd -ExecStart=/usr/sbin/mkfs.{{ type }} -F {{ device }} +ExecStart=/usr/sbin/mkfs.{{ type }} -m 0 -O ^has_journal -F {{ device }} ExecStart=/usr/bin/mount {{ device }} {{ path }} [Install] From 7d3110a365d8f8e0033ebe635a7361a868ab1a7c Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Wed, 5 Aug 2026 16:36:28 +0300 Subject: [PATCH 07/30] feature: kubelet --- kubemarine/patches/__init__.py | 40 ++++++++++++++++++- .../resources/configurations/defaults.yaml | 2 + 2 files changed, 41 insertions(+), 1 deletion(-) diff --git a/kubemarine/patches/__init__.py b/kubemarine/patches/__init__.py index 386c9e7b0..0e6dc4270 100644 --- a/kubemarine/patches/__init__.py +++ b/kubemarine/patches/__init__.py @@ -25,6 +25,42 @@ from kubemarine.core.action import Action from kubemarine.core.patch import Patch, RegularPatch from kubemarine.core.resources import DynamicResources +from kubemarine.kubernetes import components + + +class _KubeletLogLimitsPatchAction(Action): + def __init__(self) -> None: + super().__init__('kubelet-log-limits', recreate_inventory=True) + + def run(self, res: DynamicResources) -> None: + inventory = res.inventory() + kubelet = inventory.setdefault('services', {}).setdefault('kubeadm_kubelet', {}) + changed = False + for key, value in (('containerLogMaxSize', '5Mi'), ('containerLogMaxFiles', 2)): + if key not in kubelet: + kubelet[key] = value + changed = True + + if not changed: + res.logger().info("containerLogMaxSize/containerLogMaxFiles already set, skipping.") + return + + cluster = res.cluster() + cluster.nodes['control-plane'].call(components.reconfigure_components, components=['kubelet']) + + +class _KubeletLogLimitsPatch(RegularPatch): + def __init__(self) -> None: + super().__init__('kubelet-log-limits') + + @property + def action(self) -> Action: + return _KubeletLogLimitsPatchAction() + + @property + def description(self) -> str: + return ("Adds containerLogMaxSize: 5Mi and containerLogMaxFiles: 2 " + "to services.kubeadm_kubelet and applies the new kubelet config on all nodes.") class _FsmountPatchAction(Action): @@ -77,10 +113,12 @@ def action(self) -> Action: @property def description(self) -> str: - return "Sets up fsmount items (e.g. zram) with default settings on all existing cluster nodes." + return ("Sets up zram volume for /var/log/pods with default settings on all existing cluster nodes. " + "Be careful it erases all of the data inside /var/log/pods folder") patches: List[Patch] = [ + _KubeletLogLimitsPatch(), _FsmountPatch(), ] """ diff --git a/kubemarine/resources/configurations/defaults.yaml b/kubemarine/resources/configurations/defaults.yaml index ce96a32b6..ee42c7a52 100644 --- a/kubemarine/resources/configurations/defaults.yaml +++ b/kubemarine/resources/configurations/defaults.yaml @@ -40,6 +40,8 @@ services: cgroupDriver: systemd serializeImagePulls: false tlsCipherSuites: [TLS_ECDHE_ECDSA_WITH_AES_128_GCM_SHA256,TLS_ECDHE_RSA_WITH_AES_128_GCM_SHA256,TLS_ECDHE_ECDSA_WITH_CHACHA20_POLY1305,TLS_ECDHE_RSA_WITH_AES_256_GCM_SHA384,TLS_ECDHE_RSA_WITH_CHACHA20_POLY1305,TLS_ECDHE_ECDSA_WITH_AES_256_GCM_SHA384,TLS_RSA_WITH_AES_256_GCM_SHA384,TLS_RSA_WITH_AES_128_GCM_SHA256] + containerLogMaxSize: 5Mi + containerLogMaxFiles: 2 kubeadm_kube-proxy: apiVersion: kubeproxy.config.k8s.io/v1alpha1 kind: KubeProxyConfiguration From ee3c0fc2c7126d9bc0e26bae2dad7532a79fb1cc Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Wed, 5 Aug 2026 16:51:33 +0300 Subject: [PATCH 08/30] feature: documentation --- docs/public/Installation.md | 68 +++++++++++++++++++++++++++++++++++++ 1 file changed, 68 insertions(+) diff --git a/docs/public/Installation.md b/docs/public/Installation.md index 93c3e1b50..085258b71 100644 --- a/docs/public/Installation.md +++ b/docs/public/Installation.md @@ -45,6 +45,7 @@ This section provides information about the inventory, features, and steps for i - [thirdparties](#thirdparties) - [CRI](#cri) - [modprobe](#modprobe) + - [fsmount](#fsmount) - [sysctl](#sysctl) - [audit](#audit) - [Kubernetes Policy](#audit-kubernetes-policy) @@ -2497,6 +2498,73 @@ The following settings are supported in the extended format: **Warning**: If changes to the hosts `modprobe` configurations are detected, a reboot is scheduled. After the reboot, the new parameters are validated to match the expected configuration. +#### fsmount + +*Installation task*: `prepare.system.fsmount` + +*Can cause reboot*: No + +*Can restart service*: No + +*Overwrite files*: Yes, the rendered systemd unit file is uploaded to the path specified in `template.destination`, backup is created + +*OS specific*: No + +The `services.fsmount` section configures filesystems to be mounted on cluster nodes via systemd units. +Each item defines a device to format and mount, a Jinja2 template for the systemd unit that performs the setup, and an optional preparation script that runs before the unit is installed. + +By default, a zram-based filesystem is configured to back `/var/log/pods` on all `control-plane` and `worker` nodes: + +```yaml +services: + fsmount: + - name: zram + size: 1G + type: ext4 + device: /dev/zram0 + path: /var/log/pods + template: + source: templates/zram-setup.service.j2 + destination: /etc/systemd/system/zram-setup.service + preparation_script: resources/scripts/zram.sh + groups: [control-plane, worker] +``` + +You can add additional entries to the list or override the default item using [List Merge Strategy](#list-merge-strategy). + +The following parameters are supported for each item: + +|Parameter|Mandatory|Default Value|Description| +|---|---|---|---| +|**name**|**yes**| |Identifier for this mount entry, used in logs and dump filenames.| +|**device**|**yes**| |The device file to mount (e.g. `/dev/sdd1`, `/dev/zram0`).| +|**path**|**yes**| |The mount point path on the node.| +|**template.source**|**yes**| |Path to the Jinja2 template for the systemd unit. Can be an internal resource path (relative to the Kubemarine package) or an external absolute path.| +|**template.destination**|**yes**| |Absolute path on the node where the rendered unit file is placed.| +|**size**|no| |Size of the filesystem, passed to the systemd unit template (e.g. `1G`). Required for virtual devices such as zram.| +|**type**|no| |Filesystem type passed to the template (e.g. `ext4`, `tmpfs`).| +|**preparation_script**|no| |Path to a shell script executed before the systemd unit is installed. If the script exits with a non-zero code, the mount entry is skipped for that node. Can be an internal resource path or an external absolute path.| +|**groups**|no|`[control-plane, worker, balancer]`|The list of node roles where this mount should be applied.| +|**nodes**|no| |The list of specific node names where this mount should be applied.| + +**Notes**: +* You can specify `groups` and `nodes` at the same time; both are merged to determine the target nodes. +* If neither `groups` nor `nodes` is specified, the mount is applied to all nodes. +* If the mount path is already present in `/proc/mounts`, the entry is skipped for that node. +* The preparation script is uploaded to the node, executed, and then removed. It is intended for operations such as loading kernel modules or installing required packages. + +**Warning**: The `preparation_script` failure is non-fatal. If the script fails, only that specific mount entry is skipped on the affected node; other entries continue to be processed. + +The template receives the following variables for rendering: + +|Variable|Source| +|---|---| +|`name`|`name` field| +|`device`|`device` field| +|`path`|`path` field| +|`size`|`size` field (empty string if not set)| +|`type`|`type` field (empty string if not set)| + #### sysctl *Installation task*: `prepare.system.sysctl` From e1bac8e8185e956c1fba398fe534f2d18644afcf Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Thu, 6 Aug 2026 08:15:30 +0300 Subject: [PATCH 09/30] feature: docs --- docs/public/Installation.md | 24 ++++++++++++------------ 1 file changed, 12 insertions(+), 12 deletions(-) diff --git a/docs/public/Installation.md b/docs/public/Installation.md index 085258b71..6e2437fc6 100644 --- a/docs/public/Installation.md +++ b/docs/public/Installation.md @@ -2534,18 +2534,18 @@ You can add additional entries to the list or override the default item using [L The following parameters are supported for each item: -|Parameter|Mandatory|Default Value|Description| -|---|---|---|---| -|**name**|**yes**| |Identifier for this mount entry, used in logs and dump filenames.| -|**device**|**yes**| |The device file to mount (e.g. `/dev/sdd1`, `/dev/zram0`).| -|**path**|**yes**| |The mount point path on the node.| -|**template.source**|**yes**| |Path to the Jinja2 template for the systemd unit. Can be an internal resource path (relative to the Kubemarine package) or an external absolute path.| -|**template.destination**|**yes**| |Absolute path on the node where the rendered unit file is placed.| -|**size**|no| |Size of the filesystem, passed to the systemd unit template (e.g. `1G`). Required for virtual devices such as zram.| -|**type**|no| |Filesystem type passed to the template (e.g. `ext4`, `tmpfs`).| -|**preparation_script**|no| |Path to a shell script executed before the systemd unit is installed. If the script exits with a non-zero code, the mount entry is skipped for that node. Can be an internal resource path or an external absolute path.| -|**groups**|no|`[control-plane, worker, balancer]`|The list of node roles where this mount should be applied.| -|**nodes**|no| |The list of specific node names where this mount should be applied.| +|Parameter|Mandatory|Description| +|---|---|---| +|**name**|**yes**|Identifier for this mount entry, used in logs and dump filenames.| +|**device**|**yes**|The device file to mount (e.g. `/dev/sdd1`, `/dev/zram0`).| +|**path**|**yes**|The mount point path on the node.| +|**template.source**|**yes**|Path to the Jinja2 template for the systemd unit. Can be an internal resource path (relative to the Kubemarine package) or an external absolute path.| +|**template.destination**|**yes**|Absolute path on the node where the rendered unit file is placed.| +|**size**|no|Size of the filesystem, passed to the systemd unit template (e.g. `1G`). Required for virtual devices such as zram.| +|**type**|no|Filesystem type passed to the template (e.g. `ext4`, `tmpfs`).| +|**preparation_script**|no|Path to a shell script executed before the systemd unit is installed. If the script exits with a non-zero code, the mount entry is skipped for that node. Can be an internal resource path or an external absolute path.| +|**groups**|no|The list of node roles where this mount should be applied.| +|**nodes**|no|The list of specific node names where this mount should be applied.| **Notes**: * You can specify `groups` and `nodes` at the same time; both are merged to determine the target nodes. From c72cdb2e37a1c3cd65c334aa6cce2ecebfa6c218 Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Thu, 6 Aug 2026 10:04:05 +0300 Subject: [PATCH 10/30] feature: docs --- docs/public/Installation.md | 167 +++++++++++++++++++++--------------- 1 file changed, 99 insertions(+), 68 deletions(-) diff --git a/docs/public/Installation.md b/docs/public/Installation.md index 6e2437fc6..865c1132b 100644 --- a/docs/public/Installation.md +++ b/docs/public/Installation.md @@ -45,8 +45,8 @@ This section provides information about the inventory, features, and steps for i - [thirdparties](#thirdparties) - [CRI](#cri) - [modprobe](#modprobe) - - [fsmount](#fsmount) - [sysctl](#sysctl) + - [fsmount](#fsmount) - [audit](#audit) - [Kubernetes Policy](#audit-kubernetes-policy) - [Daemon](#audit-daemon) @@ -2498,73 +2498,6 @@ The following settings are supported in the extended format: **Warning**: If changes to the hosts `modprobe` configurations are detected, a reboot is scheduled. After the reboot, the new parameters are validated to match the expected configuration. -#### fsmount - -*Installation task*: `prepare.system.fsmount` - -*Can cause reboot*: No - -*Can restart service*: No - -*Overwrite files*: Yes, the rendered systemd unit file is uploaded to the path specified in `template.destination`, backup is created - -*OS specific*: No - -The `services.fsmount` section configures filesystems to be mounted on cluster nodes via systemd units. -Each item defines a device to format and mount, a Jinja2 template for the systemd unit that performs the setup, and an optional preparation script that runs before the unit is installed. - -By default, a zram-based filesystem is configured to back `/var/log/pods` on all `control-plane` and `worker` nodes: - -```yaml -services: - fsmount: - - name: zram - size: 1G - type: ext4 - device: /dev/zram0 - path: /var/log/pods - template: - source: templates/zram-setup.service.j2 - destination: /etc/systemd/system/zram-setup.service - preparation_script: resources/scripts/zram.sh - groups: [control-plane, worker] -``` - -You can add additional entries to the list or override the default item using [List Merge Strategy](#list-merge-strategy). - -The following parameters are supported for each item: - -|Parameter|Mandatory|Description| -|---|---|---| -|**name**|**yes**|Identifier for this mount entry, used in logs and dump filenames.| -|**device**|**yes**|The device file to mount (e.g. `/dev/sdd1`, `/dev/zram0`).| -|**path**|**yes**|The mount point path on the node.| -|**template.source**|**yes**|Path to the Jinja2 template for the systemd unit. Can be an internal resource path (relative to the Kubemarine package) or an external absolute path.| -|**template.destination**|**yes**|Absolute path on the node where the rendered unit file is placed.| -|**size**|no|Size of the filesystem, passed to the systemd unit template (e.g. `1G`). Required for virtual devices such as zram.| -|**type**|no|Filesystem type passed to the template (e.g. `ext4`, `tmpfs`).| -|**preparation_script**|no|Path to a shell script executed before the systemd unit is installed. If the script exits with a non-zero code, the mount entry is skipped for that node. Can be an internal resource path or an external absolute path.| -|**groups**|no|The list of node roles where this mount should be applied.| -|**nodes**|no|The list of specific node names where this mount should be applied.| - -**Notes**: -* You can specify `groups` and `nodes` at the same time; both are merged to determine the target nodes. -* If neither `groups` nor `nodes` is specified, the mount is applied to all nodes. -* If the mount path is already present in `/proc/mounts`, the entry is skipped for that node. -* The preparation script is uploaded to the node, executed, and then removed. It is intended for operations such as loading kernel modules or installing required packages. - -**Warning**: The `preparation_script` failure is non-fatal. If the script fails, only that specific mount entry is skipped on the affected node; other entries continue to be processed. - -The template receives the following variables for rendering: - -|Variable|Source| -|---|---| -|`name`|`name` field| -|`device`|`device` field| -|`path`|`path` field| -|`size`|`size` field (empty string if not set)| -|`type`|`type` field (empty string if not set)| - #### sysctl *Installation task*: `prepare.system.sysctl` @@ -2651,6 +2584,104 @@ The following settings are supported in the extended format: **Warning**: If the changes to the hosts `sysctl` configurations are detected, a reboot is scheduled. After the reboot, the new parameters are validated to match the expected configuration. +#### fsmount + +*Installation task*: `prepare.system.fsmount` + +*Can cause reboot*: No + +*Can restart service*: No + +*Overwrite files*: Yes, the rendered systemd unit file is uploaded to the path specified in `template.destination`, backup is created + +*OS specific*: No + +The `services.fsmount` section configures filesystems to be mounted on cluster nodes via systemd units. +Each item defines a device to format and mount, a Jinja2 template for the systemd unit that performs the setup, and an optional preparation script that runs before the unit is installed. + +By default, a zram-based filesystem is configured to back `/var/log/pods` on all `control-plane` and `worker` nodes: + +```yaml +services: + fsmount: + - name: zram + size: 1G + type: ext4 + device: /dev/zram0 + path: /var/log/pods + template: + source: templates/zram-setup.service.j2 + destination: /etc/systemd/system/zram-setup.service + preparation_script: resources/scripts/zram.sh + groups: [control-plane, worker] +``` + +You can add additional entries to the list or override the default item using [List Merge Strategy](#list-merge-strategy). + +The following parameters are supported for each item: + +|Parameter|Mandatory|Description| +|---|---|---| +|**name**|**yes**|Identifier for this mount entry, used in logs and dump filenames.| +|**device**|**yes**|The device file to mount (e.g. `/dev/sdd1`, `/dev/zram0`).| +|**path**|**yes**|The mount point path on the node.| +|**template.source**|**yes**|Path to the Jinja2 template for the systemd unit. Can be an internal resource path (relative to the Kubemarine package) or an external absolute path.| +|**template.destination**|**yes**|Absolute path on the node where the rendered unit file is placed.| +|**size**|no|Size of the filesystem, passed to the systemd unit template (e.g. `1G`). Required for virtual devices such as zram.| +|**type**|no|Filesystem type passed to the template (e.g. `ext4`, `tmpfs`).| +|**preparation_script**|no|Path to a shell script executed before the systemd unit is installed. If the script exits with a non-zero code, the mount entry is skipped for that node. Can be an internal resource path or an external absolute path.| +|**groups**|no|The list of node roles where this mount should be applied.| +|**nodes**|no|The list of specific node names where this mount should be applied.| + +**Notes**: +* You can specify `groups` and `nodes` at the same time; both are merged to determine the target nodes. +* If neither `groups` nor `nodes` is specified, the mount is applied to all nodes. +* If the mount path is already present in `/proc/mounts`, the entry is skipped for that node. +* The preparation script is uploaded to the node, executed, and then removed. It is intended for operations such as loading kernel modules or installing required packages. + +**Warning**: The `preparation_script` failure is non-fatal. If the script fails, only that specific mount entry is skipped on the affected node; other entries continue to be processed. + +The template receives the following variables for rendering: + +|Variable|Source| +|---|---| +|`name`|`name` field| +|`device`|`device` field| +|`path`|`path` field| +|`size`|`size` field (empty string if not set)| +|`type`|`type` field (empty string if not set)| + +There is an example how to mount particular disk on some cluster node. The cluster.yaml part: + +```yaml +services: + fsmount: + - name: mydisk + type: ext4 + device: /dev/disk/by-uuid/f05934af-6699-49d1-bd1c-01bebf3e586f + path: /mnt/data + template: + source: templates/mnt-data.mount.j2 + destination: /etc/systemd/system/mnt-data.mount + nodes: ['worker-1'] +``` + +The template file: + +```conf +[Unit] +Description=Mount My Disk + +[Mount] +What={{ device }} +Where={{ path }} +Type={{ type }} +Options=defaults + +[Install] +WantedBy=multi-user.target +``` + #### audit ##### Audit Kubernetes Policy From 00871ccfebfdd330f3fe23ddeb8431e27a4df7f1 Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Thu, 6 Aug 2026 10:16:59 +0300 Subject: [PATCH 11/30] feature: check_paas docs --- docs/public/Kubecheck.md | 15 +++++++++++++++ 1 file changed, 15 insertions(+) diff --git a/docs/public/Kubecheck.md b/docs/public/Kubecheck.md index 41637950b..bda73b4be 100644 --- a/docs/public/Kubecheck.md +++ b/docs/public/Kubecheck.md @@ -67,6 +67,7 @@ This section provides information about the Kubecheck functionality. - [215 Firewalld Status](#215-firewalld-status) - [216 Swap State](#216-swap-state) - [217 Modprobe Rules](#217-modprobe-rules) + - [236 Filesystem Mounts](#236-filesystem-mounts) - [218 Time Difference](#218-time-difference) - [219 Health Status ETCD](#219-health-status-etcd) - [220 Control Plane Configuration Status](#220-control-plane-configuration-status) @@ -657,6 +658,20 @@ The test verifies that swap is disabled on all nodes in the cluster, otherwise t The test compares the modprobe rules on the nodes with the rules specified in the inventory or with default rules. If rules does not match, the test will fail. +##### 236 Filesystem Mounts + +*Task*: `services.system.fsmount.mounts` + +The test verifies that all filesystem mounts defined in `services.fsmount` are correctly configured on `control-plane` and `worker` nodes. For each item the following is checked: + +- The mount point defined in `path` is present in `/proc/mounts`. +- The filesystem type in `/proc/mounts` matches the `type` field specified in the inventory. +- For zram-backed devices (device path starts with `/dev/zram`), the mount point is also present in the output of `zramctl --output-all`. + +If any check fails, the test reports each individual issue with the affected node name and mount path. + +**Note**: If no `services.fsmount` items are applicable to a node, that node is silently skipped. + ##### 218 Time Difference *Task*: `services.system.time` From 8a2e5a96a25852ccd7a25e64ce8da7d8d00a3b6e Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Tue, 11 Aug 2026 09:51:26 +0300 Subject: [PATCH 12/30] feat: script --- kubemarine/resources/scripts/zram.sh | 1 - 1 file changed, 1 deletion(-) diff --git a/kubemarine/resources/scripts/zram.sh b/kubemarine/resources/scripts/zram.sh index b9149cfb2..472d8c935 100644 --- a/kubemarine/resources/scripts/zram.sh +++ b/kubemarine/resources/scripts/zram.sh @@ -8,5 +8,4 @@ elif command -v yum &>/dev/null; then yum install -y kernel-modules-extra fi -modprobe zram mkdir -p /var/log/pods From d9b5a11ea299ed7c36d15c1419cb2e0dbffdf84a Mon Sep 17 00:00:00 2001 From: Dipali Said Date: Thu, 13 Aug 2026 10:36:14 +0530 Subject: [PATCH 13/30] Docs: CPCAP-14161 --- docs/public/Installation.md | 32 ++++++++++++++++++++++++-------- docs/public/Kubecheck.md | 12 ++++++------ 2 files changed, 30 insertions(+), 14 deletions(-) diff --git a/docs/public/Installation.md b/docs/public/Installation.md index 865c1132b..bbc7658ce 100644 --- a/docs/public/Installation.md +++ b/docs/public/Installation.md @@ -2588,18 +2588,34 @@ The following settings are supported in the extended format: *Installation task*: `prepare.system.fsmount` -*Can cause reboot*: No +*Can cause a reboot*: **No** -*Can restart service*: No +*Can restart a service*: **No** -*Overwrite files*: Yes, the rendered systemd unit file is uploaded to the path specified in `template.destination`, backup is created +*Overwrites files*: Yes – the rendered systemd unit file is uploaded to the location defined in `template.destination`. A backup of the previous file is kept. -*OS specific*: No +*OS‑specific*: No + +The `services.fsmount` section allows you to define additional filesystems that should be formatted and mounted on cluster nodes using **systemd** unit files. Each list entry describes: + +- **name** – an identifier used in logs and dump file names. +- **device** – the block device (e.g. `/dev/sdd1` or `/dev/zram0`). +- **path** – the mount point on the node. +- **type** – the filesystem type (`ext4`, `xfs`, `tmpfs`, …). Required for virtual devices such as *zram*. +- **size** – the desired size (e.g. `1G`). Required for virtual devices. +- **template.source** – path to the Jinja2 template that renders the systemd unit. +- **template.destination** – absolute path on the target node where the rendered unit will be placed. +- **preparation_script** – optional script that runs on the node before the unit is installed. If the script exits with a non‑zero status the corresponding mount entry is skipped for that node. +- **groups** – optional list of node roles (`control-plane`, `worker`, `balancer`, …) this entry applies to. +- **nodes** – optional list of specific node names this entry applies to. + +If both `groups` and `nodes` are supplied they are merged; if neither is provided the mount is applied to **all** nodes. + +The mount is skipped on a node when the target path already appears in `/proc/mounts`. -The `services.fsmount` section configures filesystems to be mounted on cluster nodes via systemd units. -Each item defines a device to format and mount, a Jinja2 template for the systemd unit that performs the setup, and an optional preparation script that runs before the unit is installed. +**Warning** – a failure of the `preparation_script` is non‑fatal; only the affected mount entry is omitted while the rest of the configuration continues. -By default, a zram-based filesystem is configured to back `/var/log/pods` on all `control-plane` and `worker` nodes: +**Default configuration** ```yaml services: @@ -2616,7 +2632,7 @@ services: groups: [control-plane, worker] ``` -You can add additional entries to the list or override the default item using [List Merge Strategy](#list-merge-strategy). +You can add further entries to the list or replace the default one using the **list‑merge strategy** described earlier. The following parameters are supported for each item: diff --git a/docs/public/Kubecheck.md b/docs/public/Kubecheck.md index bda73b4be..6fa046d04 100644 --- a/docs/public/Kubecheck.md +++ b/docs/public/Kubecheck.md @@ -662,15 +662,15 @@ rules does not match, the test will fail. *Task*: `services.system.fsmount.mounts` -The test verifies that all filesystem mounts defined in `services.fsmount` are correctly configured on `control-plane` and `worker` nodes. For each item the following is checked: +This check validates that every filesystem mount defined in `services.fsmount` is correctly configured on `control‑plane` and `worker` nodes. For each entry the following conditions are verified: -- The mount point defined in `path` is present in `/proc/mounts`. -- The filesystem type in `/proc/mounts` matches the `type` field specified in the inventory. -- For zram-backed devices (device path starts with `/dev/zram`), the mount point is also present in the output of `zramctl --output-all`. +- The mount point specified by `path` appears in `/proc/mounts`. +- The filesystem type reported in `/proc/mounts` matches the `type` declared in the inventory. +- For ZRAM‑backed devices (i.e., the `device` value begins with `/dev/zram`), the mount point is also listed in the output of `zramctl --output‑all`. -If any check fails, the test reports each individual issue with the affected node name and mount path. +If any validation fails, the test reports the affected node and mount path with a detailed error message. -**Note**: If no `services.fsmount` items are applicable to a node, that node is silently skipped. +**Note**: Nodes that have no applicable `services.fsmount` entries are skipped silently. ##### 218 Time Difference From 4ca533952c54baab3ceef03f23b0366c261c06ea Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Mon, 17 Aug 2026 14:48:47 +0300 Subject: [PATCH 14/30] feature: rework --- docs/public/Installation.md | 71 +++----------- kubemarine/patches/__init__.py | 97 ------------------- .../resources/configurations/defaults.yaml | 14 +-- 3 files changed, 12 insertions(+), 170 deletions(-) diff --git a/docs/public/Installation.md b/docs/public/Installation.md index bbc7658ce..561009e55 100644 --- a/docs/public/Installation.md +++ b/docs/public/Installation.md @@ -2596,43 +2596,7 @@ The following settings are supported in the extended format: *OS‑specific*: No -The `services.fsmount` section allows you to define additional filesystems that should be formatted and mounted on cluster nodes using **systemd** unit files. Each list entry describes: - -- **name** – an identifier used in logs and dump file names. -- **device** – the block device (e.g. `/dev/sdd1` or `/dev/zram0`). -- **path** – the mount point on the node. -- **type** – the filesystem type (`ext4`, `xfs`, `tmpfs`, …). Required for virtual devices such as *zram*. -- **size** – the desired size (e.g. `1G`). Required for virtual devices. -- **template.source** – path to the Jinja2 template that renders the systemd unit. -- **template.destination** – absolute path on the target node where the rendered unit will be placed. -- **preparation_script** – optional script that runs on the node before the unit is installed. If the script exits with a non‑zero status the corresponding mount entry is skipped for that node. -- **groups** – optional list of node roles (`control-plane`, `worker`, `balancer`, …) this entry applies to. -- **nodes** – optional list of specific node names this entry applies to. - -If both `groups` and `nodes` are supplied they are merged; if neither is provided the mount is applied to **all** nodes. - -The mount is skipped on a node when the target path already appears in `/proc/mounts`. - -**Warning** – a failure of the `preparation_script` is non‑fatal; only the affected mount entry is omitted while the rest of the configuration continues. - -**Default configuration** - -```yaml -services: - fsmount: - - name: zram - size: 1G - type: ext4 - device: /dev/zram0 - path: /var/log/pods - template: - source: templates/zram-setup.service.j2 - destination: /etc/systemd/system/zram-setup.service - preparation_script: resources/scripts/zram.sh - groups: [control-plane, worker] -``` - -You can add further entries to the list or replace the default one using the **list‑merge strategy** described earlier. +The `services.fsmount` section allows you to define additional filesystems that should be formatted and mounted on cluster nodes using **systemd** unit files. The default configuration is empty. The following parameters are supported for each item: @@ -2667,35 +2631,22 @@ The template receives the following variables for rendering: |`size`|`size` field (empty string if not set)| |`type`|`type` field (empty string if not set)| -There is an example how to mount particular disk on some cluster node. The cluster.yaml part: +There is an example how to mount ZRAM disk on all cluster nodes. The cluster.yaml part: ```yaml services: fsmount: - - name: mydisk + - name: zram + enabled: false + size: 1G type: ext4 - device: /dev/disk/by-uuid/f05934af-6699-49d1-bd1c-01bebf3e586f - path: /mnt/data + device: /dev/zram0 + path: /var/log/pods template: - source: templates/mnt-data.mount.j2 - destination: /etc/systemd/system/mnt-data.mount - nodes: ['worker-1'] -``` - -The template file: - -```conf -[Unit] -Description=Mount My Disk - -[Mount] -What={{ device }} -Where={{ path }} -Type={{ type }} -Options=defaults - -[Install] -WantedBy=multi-user.target + source: templates/zram-setup.service.j2 + destination: /etc/systemd/system/zram-setup.service + preparation_script: resources/scripts/zram.sh + groups: [control-plane, worker] ``` #### audit diff --git a/kubemarine/patches/__init__.py b/kubemarine/patches/__init__.py index 0e6dc4270..d17bb939a 100644 --- a/kubemarine/patches/__init__.py +++ b/kubemarine/patches/__init__.py @@ -21,105 +21,8 @@ from typing import List -from kubemarine import fsmount, kubernetes, system -from kubemarine.core.action import Action -from kubemarine.core.patch import Patch, RegularPatch -from kubemarine.core.resources import DynamicResources -from kubemarine.kubernetes import components - - -class _KubeletLogLimitsPatchAction(Action): - def __init__(self) -> None: - super().__init__('kubelet-log-limits', recreate_inventory=True) - - def run(self, res: DynamicResources) -> None: - inventory = res.inventory() - kubelet = inventory.setdefault('services', {}).setdefault('kubeadm_kubelet', {}) - changed = False - for key, value in (('containerLogMaxSize', '5Mi'), ('containerLogMaxFiles', 2)): - if key not in kubelet: - kubelet[key] = value - changed = True - - if not changed: - res.logger().info("containerLogMaxSize/containerLogMaxFiles already set, skipping.") - return - - cluster = res.cluster() - cluster.nodes['control-plane'].call(components.reconfigure_components, components=['kubelet']) - - -class _KubeletLogLimitsPatch(RegularPatch): - def __init__(self) -> None: - super().__init__('kubelet-log-limits') - - @property - def action(self) -> Action: - return _KubeletLogLimitsPatchAction() - - @property - def description(self) -> str: - return ("Adds containerLogMaxSize: 5Mi and containerLogMaxFiles: 2 " - "to services.kubeadm_kubelet and applies the new kubelet config on all nodes.") - - -class _FsmountPatchAction(Action): - def __init__(self) -> None: - super().__init__('fsmount') - - def run(self, res: DynamicResources) -> None: - cluster = res.cluster() - first_control_plane = cluster.nodes['control-plane'].get_first_member() - timeout_config = cluster.inventory['globals']['expect']['pods']['kubernetes'] - - for node in cluster.nodes['all'].get_ordered_members_list(): - node_name = node.get_node_name() - node_config = node.get_config() - is_k8s_node = 'control-plane' in node_config['roles'] or 'worker' in node_config['roles'] - - if is_k8s_node: - cluster.log.debug(f"Draining node {node_name!r} before fsmount setup") - first_control_plane.sudo( - kubernetes.prepare_drain_command(cluster, node_name, disable_eviction=False), - warn=True, pty=True) - - applicable = fsmount.get_applicable_items(cluster, node) - for item in applicable: - mount_path = item['path'].rstrip('/') - cluster.log.debug(f"Removing files in {mount_path!r} on {node_name!r}") - node.sudo(f"rm -rf {mount_path}/*") - - node.call(fsmount.setup_fsmount) - - cluster.log.debug(f"Rebooting node {node_name!r} after fsmount setup") - system.perform_group_reboot(node) - - if is_k8s_node: - cluster.log.debug(f"Uncordoning node {node_name!r} after reboot") - first_control_plane.wait_command_successful( - f"kubectl uncordon {node_name}", - hide=False, pty=True, - timeout=timeout_config['timeout'], - retries=timeout_config['retries']) - - -class _FsmountPatch(RegularPatch): - def __init__(self) -> None: - super().__init__('fsmount') - - @property - def action(self) -> Action: - return _FsmountPatchAction() - - @property - def description(self) -> str: - return ("Sets up zram volume for /var/log/pods with default settings on all existing cluster nodes. " - "Be careful it erases all of the data inside /var/log/pods folder") - patches: List[Patch] = [ - _KubeletLogLimitsPatch(), - _FsmountPatch(), ] """ List of patches that is sorted according to the Patch.priority() before execution. diff --git a/kubemarine/resources/configurations/defaults.yaml b/kubemarine/resources/configurations/defaults.yaml index ee42c7a52..a65d84128 100644 --- a/kubemarine/resources/configurations/defaults.yaml +++ b/kubemarine/resources/configurations/defaults.yaml @@ -40,8 +40,6 @@ services: cgroupDriver: systemd serializeImagePulls: false tlsCipherSuites: [TLS_ECDHE_ECDSA_WITH_AES_128_GCM_SHA256,TLS_ECDHE_RSA_WITH_AES_128_GCM_SHA256,TLS_ECDHE_ECDSA_WITH_CHACHA20_POLY1305,TLS_ECDHE_RSA_WITH_AES_256_GCM_SHA384,TLS_ECDHE_RSA_WITH_CHACHA20_POLY1305,TLS_ECDHE_ECDSA_WITH_AES_256_GCM_SHA384,TLS_RSA_WITH_AES_256_GCM_SHA384,TLS_RSA_WITH_AES_128_GCM_SHA256] - containerLogMaxSize: 5Mi - containerLogMaxFiles: 2 kubeadm_kube-proxy: apiVersion: kubeproxy.config.k8s.io/v1alpha1 kind: KubeProxyConfiguration @@ -165,17 +163,7 @@ services: debian: *modprobe-default-modules ubuntu26.04: *modprobe-default-modules - fsmount: - - name: zram - size: 1G - type: ext4 - device: /dev/zram0 - path: /var/log/pods - template: - source: templates/zram-setup.service.j2 - destination: /etc/systemd/system/zram-setup.service - preparation_script: resources/scripts/zram.sh - groups: [control-plane, worker] + fsmount: [] sysctl: net.ipv4.ip_nonlocal_bind: From 816ae88ea07ab4d39dc3fcf091ee54796c15e485 Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Mon, 17 Aug 2026 14:52:49 +0300 Subject: [PATCH 15/30] feature: rework --- kubemarine/patches/__init__.py | 1 + 1 file changed, 1 insertion(+) diff --git a/kubemarine/patches/__init__.py b/kubemarine/patches/__init__.py index d17bb939a..7958d24dd 100644 --- a/kubemarine/patches/__init__.py +++ b/kubemarine/patches/__init__.py @@ -21,6 +21,7 @@ from typing import List +from kubemarine.core.patch import Patch patches: List[Patch] = [ ] From 7e8003766656bc84405ae9e5e255baed3f92c00a Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Mon, 17 Aug 2026 15:16:46 +0300 Subject: [PATCH 16/30] feature: mypy --- kubemarine/__main__.py | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/kubemarine/__main__.py b/kubemarine/__main__.py index 29b1781ab..2d001150f 100755 --- a/kubemarine/__main__.py +++ b/kubemarine/__main__.py @@ -50,8 +50,8 @@ if (1, 1) in ir[0] and 'tag:yaml.org,2002:float' in ir[1]: float_patched_resolver = (ir[1], ir[2], ir[3]) # Globally change behaviour of yaml.safe_load and yaml.dump - yaml.Dumper.add_implicit_resolver(*float_patched_resolver) # type: ignore[no-untyped-call] - yaml.SafeLoader.add_implicit_resolver(*float_patched_resolver) # type: ignore[no-untyped-call] + yaml.Dumper.add_implicit_resolver(*float_patched_resolver) # type: ignore[no-untyped-call, unused-ignore] + yaml.SafeLoader.add_implicit_resolver(*float_patched_resolver) # type: ignore[no-untyped-call, unused-ignore] break From e5719b207071713e69bf9bb4ad7e9a84911ad98c Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Wed, 19 Aug 2026 08:49:55 +0300 Subject: [PATCH 17/30] feature: new procedure --- docs/public/Maintenance.md | 51 ++++++++++++ kubemarine/__main__.py | 4 + kubemarine/core/resources.py | 1 + kubemarine/fsmount.py | 38 +++++++-- kubemarine/procedures/__init__.py | 2 + kubemarine/procedures/mount_fs.py | 95 ++++++++++++++++++++++ kubemarine/resources/schemas/mount_fs.json | 4 + 7 files changed, 186 insertions(+), 9 deletions(-) create mode 100644 kubemarine/procedures/mount_fs.py create mode 100644 kubemarine/resources/schemas/mount_fs.json diff --git a/docs/public/Maintenance.md b/docs/public/Maintenance.md index 892c48bc8..731f7f04a 100644 --- a/docs/public/Maintenance.md +++ b/docs/public/Maintenance.md @@ -14,6 +14,7 @@ This section describes the features and steps for performing maintenance procedu - [Reconfigure Procedure](#reconfigure-procedure) - [Manage PSS Procedure](#manage-pss-procedure) - [Reboot Procedure](#reboot-procedure) + - [Mount Filesystems Procedure](#mount-filesystems-procedure) - [Certificate Renew Procedure](#certificate-renew-procedure) - [Etcd Member Reunion](#etcd-member-reunion) - [Procedure Execution](#procedure-execution) @@ -1357,6 +1358,56 @@ nodes: ``` +## Mount Filesystems Procedure + +The `mount_fs` procedure sets up filesystems on cluster nodes according to a required `procedure.yaml`. +The format of `procedure.yaml` is the same as the `services.fsmount` list in `cluster.yaml` — an array of fsmount items specifying which filesystems to mount and on which nodes. + +For each node that has at least one applicable fsmount item, the procedure: + +1. Drains the node (if it is a `control-plane` or `worker` node) to safely evacuate workloads. +2. Removes existing data from each configured mount path to ensure a clean state. +3. Installs and enables the corresponding systemd mount units. +4. Reboots the node so that the new mounts are activated at the OS level. +5. Uncordons the node (if it is a `control-plane` or `worker` node) to make it schedulable again. + +Nodes that have no applicable items are skipped entirely. + +**Note**: Data inside the configured mount paths is erased before the mount is set up. Back up any important data before running this procedure. + +### Mount Filesystems Procedure Parameters + +The procedure accepts required positional argument with the path to the procedure inventory file. + +The JSON schema for procedure inventory is available by [URL](/kubemarine/resources/schemas/mount_fs.json?raw=1). +For more information, see [Validation by JSON Schemas](Installation.md#inventory-validation). + +The procedure inventory has the same structure as the `services.fsmount` section in `cluster.yaml`. +For a description of the available fields, refer to the [fsmount](Installation.md#fsmount) section in _Kubemarine Installation Procedure_. + +Example: + +```yaml +- name: zram-pods + device: /dev/zram0 + path: /var/log/pods + type: zram + size: 1G + template: + source: templates/zram.mount.j2 + destination: /etc/systemd/system/var-log-pods.mount + groups: + - control-plane + - worker +``` + +### Mount Filesystems Procedure Tasks Tree + +The `mount_fs` procedure executes the following sequence of tasks: + +* mount_filesystems +* overview + ## Certificate Renew Procedure The `cert_renew` procedure allows you to renew some certificates on an existing Kubernetes cluster. diff --git a/kubemarine/__main__.py b/kubemarine/__main__.py index 2d001150f..a26942260 100755 --- a/kubemarine/__main__.py +++ b/kubemarine/__main__.py @@ -97,6 +97,10 @@ 'description': "Renew certificates on Kubernetes cluster", 'group': 'maintenance' }, + 'mount_fs': { + 'description': "Mount filesystems described in the fsmount inventory section", + 'group': 'maintenance' + }, 'reboot': { 'description': "Reboot Kubernetes nodes", 'group': 'maintenance' diff --git a/kubemarine/core/resources.py b/kubemarine/core/resources.py index 1b8761a5b..f53b31aa9 100644 --- a/kubemarine/core/resources.py +++ b/kubemarine/core/resources.py @@ -409,6 +409,7 @@ def enrichment_functions(self) -> List[c.EnrichmentFunction]: kubemarine.cri.enrich_upgrade_inventory, kubemarine.plugins.nginx_ingress.cert_renew_enrichment, kubemarine.plugins.envoy_gateway.cert_renew_enrichment, + kubemarine.fsmount.enrich_procedure_inventory, kubemarine.sysctl.enrich_reconfigure_inventory, kubemarine.core.inventory.enrich_reconfigure_inventory, # Enrichment of procedure inventory should be finished at this step. diff --git a/kubemarine/fsmount.py b/kubemarine/fsmount.py index 5b71f842e..5cda987d3 100644 --- a/kubemarine/fsmount.py +++ b/kubemarine/fsmount.py @@ -23,6 +23,24 @@ from kubemarine.core.group import NodeGroup, CollectorCallback +@enrichment(EnrichmentStage.PROCEDURE, procedures=['mount_fs']) +def enrich_procedure_inventory(cluster: KubernetesCluster) -> None: + proc_items: List[dict] = cluster.procedure_inventory # type: ignore[assignment] + if not proc_items: + return + + existing: List[dict] = cluster.inventory.setdefault('services', {}).setdefault('fsmount', []) + existing_by_name = {item['name']: i for i, item in enumerate(existing)} + + for item in proc_items: + idx = existing_by_name.get(item['name']) + if idx is not None: + existing[idx] = utils.deepcopy_yaml(item) + else: + existing.append(utils.deepcopy_yaml(item)) + existing_by_name[item['name']] = len(existing) - 1 + + @enrichment(EnrichmentStage.FULL) def enrich_inventory(cluster: KubernetesCluster) -> None: fsmount_list: List[dict] = cluster.inventory.get('services', {}).get('fsmount', []) @@ -49,8 +67,10 @@ def enrich_inventory(cluster: KubernetesCluster) -> None: f"provided for fsmount item {item['name']!r}.") -def get_applicable_items(cluster: KubernetesCluster, node: NodeGroup) -> List[dict]: - fsmount_list: List[dict] = cluster.inventory.get('services', {}).get('fsmount', []) +def get_applicable_items(cluster: KubernetesCluster, node: NodeGroup, + fsmount_list: List[dict] = None) -> List[dict]: + if fsmount_list is None: + fsmount_list = cluster.inventory.get('services', {}).get('fsmount', []) applicable = [] for item in fsmount_list: groups: Union[List[str], None] = item.get('groups') @@ -88,12 +108,12 @@ def _parse_mounts(mounts_output: str) -> dict: return result -def is_mounted(group: NodeGroup) -> bool: +def is_mounted(group: NodeGroup, fsmount_list: List[dict] = None) -> bool: cluster: KubernetesCluster = group.cluster results = group.sudo("cat /proc/mounts") for node in group.get_ordered_members_list(): - applicable = get_applicable_items(cluster, node) + applicable = get_applicable_items(cluster, node, fsmount_list) if not applicable: continue host = node.get_host() @@ -106,7 +126,7 @@ def is_mounted(group: NodeGroup) -> bool: return True -def check_mounts(group: NodeGroup) -> List[str]: +def check_mounts(group: NodeGroup, fsmount_list: List[dict] = None) -> List[str]: """Return a list of human-readable error strings for missing or wrong-type mounts.""" cluster: KubernetesCluster = group.cluster @@ -120,7 +140,7 @@ def check_mounts(group: NodeGroup) -> List[str]: errors = [] for node in group.get_ordered_members_list(): - applicable = get_applicable_items(cluster, node) + applicable = get_applicable_items(cluster, node, fsmount_list) if not applicable: continue host = node.get_host() @@ -143,17 +163,17 @@ def check_mounts(group: NodeGroup) -> List[str]: return errors -def setup_fsmount(group: NodeGroup) -> bool: +def setup_fsmount(group: NodeGroup, fsmount_list: List[dict] = None) -> bool: cluster: KubernetesCluster = group.cluster logger = cluster.log - if is_mounted(group): + if is_mounted(group, fsmount_list): logger.debug("Skipped - all required filesystems are already mounted") return False changed = False for node in group.get_ordered_members_list(): - applicable = get_applicable_items(cluster, node) + applicable = get_applicable_items(cluster, node, fsmount_list) if not applicable: continue diff --git a/kubemarine/procedures/__init__.py b/kubemarine/procedures/__init__.py index 60ee5b593..4fe390735 100644 --- a/kubemarine/procedures/__init__.py +++ b/kubemarine/procedures/__init__.py @@ -53,6 +53,8 @@ def _import_tasks_procedure(name: str) -> TasksProcedure: from kubemarine.procedures import install as procedure elif name == "manage_pss": from kubemarine.procedures import manage_pss as procedure + elif name == "mount_fs": + from kubemarine.procedures import mount_fs as procedure elif name == "reboot": from kubemarine.procedures import reboot as procedure elif name == "reconfigure": diff --git a/kubemarine/procedures/mount_fs.py b/kubemarine/procedures/mount_fs.py new file mode 100644 index 000000000..f145acab5 --- /dev/null +++ b/kubemarine/procedures/mount_fs.py @@ -0,0 +1,95 @@ +# Copyright 2021-2022 NetCracker Technology Corporation +# +# Licensed under the Apache License, Version 2.0 (the "License"); +# you may not use this file except in compliance with the License. +# You may obtain a copy of the License at +# +# http://www.apache.org/licenses/LICENSE-2.0 +# +# Unless required by applicable law or agreed to in writing, software +# distributed under the License is distributed on an "AS IS" BASIS, +# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. +# See the License for the specific language governing permissions and +# limitations under the License. + +from collections import OrderedDict +from typing import List, Any + +from kubemarine import fsmount, kubernetes, system +from kubemarine.core import flow +from kubemarine.core.cluster import KubernetesCluster +from kubemarine.procedures import install + + +def mount_filesystems(cluster: KubernetesCluster) -> None: + # procedure_inventory is the raw fsmount list from procedure.yaml + fsmount_list: List[dict] = cluster.procedure_inventory # type: ignore[assignment] + + first_control_plane = cluster.nodes['control-plane'].get_first_member() + timeout_config = cluster.inventory['globals']['expect']['pods']['kubernetes'] + + for node in cluster.nodes['all'].get_ordered_members_list(): + node_name = node.get_node_name() + node_config = node.get_config() + is_k8s_node = 'control-plane' in node_config['roles'] or 'worker' in node_config['roles'] + + applicable = fsmount.get_applicable_items(cluster, node, fsmount_list) + if not applicable: + continue + + if is_k8s_node: + cluster.log.debug(f"Draining node {node_name!r} before fsmount setup") + first_control_plane.sudo( + kubernetes.prepare_drain_command(cluster, node_name, disable_eviction=False), + warn=True, pty=True) + + for item in applicable: + mount_path = item['path'].rstrip('/') + cluster.log.debug(f"Removing files in {mount_path!r} on {node_name!r}") + node.sudo(f"rm -rf {mount_path}/*") + + node.call(fsmount.setup_fsmount, fsmount_list=fsmount_list) + + cluster.log.debug(f"Rebooting node {node_name!r} after fsmount setup") + system.perform_group_reboot(node) + + if is_k8s_node: + cluster.log.debug(f"Uncordoning node {node_name!r} after reboot") + first_control_plane.wait_command_successful( + f"kubectl uncordon {node_name}", + hide=False, pty=True, + timeout=timeout_config['timeout'], + retries=timeout_config['retries']) + + +tasks = OrderedDict({ + "mount_filesystems": mount_filesystems, + "overview": install.overview, +}) + + +class MountFsAction(flow.TasksAction): + def __init__(self) -> None: + super().__init__('mount_fs', tasks, recreate_inventory=True) + + +def create_context(cli_arguments: List[str] = None) -> dict: + cli_help = ''' + Script for mounting filesystems defined in procedure.yaml. + + How to use: + + ''' + + parser = flow.new_procedure_parser(cli_help, tasks=tasks) + context = flow.create_context(parser, cli_arguments, procedure='mount_fs') + return context + + +def main(cli_arguments: List[str] = None) -> None: + context = create_context(cli_arguments) + flow.ActionsFlow([MountFsAction()]).run_flow(context) + + +if __name__ == '__main__': + main() diff --git a/kubemarine/resources/schemas/mount_fs.json b/kubemarine/resources/schemas/mount_fs.json new file mode 100644 index 000000000..0930d5059 --- /dev/null +++ b/kubemarine/resources/schemas/mount_fs.json @@ -0,0 +1,4 @@ +{ + "$schema": "http://json-schema.org/draft-07/schema", + "$ref": "definitions/services/fsmount.json" +} From 8ae5bd2140a7c8406d7ad1f4a0dff2e51dd62a41 Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Wed, 19 Aug 2026 08:59:58 +0300 Subject: [PATCH 18/30] feature: new procedure --- kubemarine/procedures/mount_fs.py | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/kubemarine/procedures/mount_fs.py b/kubemarine/procedures/mount_fs.py index f145acab5..e816229ee 100644 --- a/kubemarine/procedures/mount_fs.py +++ b/kubemarine/procedures/mount_fs.py @@ -13,7 +13,7 @@ # limitations under the License. from collections import OrderedDict -from typing import List, Any +from typing import List from kubemarine import fsmount, kubernetes, system from kubemarine.core import flow From 9b658fa1bb674a639ae5a35bf83ae6b872f48083 Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Thu, 20 Aug 2026 09:05:41 +0300 Subject: [PATCH 19/30] feature: reboot --- docs/public/Maintenance.md | 44 +++++++++++++++------- kubemarine/fsmount.py | 2 +- kubemarine/procedures/mount_fs.py | 36 +++++++++++------- kubemarine/resources/schemas/mount_fs.json | 13 ++++++- 4 files changed, 65 insertions(+), 30 deletions(-) diff --git a/docs/public/Maintenance.md b/docs/public/Maintenance.md index 731f7f04a..6e405b552 100644 --- a/docs/public/Maintenance.md +++ b/docs/public/Maintenance.md @@ -1361,17 +1361,22 @@ nodes: ## Mount Filesystems Procedure The `mount_fs` procedure sets up filesystems on cluster nodes according to a required `procedure.yaml`. -The format of `procedure.yaml` is the same as the `services.fsmount` list in `cluster.yaml` — an array of fsmount items specifying which filesystems to mount and on which nodes. -For each node that has at least one applicable fsmount item, the procedure: +For each node that has at least one applicable fsmount item, the procedure behaves differently depending on the `reboot` option: +**With `reboot: true` (default)**: 1. Drains the node (if it is a `control-plane` or `worker` node) to safely evacuate workloads. 2. Removes existing data from each configured mount path to ensure a clean state. 3. Installs and enables the corresponding systemd mount units. 4. Reboots the node so that the new mounts are activated at the OS level. 5. Uncordons the node (if it is a `control-plane` or `worker` node) to make it schedulable again. +**With `reboot: false`**: +1. Removes existing data from each configured mount path. +2. Installs and enables the corresponding systemd mount units immediately without a reboot. + Nodes that have no applicable items are skipped entirely. +After a successful run, `cluster.yaml` is updated with the fsmount items from `procedure.yaml`. **Note**: Data inside the configured mount paths is erased before the mount is set up. Back up any important data before running this procedure. @@ -1382,23 +1387,34 @@ The procedure accepts required positional argument with the path to the procedur The JSON schema for procedure inventory is available by [URL](/kubemarine/resources/schemas/mount_fs.json?raw=1). For more information, see [Validation by JSON Schemas](Installation.md#inventory-validation). -The procedure inventory has the same structure as the `services.fsmount` section in `cluster.yaml`. +#### reboot Parameter + +Controls whether each node is drained and rebooted after the mount units are installed. + +* `true` (default) — drain, install, reboot, uncordon. Use this when the filesystem must be cleanly initialized before any workloads run on the node. +* `false` — install and enable the units in-place without a reboot. Use this when the filesystem can be activated live. + +#### fsmount Parameter + +The list of filesystem mount items to apply. The structure is identical to the `services.fsmount` section in `cluster.yaml`. For a description of the available fields, refer to the [fsmount](Installation.md#fsmount) section in _Kubemarine Installation Procedure_. Example: ```yaml -- name: zram-pods - device: /dev/zram0 - path: /var/log/pods - type: zram - size: 1G - template: - source: templates/zram.mount.j2 - destination: /etc/systemd/system/var-log-pods.mount - groups: - - control-plane - - worker +reboot: true +fsmount: + - name: zram-pods + device: /dev/zram0 + path: /var/log/pods + type: zram + size: 1G + template: + source: templates/zram.mount.j2 + destination: /etc/systemd/system/var-log-pods.mount + groups: + - control-plane + - worker ``` ### Mount Filesystems Procedure Tasks Tree diff --git a/kubemarine/fsmount.py b/kubemarine/fsmount.py index 5cda987d3..735ef5c77 100644 --- a/kubemarine/fsmount.py +++ b/kubemarine/fsmount.py @@ -25,7 +25,7 @@ @enrichment(EnrichmentStage.PROCEDURE, procedures=['mount_fs']) def enrich_procedure_inventory(cluster: KubernetesCluster) -> None: - proc_items: List[dict] = cluster.procedure_inventory # type: ignore[assignment] + proc_items: List[dict] = cluster.procedure_inventory.get('fsmount', []) if not proc_items: return diff --git a/kubemarine/procedures/mount_fs.py b/kubemarine/procedures/mount_fs.py index e816229ee..97aa12151 100644 --- a/kubemarine/procedures/mount_fs.py +++ b/kubemarine/procedures/mount_fs.py @@ -22,13 +22,20 @@ def mount_filesystems(cluster: KubernetesCluster) -> None: - # procedure_inventory is the raw fsmount list from procedure.yaml - fsmount_list: List[dict] = cluster.procedure_inventory # type: ignore[assignment] + proc_inv = cluster.procedure_inventory + fsmount_list: List[dict] = proc_inv.get('fsmount', []) + do_reboot: bool = proc_inv.get('reboot', True) first_control_plane = cluster.nodes['control-plane'].get_first_member() timeout_config = cluster.inventory['globals']['expect']['pods']['kubernetes'] - for node in cluster.nodes['all'].get_ordered_members_list(): + target_nodes = cluster.make_group([]) + for item in fsmount_list: + target_nodes = target_nodes.include_group( + cluster.create_group_from_groups_nodes_names( + item.get('groups') or [], item.get('nodes') or [])) + + for node in target_nodes.get_ordered_members_list(): node_name = node.get_node_name() node_config = node.get_config() is_k8s_node = 'control-plane' in node_config['roles'] or 'worker' in node_config['roles'] @@ -37,7 +44,7 @@ def mount_filesystems(cluster: KubernetesCluster) -> None: if not applicable: continue - if is_k8s_node: + if do_reboot and is_k8s_node: cluster.log.debug(f"Draining node {node_name!r} before fsmount setup") first_control_plane.sudo( kubernetes.prepare_drain_command(cluster, node_name, disable_eviction=False), @@ -50,16 +57,17 @@ def mount_filesystems(cluster: KubernetesCluster) -> None: node.call(fsmount.setup_fsmount, fsmount_list=fsmount_list) - cluster.log.debug(f"Rebooting node {node_name!r} after fsmount setup") - system.perform_group_reboot(node) - - if is_k8s_node: - cluster.log.debug(f"Uncordoning node {node_name!r} after reboot") - first_control_plane.wait_command_successful( - f"kubectl uncordon {node_name}", - hide=False, pty=True, - timeout=timeout_config['timeout'], - retries=timeout_config['retries']) + if do_reboot: + cluster.log.debug(f"Rebooting node {node_name!r} after fsmount setup") + system.perform_group_reboot(node) + + if is_k8s_node: + cluster.log.debug(f"Uncordoning node {node_name!r} after reboot") + first_control_plane.wait_command_successful( + f"kubectl uncordon {node_name}", + hide=False, pty=True, + timeout=timeout_config['timeout'], + retries=timeout_config['retries']) tasks = OrderedDict({ diff --git a/kubemarine/resources/schemas/mount_fs.json b/kubemarine/resources/schemas/mount_fs.json index 0930d5059..3436bb746 100644 --- a/kubemarine/resources/schemas/mount_fs.json +++ b/kubemarine/resources/schemas/mount_fs.json @@ -1,4 +1,15 @@ { "$schema": "http://json-schema.org/draft-07/schema", - "$ref": "definitions/services/fsmount.json" + "type": "object", + "properties": { + "reboot": { + "type": "boolean", + "description": "Whether to drain and reboot the node after mounting. Defaults to true." + }, + "fsmount": { + "$ref": "definitions/services/fsmount.json" + } + }, + "required": ["fsmount"], + "additionalProperties": false } From 7353222ed693c6b125a980f73aac66c42a71d95b Mon Sep 17 00:00:00 2001 From: Dipali Said Date: Thu, 20 Aug 2026 16:58:18 +0530 Subject: [PATCH 20/30] Docs: [CPCAP-14161] Review --- docs/public/Installation.md | 79 +++++++++++++++++++------------------ docs/public/Kubecheck.md | 12 +++--- docs/public/Maintenance.md | 59 ++++++++++++++------------- 3 files changed, 79 insertions(+), 71 deletions(-) diff --git a/docs/public/Installation.md b/docs/public/Installation.md index 561009e55..14e5a0892 100644 --- a/docs/public/Installation.md +++ b/docs/public/Installation.md @@ -2592,44 +2592,47 @@ The following settings are supported in the extended format: *Can restart a service*: **No** -*Overwrites files*: Yes – the rendered systemd unit file is uploaded to the location defined in `template.destination`. A backup of the previous file is kept. - -*OS‑specific*: No - -The `services.fsmount` section allows you to define additional filesystems that should be formatted and mounted on cluster nodes using **systemd** unit files. The default configuration is empty. - -The following parameters are supported for each item: - -|Parameter|Mandatory|Description| -|---|---|---| -|**name**|**yes**|Identifier for this mount entry, used in logs and dump filenames.| -|**device**|**yes**|The device file to mount (e.g. `/dev/sdd1`, `/dev/zram0`).| -|**path**|**yes**|The mount point path on the node.| -|**template.source**|**yes**|Path to the Jinja2 template for the systemd unit. Can be an internal resource path (relative to the Kubemarine package) or an external absolute path.| -|**template.destination**|**yes**|Absolute path on the node where the rendered unit file is placed.| -|**size**|no|Size of the filesystem, passed to the systemd unit template (e.g. `1G`). Required for virtual devices such as zram.| -|**type**|no|Filesystem type passed to the template (e.g. `ext4`, `tmpfs`).| -|**preparation_script**|no|Path to a shell script executed before the systemd unit is installed. If the script exits with a non-zero code, the mount entry is skipped for that node. Can be an internal resource path or an external absolute path.| -|**groups**|no|The list of node roles where this mount should be applied.| -|**nodes**|no|The list of specific node names where this mount should be applied.| - -**Notes**: -* You can specify `groups` and `nodes` at the same time; both are merged to determine the target nodes. -* If neither `groups` nor `nodes` is specified, the mount is applied to all nodes. -* If the mount path is already present in `/proc/mounts`, the entry is skipped for that node. -* The preparation script is uploaded to the node, executed, and then removed. It is intended for operations such as loading kernel modules or installing required packages. - -**Warning**: The `preparation_script` failure is non-fatal. If the script fails, only that specific mount entry is skipped on the affected node; other entries continue to be processed. - -The template receives the following variables for rendering: - -|Variable|Source| -|---|---| -|`name`|`name` field| -|`device`|`device` field| -|`path`|`path` field| -|`size`|`size` field (empty string if not set)| -|`type`|`type` field (empty string if not set)| +*Overwrites files*: **Yes** – the rendered systemd unit file is uploaded to the location defined by `template.destination`. A backup of any existing file is retained. + +*OS‑specific*: **No** + +The `services.fsmount` section allows you to declare additional filesystems that should be formatted and mounted on cluster nodes using **systemd** unit files. By default the section is empty. + +Each entry may contain the following keys: + +| Parameter | Mandatory | Description | +|------------------------|-----------|-------------| +| **name** | **yes** | Identifier for the mount entry; used in logs and dump filenames. | +| **enabled** | no | Whether the entry is processed. Defaults to **true**. | +| **device** | **yes** | Device file to mount (e.g. `/dev/sdd1`, `/dev/zram0`). | +| **path** | **yes** | Target mount point on the node. | +| **template.source** | **yes** | Path to the Jinja2 template that generates the systemd unit. Can be an internal resource (relative to the Kubemarine package) or an absolute external path. | +| **template.destination**| **yes** | Absolute path on the node where the rendered unit file will be placed. | +| **size** | no | Desired filesystem size (e.g. `1G`). Required for virtual devices such as zram. | +| **type** | no | Filesystem type (e.g. `ext4`, `tmpfs`). | +| **preparation_script** | no | Path to a shell script that runs before the systemd unit is installed. If the script exits with a non‑zero status, the mount entry is skipped on that node. The script may be an internal resource or an absolute external path. | +| **groups** | no | List of node roles (e.g. `control-plane`, `worker`) to which the mount should be applied. | +| **nodes** | no | List of specific node names to which the mount should be applied. | + +**Notes** + +* You may specify both `groups` and `nodes`; the resulting node set is the union of both selectors. +* If neither `groups` nor `nodes` is provided, the mount is applied to **all** nodes. +* If the mount point already appears in `/proc/mounts`, the entry is skipped for that node. + * The preparation script is uploaded, executed, and then removed. It is useful for loading kernel modules or installing prerequisite packages. + * If `enabled` is set to **false**, the mount entry is ignored entirely. + +**Warning**: Failure of the `preparation_script` is non‑fatal. Only the mount entry for the affected node is omitted; other entries continue to be processed. + +The following variables are made available to the Jinja2 template: + +| Variable | Source | +|----------|--------| +| `name` | `name` field | +| `device` | `device` field | +| `path` | `path` field | +| `size` | `size` field (empty string if omitted) | +| `type` | `type` field (empty string if omitted) | There is an example how to mount ZRAM disk on all cluster nodes. The cluster.yaml part: diff --git a/docs/public/Kubecheck.md b/docs/public/Kubecheck.md index 6fa046d04..e2fbb0147 100644 --- a/docs/public/Kubecheck.md +++ b/docs/public/Kubecheck.md @@ -662,15 +662,17 @@ rules does not match, the test will fail. *Task*: `services.system.fsmount.mounts` -This check validates that every filesystem mount defined in `services.fsmount` is correctly configured on `control‑plane` and `worker` nodes. For each entry the following conditions are verified: +This check validates that every **enabled** filesystem mount defined in ``services.fsmount`` is correctly configured on ``control‑plane`` and ``worker`` nodes. Entries with ``enabled: false`` are ignored. -- The mount point specified by `path` appears in `/proc/mounts`. -- The filesystem type reported in `/proc/mounts` matches the `type` declared in the inventory. -- For ZRAM‑backed devices (i.e., the `device` value begins with `/dev/zram`), the mount point is also listed in the output of `zramctl --output‑all`. +For each applicable entry the following conditions are verified: + +* The mount point specified by ``path`` appears in ``/proc/mounts``. +* If the ``type`` field is defined in the inventory, the filesystem type reported in ``/proc/mounts`` must match it. When ``type`` is omitted only the presence of the mount point is checked. +* For ZRAM‑backed devices (i.e., the ``device`` value starts with ``/dev/zram``), the mount point must also be listed in the output of ``zramctl --output‑all``. If any validation fails, the test reports the affected node and mount path with a detailed error message. -**Note**: Nodes that have no applicable `services.fsmount` entries are skipped silently. +**Note**: Nodes that have no applicable ``services.fsmount`` entries are skipped silently. ##### 218 Time Difference diff --git a/docs/public/Maintenance.md b/docs/public/Maintenance.md index 6e405b552..45fd68633 100644 --- a/docs/public/Maintenance.md +++ b/docs/public/Maintenance.md @@ -1357,49 +1357,52 @@ nodes: - name: control-plane-3 ``` - ## Mount Filesystems Procedure -The `mount_fs` procedure sets up filesystems on cluster nodes according to a required `procedure.yaml`. +The `mount_fs` procedure configures filesystems on cluster nodes based on the supplied `procedure.yaml` inventory. + +For each node that contains at least one applicable **fsmount** entry, the behaviour depends on the ``reboot`` option: -For each node that has at least one applicable fsmount item, the procedure behaves differently depending on the `reboot` option: +**When ``reboot: true`` (the default):** +1. **Drain** the node (if it is a ``control-plane`` or ``worker``) to safely evacuate workloads. +2. **Remove** any existing data from each configured mount path, ensuring a clean state. +3. **Install and enable** the corresponding systemd mount units. +4. **Reboot** the node so the new mounts become active at the operating‑system level. +5. **Uncordon** the node (again, for ``control-plane`` or ``worker``) to make it schedulable. -**With `reboot: true` (default)**: -1. Drains the node (if it is a `control-plane` or `worker` node) to safely evacuate workloads. -2. Removes existing data from each configured mount path to ensure a clean state. -3. Installs and enables the corresponding systemd mount units. -4. Reboots the node so that the new mounts are activated at the OS level. -5. Uncordons the node (if it is a `control-plane` or `worker` node) to make it schedulable again. +**When ``reboot: false``:** +1. **Remove** existing data from each configured mount path. +2. **Install and enable** the systemd mount units immediately, without a reboot. -**With `reboot: false`**: -1. Removes existing data from each configured mount path. -2. Installs and enables the corresponding systemd mount units immediately without a reboot. +Nodes that have no applicable items are skipped entirely. After a successful execution, ``cluster.yaml`` is updated with the fsmount items defined in the procedure file. -Nodes that have no applicable items are skipped entirely. -After a successful run, `cluster.yaml` is updated with the fsmount items from `procedure.yaml`. +**Note:** Data inside the configured mount paths is erased before the mounts are created. Back up any important data prior to running this procedure. -**Note**: Data inside the configured mount paths is erased before the mount is set up. Back up any important data before running this procedure. +**Note:** Data inside the configured mount paths is erased before the mounts are created. Back up any important data +prior to running this procedure. ### Mount Filesystems Procedure Parameters -The procedure accepts required positional argument with the path to the procedure inventory file. +The procedure requires a positional argument that points to the procedure inventory file. -The JSON schema for procedure inventory is available by [URL](/kubemarine/resources/schemas/mount_fs.json?raw=1). -For more information, see [Validation by JSON Schemas](Installation.md#inventory-validation). +The JSON schema for the inventory is available at +[URL](/kubemarine/resources/schemas/mount_fs.json?raw=1). See +[Validation by JSON Schemas](Installation.md#inventory-validation) for further details. -#### reboot Parameter +#### ``reboot`` Parameter Controls whether each node is drained and rebooted after the mount units are installed. -* `true` (default) — drain, install, reboot, uncordon. Use this when the filesystem must be cleanly initialized before any workloads run on the node. -* `false` — install and enable the units in-place without a reboot. Use this when the filesystem can be activated live. +* ``true`` (default) – drain, install, reboot, then uncordon. Use this when the filesystem must be cleanly initialised before any + workloads run on the node. +* ``false`` – install and enable the units in‑place without a reboot. Use this when the filesystem can be activated live. -#### fsmount Parameter +#### ``fsmount`` Parameter -The list of filesystem mount items to apply. The structure is identical to the `services.fsmount` section in `cluster.yaml`. -For a description of the available fields, refer to the [fsmount](Installation.md#fsmount) section in _Kubemarine Installation Procedure_. +The list of filesystem mount items to apply. Its structure is identical to the ``services.fsmount`` section in ``cluster.yaml``. +For a description of the available fields, refer to the [fsmount](Installation.md#fsmount) section in the _Kubemarine Installation Procedure_. -Example: +**Example:** ```yaml reboot: true @@ -1419,10 +1422,10 @@ fsmount: ### Mount Filesystems Procedure Tasks Tree -The `mount_fs` procedure executes the following sequence of tasks: +The ``mount_fs`` procedure executes the following sequence of tasks: -* mount_filesystems -* overview +* ``mount_filesystems`` +* ``overview`` ## Certificate Renew Procedure From 6e27cdc37e17978a350c19db98fc787c7b64e279 Mon Sep 17 00:00:00 2001 From: Dipali Said Date: Thu, 20 Aug 2026 16:59:22 +0530 Subject: [PATCH 21/30] Docs: [CPCAP-14161] Review --- docs/public/Maintenance.md | 27 ++++++++++++++------------- 1 file changed, 14 insertions(+), 13 deletions(-) diff --git a/docs/public/Maintenance.md b/docs/public/Maintenance.md index 45fd68633..8fdaa3868 100644 --- a/docs/public/Maintenance.md +++ b/docs/public/Maintenance.md @@ -1357,28 +1357,29 @@ nodes: - name: control-plane-3 ``` + ## Mount Filesystems Procedure -The `mount_fs` procedure configures filesystems on cluster nodes based on the supplied `procedure.yaml` inventory. +The `mount_fs` procedure configures filesystems on cluster nodes based on the supplied `procedure.yaml`. -For each node that contains at least one applicable **fsmount** entry, the behaviour depends on the ``reboot`` option: +For each node that contains at least one applicable **fsmount** entry, the behaviour depends on the value of the +``reboot`` option: **When ``reboot: true`` (the default):** -1. **Drain** the node (if it is a ``control-plane`` or ``worker``) to safely evacuate workloads. -2. **Remove** any existing data from each configured mount path, ensuring a clean state. -3. **Install and enable** the corresponding systemd mount units. -4. **Reboot** the node so the new mounts become active at the operating‑system level. -5. **Uncordon** the node (again, for ``control-plane`` or ``worker``) to make it schedulable. +1. Drains the node (if it is a ``control-plane`` or ``worker``) to safely evacuate workloads. +2. Removes any existing data from each configured mount path, ensuring a clean state. +3. Installs and enables the corresponding systemd mount units. +4. Reboots the node so the new mounts become active at the operating‑system level. +5. Uncordons the node (again, for ``control-plane`` or ``worker``) to make it schedulable. **When ``reboot: false``:** -1. **Remove** existing data from each configured mount path. -2. **Install and enable** the systemd mount units immediately, without a reboot. - -Nodes that have no applicable items are skipped entirely. After a successful execution, ``cluster.yaml`` is updated with the fsmount items defined in the procedure file. +1. Removes existing data from each configured mount path. +2. Installs and enables the systemd mount units immediately, without performing a reboot. -**Note:** Data inside the configured mount paths is erased before the mounts are created. Back up any important data prior to running this procedure. +Nodes that have no applicable items are skipped entirely. After a successful execution, ``cluster.yaml`` is updated +with the fsmount items defined in the procedure file. -**Note:** Data inside the configured mount paths is erased before the mounts are created. Back up any important data +**Note:** Data inside the configured mount paths is erased before the mount is set up. Back up any important data prior to running this procedure. ### Mount Filesystems Procedure Parameters From c1911c3e231cc31e87ddf83b443ff8d9326b2de7 Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Thu, 20 Aug 2026 16:52:17 +0300 Subject: [PATCH 22/30] feature: docs; schema --- docs/public/Installation.md | 11 ++++++++++- docs/public/Maintenance.md | 9 +++++++++ .../schemas/definitions/services/kubeadm_kubelet.json | 6 +++++- kubemarine/resources/schemas/reconfigure.json | 4 +++- 4 files changed, 27 insertions(+), 3 deletions(-) diff --git a/docs/public/Installation.md b/docs/public/Installation.md index 14e5a0892..1cc398eff 100644 --- a/docs/public/Installation.md +++ b/docs/public/Installation.md @@ -2634,7 +2634,7 @@ The following variables are made available to the Jinja2 template: | `size` | `size` field (empty string if omitted) | | `type` | `type` field (empty string if omitted) | -There is an example how to mount ZRAM disk on all cluster nodes. The cluster.yaml part: +This functionality could be used to mount ZRAM volume on all cluster nodes for pods logs. The cluster.yaml part is the following: ```yaml services: @@ -2652,6 +2652,15 @@ services: groups: [control-plane, worker] ``` +The `size` of the ZRAM volume must be chosen according to the kubelet configuration (`containerLogMax` options). For the current case the recommended numbers are the following. They must be set in `kubeadm_kubelet` section: + +```yaml +services: + kubeadm_kubelet: + containerLogMaxSize: 5Mi + containerLogMaxFiles: 2 +``` + #### audit ##### Audit Kubernetes Policy diff --git a/docs/public/Maintenance.md b/docs/public/Maintenance.md index 8fdaa3868..fad39ac7d 100644 --- a/docs/public/Maintenance.md +++ b/docs/public/Maintenance.md @@ -1381,6 +1381,15 @@ with the fsmount items defined in the procedure file. **Note:** Data inside the configured mount paths is erased before the mount is set up. Back up any important data prior to running this procedure. +**Note:** For the case when ZRAM volume is monted to `/var/log/pods` the kubelet service must be reconfigured +with the recomended parameters. The procedure.yaml is the following: + +```yaml +services: + kubeadm_kubelet: + containerLogMaxSize: 5Mi + containerLogMaxFiles: 2 +``` ### Mount Filesystems Procedure Parameters diff --git a/kubemarine/resources/schemas/definitions/services/kubeadm_kubelet.json b/kubemarine/resources/schemas/definitions/services/kubeadm_kubelet.json index 00dc110d9..f9af7e494 100644 --- a/kubemarine/resources/schemas/definitions/services/kubeadm_kubelet.json +++ b/kubemarine/resources/schemas/definitions/services/kubeadm_kubelet.json @@ -10,6 +10,8 @@ "cgroupDriver": {"type": "string", "default": "systemd"}, "maxPods": {"$ref": "#/definitions/MaxPods"}, "serializeImagePulls": {"$ref": "#/definitions/SerializeImagePulls"}, + "containerLogMaxSize": {"$ref": "#/definitions/ContainerLogMaxSize"}, + "containerLogMaxFiles": {"$ref": "#/definitions/ContainerLogMaxFiles"}, "apiVersion": {"type": ["string"], "default": "kubelet.config.k8s.io/v1beta1"}, "kind": {"enum": ["KubeletConfiguration"], "default": "KubeletConfiguration"} }, @@ -17,6 +19,8 @@ "ProtectKernelDefaults": {"type": "boolean", "default": true}, "PodPidsLimit": {"type": "integer", "default": 4096}, "MaxPods": {"type": "integer", "default": 110}, - "SerializeImagePulls": {"type": "boolean", "default": false} + "SerializeImagePulls": {"type": "boolean", "default": false}, + "ContainerLogMaxSize": {"type": "string", "default": "10Mi"}, + "ContainerLogMaxFiles": {"type": "integer", "minimum": 2, "default": 5} } } diff --git a/kubemarine/resources/schemas/reconfigure.json b/kubemarine/resources/schemas/reconfigure.json index 2e0e9ac55..e753fbf14 100644 --- a/kubemarine/resources/schemas/reconfigure.json +++ b/kubemarine/resources/schemas/reconfigure.json @@ -103,7 +103,9 @@ "protectKernelDefaults": {"$ref": "definitions/services/kubeadm_kubelet.json#/definitions/ProtectKernelDefaults"}, "podPidsLimit": {"$ref": "definitions/services/kubeadm_kubelet.json#/definitions/PodPidsLimit"}, "maxPods": {"$ref": "definitions/services/kubeadm_kubelet.json#/definitions/MaxPods"}, - "serializeImagePulls": {"$ref": "definitions/services/kubeadm_kubelet.json#/definitions/SerializeImagePulls"} + "serializeImagePulls": {"$ref": "definitions/services/kubeadm_kubelet.json#/definitions/SerializeImagePulls"}, + "containerLogMaxSize": {"$ref": "definitions/services/kubeadm_kubelet.json#/definitions/ContainerLogMaxSize"}, + "containerLogMaxFiles": {"$ref": "definitions/services/kubeadm_kubelet.json#/definitions/ContainerLogMaxFiles"} }, "additionalProperties": false }, From 9e8110192e2b00f2347d45d0996a51c4798ebcfa Mon Sep 17 00:00:00 2001 From: Dipali Said Date: Fri, 21 Aug 2026 11:25:18 +0530 Subject: [PATCH 23/30] Docs: CPCAP-14161 Doc review --- docs/public/Installation.md | 2 +- docs/public/Maintenance.md | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/docs/public/Installation.md b/docs/public/Installation.md index 1cc398eff..68a4670da 100644 --- a/docs/public/Installation.md +++ b/docs/public/Installation.md @@ -2652,7 +2652,7 @@ services: groups: [control-plane, worker] ``` -The `size` of the ZRAM volume must be chosen according to the kubelet configuration (`containerLogMax` options). For the current case the recommended numbers are the following. They must be set in `kubeadm_kubelet` section: +The `size` of the ZRAM volume must be selected according to the kubelet configuration (`containerLogMax` options). For the current case, the following values are recommended and they must be set in the `kubeadm_kubelet` section: ```yaml services: diff --git a/docs/public/Maintenance.md b/docs/public/Maintenance.md index fad39ac7d..3c4473ac5 100644 --- a/docs/public/Maintenance.md +++ b/docs/public/Maintenance.md @@ -1379,10 +1379,10 @@ For each node that contains at least one applicable **fsmount** entry, the behav Nodes that have no applicable items are skipped entirely. After a successful execution, ``cluster.yaml`` is updated with the fsmount items defined in the procedure file. -**Note:** Data inside the configured mount paths is erased before the mount is set up. Back up any important data +**Note**: Data inside the configured mount paths is erased before the mount is set up. Back up any important data prior to running this procedure. -**Note:** For the case when ZRAM volume is monted to `/var/log/pods` the kubelet service must be reconfigured -with the recomended parameters. The procedure.yaml is the following: +**Note**: For the case when ZRAM volume is mounted to `/var/log/pods`, the kubelet service must be reconfigured +with the recomended parameters. The `procedure.yaml` is as follows: ```yaml services: From 6b41df662a81a0e605631d79c057f5416c5ee66d Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Mon, 24 Aug 2026 14:05:31 +0300 Subject: [PATCH 24/30] feature: docs --- docs/public/Installation.md | 2 ++ docs/public/Maintenance.md | 2 ++ 2 files changed, 4 insertions(+) diff --git a/docs/public/Installation.md b/docs/public/Installation.md index 68a4670da..288de703b 100644 --- a/docs/public/Installation.md +++ b/docs/public/Installation.md @@ -2661,6 +2661,8 @@ services: containerLogMaxFiles: 2 ``` +**Warning**: Pay attention, the OS must provide `zram` module. + #### audit ##### Audit Kubernetes Policy diff --git a/docs/public/Maintenance.md b/docs/public/Maintenance.md index 3c4473ac5..81be78331 100644 --- a/docs/public/Maintenance.md +++ b/docs/public/Maintenance.md @@ -1391,6 +1391,8 @@ services: containerLogMaxFiles: 2 ``` +**Warning**: Pay attention, the OS must provide `zram` module. + ### Mount Filesystems Procedure Parameters The procedure requires a positional argument that points to the procedure inventory file. From 855b46fc5cafa1f0e49ef178984d84f112c55793 Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Wed, 26 Aug 2026 13:06:28 +0300 Subject: [PATCH 25/30] feat: mg --- docs/public/Maintenance.md | 16 ++++++++-------- 1 file changed, 8 insertions(+), 8 deletions(-) diff --git a/docs/public/Maintenance.md b/docs/public/Maintenance.md index 81be78331..63a908ba6 100644 --- a/docs/public/Maintenance.md +++ b/docs/public/Maintenance.md @@ -1419,17 +1419,17 @@ For a description of the available fields, refer to the [fsmount](Installation.m ```yaml reboot: true fsmount: - - name: zram-pods + - name: zram + enabled: false + size: 1G + type: ext4 device: /dev/zram0 path: /var/log/pods - type: zram - size: 1G template: - source: templates/zram.mount.j2 - destination: /etc/systemd/system/var-log-pods.mount - groups: - - control-plane - - worker + source: templates/zram-setup.service.j2 + destination: /etc/systemd/system/zram-setup.service + preparation_script: resources/scripts/zram.sh + groups: [control-plane, worker] ``` ### Mount Filesystems Procedure Tasks Tree From ddaea5f579e103723cf22981cb54aafcdea60162 Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Wed, 26 Aug 2026 16:22:48 +0300 Subject: [PATCH 26/30] feat: ig/mg --- docs/public/Installation.md | 1 - docs/public/Maintenance.md | 1 - 2 files changed, 2 deletions(-) diff --git a/docs/public/Installation.md b/docs/public/Installation.md index 288de703b..515f152bc 100644 --- a/docs/public/Installation.md +++ b/docs/public/Installation.md @@ -2640,7 +2640,6 @@ This functionality could be used to mount ZRAM volume on all cluster nodes for p services: fsmount: - name: zram - enabled: false size: 1G type: ext4 device: /dev/zram0 diff --git a/docs/public/Maintenance.md b/docs/public/Maintenance.md index 63a908ba6..437ee1153 100644 --- a/docs/public/Maintenance.md +++ b/docs/public/Maintenance.md @@ -1420,7 +1420,6 @@ For a description of the available fields, refer to the [fsmount](Installation.m reboot: true fsmount: - name: zram - enabled: false size: 1G type: ext4 device: /dev/zram0 From 2ab1f49cb7ba5288415c8b8b3fcbead3944c8d56 Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Fri, 28 Aug 2026 19:00:38 +0300 Subject: [PATCH 27/30] feat: preparation_script --- docs/public/Installation.md | 15 +++++--- docs/public/Maintenance.md | 6 ++- kubemarine/fsmount.py | 37 ++++++++++--------- .../schemas/definitions/services/fsmount.json | 7 ++-- 4 files changed, 39 insertions(+), 26 deletions(-) diff --git a/docs/public/Installation.md b/docs/public/Installation.md index 515f152bc..72695f8b3 100644 --- a/docs/public/Installation.md +++ b/docs/public/Installation.md @@ -2588,7 +2588,7 @@ The following settings are supported in the extended format: *Installation task*: `prepare.system.fsmount` -*Can cause a reboot*: **No** +*Can cause a reboot*: **Yes** – when `preparation_scripts` contains more than one entry, the node is rebooted between consecutive scripts. *Can restart a service*: **No** @@ -2610,7 +2610,7 @@ Each entry may contain the following keys: | **template.destination**| **yes** | Absolute path on the node where the rendered unit file will be placed. | | **size** | no | Desired filesystem size (e.g. `1G`). Required for virtual devices such as zram. | | **type** | no | Filesystem type (e.g. `ext4`, `tmpfs`). | -| **preparation_script** | no | Path to a shell script that runs before the systemd unit is installed. If the script exits with a non‑zero status, the mount entry is skipped on that node. The script may be an internal resource or an absolute external path. | +| **preparation_scripts** | no | Ordered list of shell scripts that run before the systemd unit is installed. Each script may be an internal resource (relative to the Kubemarine package) or an absolute external path. A node reboot is performed between consecutive scripts. If any script exits with a non‑zero status, the procedure fails. For items with `type: zram`, defaults to `[resources/scripts/upgrade_kernel.sh, resources/scripts/zram.sh]`. | | **groups** | no | List of node roles (e.g. `control-plane`, `worker`) to which the mount should be applied. | | **nodes** | no | List of specific node names to which the mount should be applied. | @@ -2619,10 +2619,11 @@ Each entry may contain the following keys: * You may specify both `groups` and `nodes`; the resulting node set is the union of both selectors. * If neither `groups` nor `nodes` is provided, the mount is applied to **all** nodes. * If the mount point already appears in `/proc/mounts`, the entry is skipped for that node. - * The preparation script is uploaded, executed, and then removed. It is useful for loading kernel modules or installing prerequisite packages. + * Each preparation script is uploaded to the node, executed, and then removed. Scripts are useful for loading kernel modules, upgrading the kernel, or installing prerequisite packages. + * When more than one preparation script is specified, the node is rebooted between consecutive scripts so that kernel or module changes take effect before the next script runs. * If `enabled` is set to **false**, the mount entry is ignored entirely. -**Warning**: Failure of the `preparation_script` is non‑fatal. Only the mount entry for the affected node is omitted; other entries continue to be processed. +**Warning**: Failure of any `preparation_scripts` entry is fatal — the whole procedure stops immediately. The following variables are made available to the Jinja2 template: @@ -2647,10 +2648,14 @@ services: template: source: templates/zram-setup.service.j2 destination: /etc/systemd/system/zram-setup.service - preparation_script: resources/scripts/zram.sh + preparation_scripts: + - resources/scripts/upgrade_kernel.sh + - resources/scripts/zram.sh groups: [control-plane, worker] ``` +The node is rebooted between the two scripts so the upgraded kernel is running before `zram.sh` executes. + The `size` of the ZRAM volume must be selected according to the kubelet configuration (`containerLogMax` options). For the current case, the following values are recommended and they must be set in the `kubeadm_kubelet` section: ```yaml diff --git a/docs/public/Maintenance.md b/docs/public/Maintenance.md index 437ee1153..18b334c27 100644 --- a/docs/public/Maintenance.md +++ b/docs/public/Maintenance.md @@ -1427,10 +1427,14 @@ fsmount: template: source: templates/zram-setup.service.j2 destination: /etc/systemd/system/zram-setup.service - preparation_script: resources/scripts/zram.sh + preparation_scripts: + - resources/scripts/upgrade_kernel.sh + - resources/scripts/zram.sh groups: [control-plane, worker] ``` +The node is rebooted between `upgrade_kernel.sh` and `zram.sh` so the new kernel is active before the zram module is loaded. + ### Mount Filesystems Procedure Tasks Tree The ``mount_fs`` procedure executes the following sequence of tasks: diff --git a/kubemarine/fsmount.py b/kubemarine/fsmount.py index 735ef5c77..8d2fe1dd1 100644 --- a/kubemarine/fsmount.py +++ b/kubemarine/fsmount.py @@ -50,12 +50,11 @@ def enrich_inventory(cluster: KubernetesCluster) -> None: if item.get('groups') is None and item.get('nodes') is None: continue - preparation_script = item.get('preparation_script') - if preparation_script is not None: - ext_path = utils.get_external_resource_path(preparation_script) - if not os.path.isfile(ext_path) and not os.path.isfile(utils.get_internal_resource_path(preparation_script)): + for j, script in enumerate(item.get('preparation_scripts') or []): + ext_path = utils.get_external_resource_path(script) + if not os.path.isfile(ext_path) and not os.path.isfile(utils.get_internal_resource_path(script)): raise Exception( - f"'preparation_script' file {preparation_script!r} not found " + f"'preparation_scripts[{j}]' file {script!r} not found " f"for fsmount item at {utils.pretty_path(path)}") if item.get('nodes') is not None: @@ -185,25 +184,29 @@ def setup_fsmount(group: NodeGroup, fsmount_list: List[dict] = None) -> bool: logger.debug(f"Skipping fsmount item {item['name']!r} on {node.get_node_name()}: already mounted") continue - preparation_script = item.get('preparation_script') - if preparation_script: - logger.debug(f"Running preparation script for fsmount item {item['name']!r} on {node.get_node_name()}") - ext_path = utils.get_external_resource_path(preparation_script) + preparation_scripts = item.get('preparation_scripts') or [] + for idx, script_path in enumerate(preparation_scripts): + logger.debug(f"Running preparation script [{idx}] {script_path!r} " + f"for fsmount item {item['name']!r} on {node.get_node_name()}") + ext_path = utils.get_external_resource_path(script_path) if os.path.isfile(ext_path): - script_content = utils.read_external(preparation_script) + script_content = utils.read_external(script_path) else: - script_content = utils.read_internal(preparation_script) - remote_path = f"/tmp/fsmount_{item['name']}_prep.sh" + script_content = utils.read_internal(script_path) + remote_path = f"/tmp/fsmount_{item['name']}_prep_{idx}.sh" node.put(io.StringIO(script_content), remote_path, sudo=True) node.sudo(f"chmod +x {remote_path}") prep_result = node.sudo(f"bash {remote_path}", warn=True) node.sudo(f"rm -f {remote_path}") if prep_result[host].return_code != 0: - logger.warning( - f"Preparation script for fsmount item {item['name']!r} " - f"failed on {node.get_node_name()}, skipping." - f"The output is: {prep_result[host]}") - continue + raise Exception( + f"Preparation script [{idx}] {script_path!r} for fsmount item {item['name']!r} " + f"failed on {node.get_node_name()}. Output: {prep_result[host]}") + + if idx < len(preparation_scripts) - 1: + logger.debug(f"Rebooting {node.get_node_name()!r} between preparation scripts") + from kubemarine import system as _system # lazy import to avoid circular dependency + _system.perform_group_reboot(node) unit_content = _render_unit(item) unit_destination = item['template']['destination'] diff --git a/kubemarine/resources/schemas/definitions/services/fsmount.json b/kubemarine/resources/schemas/definitions/services/fsmount.json index 440d8c1c9..6e1b297a3 100644 --- a/kubemarine/resources/schemas/definitions/services/fsmount.json +++ b/kubemarine/resources/schemas/definitions/services/fsmount.json @@ -48,9 +48,10 @@ "required": ["source", "destination"], "additionalProperties": false }, - "preparation_script": { - "type": "string", - "description": "Path to a shell script run before creating the filesystem. If it fails, this item is skipped." + "preparation_scripts": { + "type": "array", + "description": "Ordered list of shell scripts run before creating the filesystem. A reboot is performed between consecutive scripts. If any script fails, the procedure fails.", + "items": {"type": "string"} }, "groups": { "$ref": "../common/node_ref.json#/definitions/Roles", From c3b06972a6e7751cbb64fb53f0d1fef48898150f Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Fri, 28 Aug 2026 19:02:02 +0300 Subject: [PATCH 28/30] feat: preparation_script --- kubemarine/resources/scripts/upgrade_kernel.sh | 11 +++++++++++ 1 file changed, 11 insertions(+) create mode 100644 kubemarine/resources/scripts/upgrade_kernel.sh diff --git a/kubemarine/resources/scripts/upgrade_kernel.sh b/kubemarine/resources/scripts/upgrade_kernel.sh new file mode 100644 index 000000000..32518186c --- /dev/null +++ b/kubemarine/resources/scripts/upgrade_kernel.sh @@ -0,0 +1,11 @@ +#!/bin/bash +set -e + +if command -v apt-get &>/dev/null; then + apt-get update -q + apt-get install -yq --only-upgrade linux-image-generic linux-headers-generic +elif command -v dnf &>/dev/null; then + dnf upgrade -y kernel +elif command -v yum &>/dev/null; then + yum upgrade -y kernel +fi From aa1f47c3c5d740f0f1450d6f232e2fd182cce08a Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Fri, 28 Aug 2026 19:50:44 +0300 Subject: [PATCH 29/30] feat: docs --- docs/public/Installation.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/docs/public/Installation.md b/docs/public/Installation.md index 72695f8b3..348b082a5 100644 --- a/docs/public/Installation.md +++ b/docs/public/Installation.md @@ -2610,7 +2610,7 @@ Each entry may contain the following keys: | **template.destination**| **yes** | Absolute path on the node where the rendered unit file will be placed. | | **size** | no | Desired filesystem size (e.g. `1G`). Required for virtual devices such as zram. | | **type** | no | Filesystem type (e.g. `ext4`, `tmpfs`). | -| **preparation_scripts** | no | Ordered list of shell scripts that run before the systemd unit is installed. Each script may be an internal resource (relative to the Kubemarine package) or an absolute external path. A node reboot is performed between consecutive scripts. If any script exits with a non‑zero status, the procedure fails. For items with `type: zram`, defaults to `[resources/scripts/upgrade_kernel.sh, resources/scripts/zram.sh]`. | +| **preparation_scripts** | no | Ordered list of shell scripts that run before the systemd unit is installed. Each script may be an internal resource (relative to the Kubemarine package) or an absolute external path. A node reboot is performed between consecutive scripts. If any script exits with a non‑zero status, the procedure fails. | | **groups** | no | List of node roles (e.g. `control-plane`, `worker`) to which the mount should be applied. | | **nodes** | no | List of specific node names to which the mount should be applied. | From a5b7cf934ae0fa6f0c2873c8140a9eaa7138d1ec Mon Sep 17 00:00:00 2001 From: Aleksandr Arefev <39635005+alexarefev@users.noreply.github.com> Date: Fri, 28 Aug 2026 20:32:39 +0300 Subject: [PATCH 30/30] feat: lint --- kubemarine/fsmount.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/kubemarine/fsmount.py b/kubemarine/fsmount.py index 8d2fe1dd1..e37dfc251 100644 --- a/kubemarine/fsmount.py +++ b/kubemarine/fsmount.py @@ -205,8 +205,10 @@ def setup_fsmount(group: NodeGroup, fsmount_list: List[dict] = None) -> bool: if idx < len(preparation_scripts) - 1: logger.debug(f"Rebooting {node.get_node_name()!r} between preparation scripts") - from kubemarine import system as _system # lazy import to avoid circular dependency - _system.perform_group_reboot(node) + initial_boot_history = node.sudo('uptime -s') + node.sudo(cluster.globals['nodes']['boot']['reboot_command'], warn=True) + logger.debug("Waiting for boot up...") + node.wait_for_reboot(initial_boot_history) unit_content = _render_unit(item) unit_destination = item['template']['destination']