Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions nodescraper/plugins/inband/memory/analyzer_args.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,6 +23,8 @@
# SOFTWARE.
#
###############################################################################
from typing import Optional

from pydantic import Field

from nodescraper.models.analyzerargs import AnalyzerArgs
Expand All @@ -31,6 +33,12 @@


class MemoryAnalyzerArgs(AnalyzerArgs):
minimum_free_memory_percent: Optional[float] = Field(
default=None,
ge=0,
le=100,
description="Minimum available system memory percentage required.",
)
ratio: float = Field(
default=0.66,
description="Required free-memory ratio (0-1). Analysis fails if free/total < ratio.",
Expand Down
40 changes: 40 additions & 0 deletions nodescraper/plugins/inband/memory/memory_analyzer.py
Original file line number Diff line number Diff line change
Expand Up @@ -61,6 +61,46 @@ def _bytes_to_gb(n: float) -> float:
total_memory = convert_to_bytes(data.mem_total)
used_memory = total_memory - available_memory

if args.minimum_free_memory_percent is not None:
if total_memory <= 0:
self.result.status = ExecutionStatus.WARNING
self.result.message = "Total memory is unavailable"
self._log_event(
category=EventCategory.OS,
description="Cannot validate minimum free memory percentage",
priority=EventPriority.WARNING,
data={"total_memory": total_memory, "available_memory": available_memory},
console_log=True,
)
return self.result

available_percent = available_memory / total_memory * 100
if available_percent < args.minimum_free_memory_percent:
self.result.status = ExecutionStatus.ERROR
self.result.message = "Minimum free memory percentage not met"
self._log_event(
category=EventCategory.OS,
description=(
f"Available memory is {available_percent:.2f}% "
f"(minimum {args.minimum_free_memory_percent:.2f}%)"
),
priority=EventPriority.CRITICAL,
data={
"available_memory": available_memory,
"total_memory": total_memory,
"available_percent": available_percent,
"minimum_free_memory_percent": args.minimum_free_memory_percent,
},
console_log=True,
)
else:
self.result.status = ExecutionStatus.OK
self.result.message = (
f"Available memory is {available_percent:.2f}% "
f"(minimum {args.minimum_free_memory_percent:.2f}%)"
)
return self.result

threshold_bytes = convert_to_bytes(args.memory_threshold)

if total_memory > threshold_bytes:
Expand Down
38 changes: 38 additions & 0 deletions nodescraper/plugins/inband/nvme/analyzer_args.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,38 @@
###############################################################################
#
# MIT License
#
# Copyright (c) 2026 Advanced Micro Devices, Inc.
#
# Permission is hereby granted, free of charge, to any person obtaining a copy
# of this software and associated documentation files (the "Software"), to deal
# in the Software without restriction, including without limitation the rights
# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
# copies of the Software, and to permit persons to whom the Software is
# furnished to do so, subject to the following conditions:
#
# The above copyright notice and this permission notice shall be included in all
# copies or substantial portions of the Software.
#
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
# SOFTWARE.
#
###############################################################################
from typing import Optional

from pydantic import Field

from nodescraper.models import AnalyzerArgs


class NvmeAnalyzerArgs(AnalyzerArgs):
maximum_smart_error_count: Optional[int] = Field(
default=None,
ge=0,
description="Maximum allowed NVMe SMART media error count per device.",
)
101 changes: 101 additions & 0 deletions nodescraper/plugins/inband/nvme/nvme_analyzer.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,101 @@
###############################################################################
#
# MIT License
#
# Copyright (c) 2026 Advanced Micro Devices, Inc.
#
# Permission is hereby granted, free of charge, to any person obtaining a copy
# of this software and associated documentation files (the "Software"), to deal
# in the Software without restriction, including without limitation the rights
# to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
# copies of the Software, and to permit persons to whom the Software is
# furnished to do so, subject to the following conditions:
#
# The above copyright notice and this permission notice shall be included in all
# copies or substantial portions of the Software.
#
# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
# FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
# AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
# LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
# OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
# SOFTWARE.
#
###############################################################################
import re
from typing import Optional

from nodescraper.enums import EventCategory, EventPriority, ExecutionStatus
from nodescraper.interfaces import DataAnalyzer
from nodescraper.models import TaskResult

from .analyzer_args import NvmeAnalyzerArgs
from .nvmedata import NvmeDataModel


class NvmeAnalyzer(DataAnalyzer[NvmeDataModel, NvmeAnalyzerArgs]):
"""Check NVMe SMART health data."""

DATA_MODEL = NvmeDataModel

@staticmethod
def _parse_media_errors(smart_log: Optional[str]) -> Optional[int]:
if not smart_log:
return None
match = re.search(r"(?im)^\s*media_errors\s*:\s*(\d+)\s*$", smart_log)
return int(match.group(1)) if match else None

def analyze_data(
self, data: NvmeDataModel, args: Optional[NvmeAnalyzerArgs] = None
) -> TaskResult:
"""Check each NVMe device's SMART media error count."""
if args is None or args.maximum_smart_error_count is None:
self.result.status = ExecutionStatus.NOT_RAN
self.result.message = "Maximum NVMe SMART error count not provided"
return self.result

if not data.devices:
self.result.status = ExecutionStatus.NOT_RAN
self.result.message = "No NVMe data available"
return self.result

failures = []
for device, device_data in data.devices.items():
media_errors = self._parse_media_errors(device_data.smart_log)
if media_errors is None:
self._log_event(
category=EventCategory.STORAGE,
description=f"NVMe SMART media error count unavailable for {device}",
priority=EventPriority.WARNING,
data={"device": device},
console_log=True,
)
continue

if media_errors > args.maximum_smart_error_count:
failures.append((device, media_errors))

if failures:
for device, media_errors in failures:
self._log_event(
category=EventCategory.STORAGE,
description=(
f"NVMe SMART media errors exceeded threshold for {device}: "
f"{media_errors} > {args.maximum_smart_error_count}"
),
priority=EventPriority.ERROR,
data={
"device": device,
"media_errors": media_errors,
"maximum_smart_error_count": args.maximum_smart_error_count,
},
console_log=True,
)
self.result.status = ExecutionStatus.ERROR
self.result.message = "NVMe SMART media error threshold exceeded"
else:
self.result.status = ExecutionStatus.OK
self.result.message = "NVMe SMART media error counts are within limits"

return self.result
8 changes: 7 additions & 1 deletion nodescraper/plugins/inband/nvme/nvme_plugin.py
Original file line number Diff line number Diff line change
Expand Up @@ -25,13 +25,19 @@
###############################################################################
from nodescraper.base import InBandDataPlugin

from .analyzer_args import NvmeAnalyzerArgs
from .nvme_analyzer import NvmeAnalyzer
from .nvme_collector import NvmeCollector
from .nvmedata import NvmeDataModel


class NvmePlugin(InBandDataPlugin[NvmeDataModel, None, None]):
class NvmePlugin(InBandDataPlugin[NvmeDataModel, None, NvmeAnalyzerArgs]):
"""Plugin for collection and analysis of nvme data"""

DATA_MODEL = NvmeDataModel

COLLECTOR = NvmeCollector

ANALYZER = NvmeAnalyzer

ANALYZER_ARGS = NvmeAnalyzerArgs
7 changes: 6 additions & 1 deletion nodescraper/plugins/inband/os/analyzer_args.py
Original file line number Diff line number Diff line change
Expand Up @@ -23,7 +23,7 @@
# SOFTWARE.
#
###############################################################################
from typing import Union
from typing import Optional, Union

from pydantic import Field, field_validator

Expand All @@ -32,6 +32,11 @@


class OsAnalyzerArgs(AnalyzerArgs):
maximum_load_per_cpu_core: Optional[float] = Field(
default=None,
ge=0,
description="Maximum allowed 1-minute load average per CPU core.",
)
exp_os: Union[str, list] = Field(
default_factory=list,
description="Expected OS name/version string(s) to match (e.g. from lsb_release or /etc/os-release).",
Expand Down
89 changes: 72 additions & 17 deletions nodescraper/plugins/inband/os/os_analyzer.py
Original file line number Diff line number Diff line change
Expand Up @@ -48,26 +48,81 @@ def analyze_data(self, data: OsDataModel, args: Optional[OsAnalyzerArgs] = None)
Returns:
TaskResult: Result of the analysis containing status and message.
"""
if not args or not args.exp_os:
if args is None:
args = OsAnalyzerArgs()

if not args.exp_os and args.maximum_load_per_cpu_core is None:
self.result.message = "Expected OS name not provided"
self.result.status = ExecutionStatus.NOT_RAN
return self.result

for os_name in args.exp_os:
if (os_name == data.os_name and args.exact_match) or (
os_name in data.os_name and not args.exact_match
):
self.result.message = "OS name matches expected"
self.result.status = ExecutionStatus.OK
return self.result
os_name_matches = True
load_check_warning = False
load_check_error = False

if args.exp_os:
os_name_matches = False
for os_name in args.exp_os:
if (os_name == data.os_name and args.exact_match) or (
os_name in data.os_name and not args.exact_match
):
os_name_matches = True
break

if not os_name_matches:
self._log_event(
category=EventCategory.OS,
description=f"OS name mismatch! Expected: {args.exp_os}, actual: {data.os_name}",
data={"expected": args.exp_os, "actual": data.os_name},
priority=EventPriority.CRITICAL,
console_log=True,
)

if args.maximum_load_per_cpu_core is not None:
if data.load_average_1m is None or data.cpu_count is None or data.cpu_count <= 0:
load_check_warning = True
self._log_event(
category=EventCategory.OS,
description="Cannot validate maximum load per CPU core",
data={
"load_average_1m": data.load_average_1m,
"cpu_count": data.cpu_count,
},
priority=EventPriority.WARNING,
console_log=True,
)
else:
load_per_cpu_core = data.load_average_1m / data.cpu_count
if load_per_cpu_core > args.maximum_load_per_cpu_core:
load_check_error = True
self._log_event(
category=EventCategory.OS,
description=(
f"Load per CPU core is {load_per_cpu_core:.2f} "
f"(maximum {args.maximum_load_per_cpu_core:.2f})"
),
data={
"load_average_1m": data.load_average_1m,
"cpu_count": data.cpu_count,
"load_per_cpu_core": load_per_cpu_core,
"maximum_load_per_cpu_core": args.maximum_load_per_cpu_core,
},
priority=EventPriority.CRITICAL,
console_log=True,
)

self.result.message = "OS name mismatch!"
self.result.status = ExecutionStatus.ERROR
self._log_event(
category=EventCategory.OS,
description=f"OS name mismatch! Expected: {args.exp_os}, actual: {data.os_name}",
data={"expected": args.exp_os, "actual": data.os_name},
priority=EventPriority.CRITICAL,
console_log=True,
)
if not os_name_matches:
self.result.message = "OS name mismatch!"
self.result.status = ExecutionStatus.ERROR
elif load_check_error:
self.result.message = "Maximum load per CPU core exceeded"
self.result.status = ExecutionStatus.ERROR
elif load_check_warning:
self.result.message = "CPU load data is not available"
self.result.status = ExecutionStatus.WARNING
else:
self.result.message = (
"OS name matches expected" if args.exp_os else "Load per CPU core is within limit"
)
self.result.status = ExecutionStatus.OK
return self.result
29 changes: 29 additions & 0 deletions nodescraper/plugins/inband/os/os_collector.py
Original file line number Diff line number Diff line change
Expand Up @@ -45,6 +45,8 @@ class OsCollector(InBandDataCollector[OsDataModel, None]):
CMD_ESXI = "vmware -v"
PRETTY_STR = "PRETTY_NAME" # noqa: N806
CMD = f"sh -c '( lsb_release -ds || (cat /etc/*release | grep {PRETTY_STR}) || uname -om ) 2>/dev/null | head -n1'"
CMD_LOAD_AVERAGE = "cat /proc/loadavg"
CMD_CPU_COUNT = "nproc"

def collect_version(self) -> str:
"""Collect OS version.
Expand Down Expand Up @@ -135,9 +137,36 @@ def collect_data(self, args=None) -> tuple[TaskResult, Optional[OsDataModel]]:

if os_name:
os_version = self.collect_version()
load_average_1m = None
cpu_count = None
if self.system_info.os_family == OSFamily.LINUX:
load_res = self._run_sut_cmd(self.CMD_LOAD_AVERAGE)
if load_res.exit_code == 0 and load_res.stdout.strip():
try:
load_average_1m = float(load_res.stdout.split()[0])
except (ValueError, IndexError):
self._log_event(
category=EventCategory.OS,
description="Invalid 1-minute load average",
priority=EventPriority.WARNING,
)

cpu_res = self._run_sut_cmd(self.CMD_CPU_COUNT)
if cpu_res.exit_code == 0 and cpu_res.stdout.strip():
try:
cpu_count = int(cpu_res.stdout.strip())
except ValueError:
self._log_event(
category=EventCategory.OS,
description="Invalid CPU count",
priority=EventPriority.WARNING,
)

os_data = OsDataModel(
os_name=os_name,
os_version=os_version,
load_average_1m=load_average_1m,
cpu_count=cpu_count,
)
self._log_event(
category="OS_NAME_READ",
Expand Down
Loading
Loading