Merge branch 'analyze-memory-dedup-aliased-ram-symbols' into integration

This commit is contained in:
J. Nick Koston
2026-07-05 23:00:31 -05:00
1116 changed files with 21021 additions and 14895 deletions
+2 -9
View File
@@ -259,14 +259,7 @@ def lint_executable_bit(fname: Path) -> str | None:
return None
@lint_content_find_check(
"\t",
only_first=True,
exclude=[
"esphome/dashboard/static/ace.js",
"esphome/dashboard/static/ext-searchbox.js",
],
)
@lint_content_find_check("\t", only_first=True)
def lint_tabs(fname, line, col, content):
return "File contains tab character. Please convert tabs to spaces."
@@ -562,7 +555,7 @@ def lint_constants_usage():
# Maximum allowed CONF_ constants in esphome/const.py.
# This file is frozen — new constants go in esphome/components/const/__init__.py.
# Decrease this number when constants are moved out of const.py.
CONST_PY_MAX_CONF = 1013
CONST_PY_MAX_CONF = 1014
@lint_content_check(include=["esphome/const.py"])
+54 -13
View File
@@ -1,8 +1,8 @@
#!/usr/bin/env python3
"""Extract memory usage statistics from ESPHome build output.
This script parses the PlatformIO build output to extract RAM and flash
usage statistics for a compiled component. It's used by the CI workflow to
This script parses the build output to extract RAM and flash usage
statistics for a compiled component. It's used by the CI workflow to
compare memory usage between branches.
The script reads compile output from stdin and looks for the standard
@@ -10,6 +10,13 @@ PlatformIO output format:
RAM: [==== ] 36.1% (used 29548 bytes from 81920 bytes)
Flash: [=== ] 34.0% (used 348511 bytes from 1023984 bytes)
or the linker memory usage table printed by Zephyr native builds
(e.g. nRF52 with the sdk-nrf toolchain):
Memory region Used Size Region Size %age Used
FLASH: 90624 B 796 KB 11.12%
RAM: 22432 B 256 KB 8.56%
IDT_LIST: 0 GB 32 KB 0.00%
Optionally performs detailed memory analysis if a build directory is provided.
"""
@@ -34,20 +41,43 @@ _RAM_PATTERN = re.compile(r"RAM:\s+\[.*?\]\s+\d+\.\d+%\s+\(used\s+(\d+)\s+bytes"
_FLASH_PATTERN = re.compile(r"Flash:\s+\[.*?\]\s+\d+\.\d+%\s+\(used\s+(\d+)\s+bytes")
_BUILD_PATH_PATTERN = re.compile(r"Build path: (.+)")
# Zephyr native builds print the GNU ld --print-memory-usage table instead of
# the PlatformIO summary. Only the FLASH and RAM regions are real memory
# (IDT_LIST is a build-time pseudo-region discarded from the final image).
# Each cell is humanized to the largest unit that divides evenly, so used
# sizes are not always plain bytes (zero prints as "0 GB").
_ZEPHYR_RAM_PATTERN = re.compile(
r"^\s*RAM:\s+(\d+)\s*([KMG]?B)\s+\d+\s*[KMG]?B\s+\d+\.\d+%", re.MULTILINE
)
_ZEPHYR_FLASH_PATTERN = re.compile(
r"^\s*FLASH:\s+(\d+)\s*([KMG]?B)\s+\d+\s*[KMG]?B\s+\d+\.\d+%", re.MULTILINE
)
_ZEPHYR_UNIT_MULTIPLIERS = {"B": 1, "KB": 1024, "MB": 1024**2, "GB": 1024**3}
def _zephyr_bytes(matches: list[tuple[str, str]]) -> int:
"""Sum humanized (value, unit) pairs from the Zephyr memory table."""
return sum(int(value) * _ZEPHYR_UNIT_MULTIPLIERS[unit] for value, unit in matches)
def extract_from_compile_output(
output_text: str,
) -> tuple[int | None, int | None, str | None]:
"""Extract memory usage and build directory from PlatformIO compile output.
"""Extract memory usage and build directory from compile output.
Supports multiple builds (for component groups or isolated components).
When test_build_components.py creates multiple builds, this sums the
memory usage across all builds.
Looks for lines like:
Looks for PlatformIO lines like:
RAM: [==== ] 36.1% (used 29548 bytes from 81920 bytes)
Flash: [=== ] 34.0% (used 348511 bytes from 1023984 bytes)
and Zephyr (native west build) linker table rows like:
Memory region Used Size Region Size %age Used
FLASH: 90624 B 796 KB 11.12%
RAM: 22432 B 256 KB 8.56%
Also extracts build directory from lines like:
INFO Compiling app... Build path: /path/to/build
@@ -61,12 +91,20 @@ def extract_from_compile_output(
ram_matches = _RAM_PATTERN.findall(output_text)
flash_matches = _FLASH_PATTERN.findall(output_text)
if not ram_matches or not flash_matches:
# Zephyr native builds print the linker memory table instead
zephyr_ram_matches = _ZEPHYR_RAM_PATTERN.findall(output_text)
zephyr_flash_matches = _ZEPHYR_FLASH_PATTERN.findall(output_text)
if not (ram_matches or zephyr_ram_matches) or not (
flash_matches or zephyr_flash_matches
):
return None, None, None
# Sum all builds (handles multiple component groups)
total_ram = sum(int(match) for match in ram_matches)
total_flash = sum(int(match) for match in flash_matches)
total_ram += _zephyr_bytes(zephyr_ram_matches)
total_flash += _zephyr_bytes(zephyr_flash_matches)
# Extract build directory from ESPHome's explicit build path output
# Look for: INFO Compiling app... Build path: /path/to/build
@@ -202,20 +240,23 @@ def main() -> int:
)
if ram_bytes is None or flash_bytes is None:
print("Failed to extract memory usage from compile output", file=sys.stderr)
print("Expected lines like:", file=sys.stderr)
print(
" RAM: [==== ] 36.1% (used 29548 bytes from 81920 bytes)",
file=sys.stderr,
)
print(
" Flash: [=== ] 34.0% (used 348511 bytes from 1023984 bytes)",
"Failed to extract memory usage from compile output\n"
"Expected lines like:\n"
" RAM: [==== ] 36.1% (used 29548 bytes from 81920 bytes)\n"
" Flash: [=== ] 34.0% (used 348511 bytes from 1023984 bytes)\n"
"or a Zephyr linker memory usage table like:\n"
" Memory region Used Size Region Size %age Used\n"
" FLASH: 90624 B 796 KB 11.12%\n"
" RAM: 22432 B 256 KB 8.56%",
file=sys.stderr,
)
return 1
# Count how many builds were found
num_builds = len(_RAM_PATTERN.findall(compile_output))
num_builds = len(_RAM_PATTERN.findall(compile_output)) + len(
_ZEPHYR_RAM_PATTERN.findall(compile_output)
)
if num_builds > 1:
print(
+15 -4
View File
@@ -145,14 +145,16 @@ def clang_options(idedata):
# defines
cmd.extend(f"-D{define}" for define in idedata["defines"])
# add toolchain include directories using -isystem to suppress their errors
# toolchain include directories, using -isystem to suppress their errors
# idedata contains include directories for all toolchains of this platform, only use those from the one in use
toolchain_dir = os.path.normpath(f"{idedata['cxx_path']}/../../")
toolchain_includes = []
for directory in idedata["includes"]["toolchain"]:
if directory.startswith(toolchain_dir) and "picolibc" not in directory:
cmd.extend(["-isystem", directory])
toolchain_includes.extend(["-isystem", directory])
# add library include directories using -isystem to suppress their errors
# library include directories, using -isystem to suppress their errors
build_includes = []
for directory in list(idedata["includes"]["build"]):
# skip our own directories, we add those later
if (
@@ -166,7 +168,16 @@ def clang_options(idedata):
)
or (directory.startswith(f"{root_path}") and "/.pio/" in directory)
):
cmd.extend(["-isystem", directory])
build_includes.extend(["-isystem", directory])
if "zephyr" in triplet:
# Zephyr's POSIX layer shadows libc headers (sys/select.h, ...) with
# coherently-guarded versions; the real build searches the Zephyr
# include dirs before the toolchain's, and the shadowed headers clash
# (e.g. newlib's sigset_t vs Zephyr's) in the opposite order.
cmd.extend(build_includes + toolchain_includes)
else:
cmd.extend(toolchain_includes + build_includes)
# add the esphome include directory using -I
cmd.extend(["-I", root_path])
+2
View File
@@ -21,6 +21,8 @@ CLANG_TIDY_GLOBAL_FILES = (
"platformio.ini",
"requirements_dev.txt",
"esphome/idf_component.yml",
"esphome/components/esp32/__init__.py",
"esphome/components/nrf52/__init__.py",
)
# sdkconfig.defaults and per-target sdkconfig.defaults.<target> files flip the
+15 -5
View File
@@ -1338,7 +1338,7 @@ def main() -> None:
# Split components into batches for CI testing
# This intelligently groups components with similar bus configurations
component_test_batches: list[str]
component_test_batches: list[dict[str, Any]] = []
if changed_components_with_tests:
tests_dir = Path(root_path) / ESPHOME_TESTS_COMPONENTS_PATH
@@ -1363,10 +1363,20 @@ def main() -> None:
batch_size=COMPONENT_TEST_BATCH_SIZE,
directly_changed=batch_directly_changed,
)
# Convert batches to space-separated strings for CI matrix
component_test_batches = [" ".join(batch) for batch in batches]
else:
component_test_batches = []
# Convert batches to CI matrix entries: the component list plus which
# native toolchain installs the batch's test platforms need, so the
# workflow only restores the matching multi-GB toolchain caches.
for batch in batches:
platforms: set[str] = set()
for component in batch:
platforms.update(get_component_test_platforms(component))
component_test_batches.append(
{
"components": " ".join(batch),
"needs_idf": any(p.startswith("esp32") for p in platforms),
"needs_nrf": any(p.startswith("nrf52") for p in platforms),
}
)
output: dict[str, Any] = {
"core_ci": run_core_ci,
+78 -21
View File
@@ -238,6 +238,72 @@ class _ConflictWalk:
rejects: set[str]
@cache
def _get_test_config_components(component: str, platform: str) -> frozenset[str]:
"""Return the components referenced by a component's test config for a platform.
Loads ``tests/components/<component>/test.<platform>.yaml`` and extracts the
top-level component keys (and list ``platform:`` values). This lets the
conflict splitter see components that are only pulled in via a test config
(e.g. nRF52 ``network`` tests that also enable ``openthread``), which a
purely static AUTO_LOAD/CONFLICTS_WITH parse cannot discover -- notably for
components like ``api`` whose ``AUTO_LOAD`` is a callable.
Failures (missing file, parse error) are treated as empty so the splitter
never crashes on a malformed or absent test config.
"""
from esphome import yaml_util
test_file = (
Path(root_path) / "tests" / "components" / component / f"test.{platform}.yaml"
)
if not test_file.exists():
return frozenset()
try:
config = yaml_util.load_yaml(test_file)
except Exception: # noqa: BLE001 - never let a bad test config crash grouping
# Matches analyze_component_buses, which loads these same files and
# silently tolerates parse failures; surfacing it only here would be
# inconsistent and noisy.
return frozenset()
if not isinstance(config, dict):
return frozenset()
return frozenset(_extract_components_from_yaml(config))
@cache
def _conflict_walk(comp: str, platform: str) -> _ConflictWalk:
"""Build the platform-aware conflict walk for a single component.
Seeds the walk with the component itself plus any components pulled in via
its ``test.<platform>.yaml`` config, then folds in each seed's static
AUTO_LOAD closure and CONFLICTS_WITH declarations. Cached per
``(component, platform)`` since the test-config seeds are platform-specific.
"""
seeds = {comp} | set(_get_test_config_components(comp, platform))
walk = _ConflictWalk(loaded=set(seeds), rejects=set())
stack = list(seeds)
while stack:
metadata = parse_component_metadata(stack.pop())
walk.rejects |= metadata.conflicts_with
new = metadata.auto_load - walk.loaded
walk.loaded |= new
stack.extend(new)
return walk
def components_conflict(a: str, b: str, platform: str) -> bool:
"""Return True if components ``a`` and ``b`` cannot share a build on ``platform``.
Uses the same platform-aware conflict walk as :func:`split_conflicting_groups`
so callers (e.g. the no-bus redistribution in ``test_build_components.py``)
agree with how groups were originally split. The conflict relation is
symmetric even when only one side declares CONFLICTS_WITH.
"""
wa, wb = _conflict_walk(a, platform), _conflict_walk(b, platform)
return not wa.rejects.isdisjoint(wb.loaded) or not wb.rejects.isdisjoint(wa.loaded)
def split_conflicting_groups(
grouped_components: dict[tuple[str, str], list[str]],
) -> dict[tuple[str, str], list[str]]:
@@ -250,33 +316,24 @@ def split_conflicting_groups(
conflict relation is treated as symmetric even when only one side
declares it (e.g. ethernet rejects wifi but wifi does not declare the
reverse).
The walk is platform-aware: in addition to the static AUTO_LOAD closure,
each ``(component, platform)`` walk is seeded with the components found in
that component's ``test.<platform>.yaml`` config. This catches conflicts
that only exist on a given platform and are expressed through the test
config rather than static metadata -- e.g. on nRF52 the ``network``/``api``
test configs also enable ``openthread``, which ``zigbee`` declares a
conflict with, so ``api`` and ``zigbee`` end up split there. On ESP32 those
test configs have no ``openthread``, so the components still group together.
"""
batch = {c for comps in grouped_components.values() for c in comps}
walks: dict[str, _ConflictWalk] = {}
for comp in batch:
walk = _ConflictWalk(loaded={comp}, rejects=set())
stack = [comp]
while stack:
metadata = parse_component_metadata(stack.pop())
walk.rejects |= metadata.conflicts_with
new = metadata.auto_load - walk.loaded
walk.loaded |= new
stack.extend(new)
walks[comp] = walk
def conflicts(a: str, b: str) -> bool:
wa, wb = walks[a], walks[b]
return not wa.rejects.isdisjoint(wb.loaded) or not wb.rejects.isdisjoint(
wa.loaded
)
result: dict[tuple[str, str], list[str]] = {}
for (platform, signature), components in grouped_components.items():
buckets: list[list[str]] = []
for comp in components:
for bucket in buckets:
if not any(conflicts(comp, other) for other in bucket):
if not any(
components_conflict(comp, other, platform) for other in bucket
):
bucket.append(comp)
break
else:
+57 -92
View File
@@ -1,59 +1,32 @@
"""Load clang-tidy idedata for the nrf52/Zephyr environment.
The compile commands come from a configure-only build of a minimal Zephyr
project using the native sdk-nrf toolchain (see
``esphome.components.nrf52.clang_tidy``); this module extracts the include
paths, defines and compiler flags clang-tidy needs from them.
"""
import json
import os
from pathlib import Path
import re
import shlex
import subprocess
def load_idedata(environment, temp_folder, platformio_ini):
build_environment = environment.replace("-tidy", "")
build_dir = Path(temp_folder) / f"build-{build_environment}"
Path(build_dir).mkdir(exist_ok=True)
Path(build_dir / "platformio.ini").write_text(
Path(platformio_ini).read_text(encoding="utf-8"), encoding="utf-8"
)
esphome_dir = Path(build_dir / "esphome")
esphome_dir.mkdir(exist_ok=True)
Path(esphome_dir / "main.cpp").write_text(
"""
#include <zephyr/kernel.h>
int main() { return 0;}
extern "C" void zboss_signal_handler() {};
""",
encoding="utf-8",
)
zephyr_dir = Path(build_dir / "zephyr")
zephyr_dir.mkdir(exist_ok=True)
Path(zephyr_dir / "prj.conf").write_text(
"""
CONFIG_NEWLIB_LIBC=y
CONFIG_BT=y
CONFIG_ADC=y
#mcumgr begin
CONFIG_NET_BUF=y
CONFIG_ZCBOR=y
CONFIG_MCUMGR=y
CONFIG_MCUMGR_GRP_IMG=y
CONFIG_IMG_MANAGER=y
CONFIG_STREAM_FLASH=y
CONFIG_FLASH_MAP=y
CONFIG_FLASH=y
CONFIG_IMG_ERASE_PROGRESSIVELY=y
CONFIG_BOOTLOADER_MCUBOOT=y
CONFIG_MCUMGR_MGMT_NOTIFICATION_HOOKS=y
CONFIG_MCUMGR_GRP_IMG_STATUS_HOOKS=y
CONFIG_MCUMGR_GRP_IMG_UPLOAD_CHECK_HOOK=y
CONFIG_MCUMGR_TRANSPORT_UART=y
#mcumgr end
#zigbee begin
CONFIG_ZIGBEE=y
CONFIG_CRYPTO=y
CONFIG_NVS=y
CONFIG_SETTINGS=y
#zigbee end
""",
encoding="utf-8",
)
subprocess.run(["pio", "run", "-e", build_environment, "-d", build_dir], check=True)
if explicit := os.environ.get("ESPHOME_ZEPHYR_COMPILE_COMMANDS"):
compile_commands_path = Path(explicit)
else:
from esphome.components.nrf52.clang_tidy import generate_compile_commands
work_dir = (Path(temp_folder) / f"zephyr-{environment}").resolve()
compile_commands_path = generate_compile_commands(
work_dir, Path(platformio_ini)
)
if not compile_commands_path.is_file():
raise RuntimeError(f"compile_commands.json not found: {compile_commands_path}")
def extract_include_paths(command):
include_paths = []
@@ -62,7 +35,7 @@ CONFIG_SETTINGS=y
split_strings = re.split(
r"\s*-\s*(?:I|isystem)", list(filter(lambda x: x, match))[0]
)
include_paths.append(split_strings[1])
include_paths.append(split_strings[1].strip())
return include_paths
def extract_defines(command):
@@ -74,15 +47,6 @@ CONFIG_SETTINGS=y
if not any(match.startswith(prefix) for prefix in ignore_prefixes)
]
def find_cxx_path(commands):
for entry in commands:
command = entry["command"]
cxx_path = command.split()[0]
if not cxx_path.endswith("++"):
continue
return cxx_path
return None
def get_builtin_include_paths(compiler):
result = subprocess.run(
[compiler, "-E", "-x", "c++", "-", "-v"],
@@ -105,47 +69,48 @@ CONFIG_SETTINGS=y
return include_paths
def extract_cxx_flags(command):
# Extracts CXXFLAGS from the command string, excluding includes and defines.
# Extracts CXXFLAGS from the command string, excluding includes and
# defines. Anchored per token: a substring match would extract a bogus
# "-format-zero-length" from -Wno-format-zero-length.
flag_pattern = re.compile(
r"(-O[0-3s]|-g|-std=[^\s]+|-Wall|-Wextra|-Werror|--[^\s]+|-f[^\s]+|-m[^\s]+|-imacros\s*[^\s]+)"
r"^(-O[0-3s]|-g|-std=.+|-Wall|-Wextra|-Werror|--.+|-f.+|-m.+|-imacros.+)$"
)
return [
match.replace("-imacros ", "-imacros")
for match in flag_pattern.findall(command)
]
flags = []
tokens = shlex.split(command)
for i, token in enumerate(tokens):
if token == "-imacros" and i + 1 < len(tokens):
flags.append(f"-imacros{tokens[i + 1]}")
elif flag_pattern.match(token):
flags.append(token)
return flags
def transform_to_idedata_format(compile_commands):
cxx_path = find_cxx_path(compile_commands)
idedata = {
# Use only the tidy app TU (main.cpp): as the app target, its compile
# command already carries the full Zephyr include set. Unioning every
# TU instead would drag in per-library internal include dirs (e.g. the
# Zephyr POSIX shim, whose signal.h redefines newlib's sigset_t) that
# no ESPHome source compiles against.
entry = next(
(e for e in compile_commands if e["file"].endswith("main.cpp")), None
)
if entry is None:
raise RuntimeError("tidy main.cpp not found in compile_commands.json")
command = entry["command"]
# Find the compiler by name: the command may be prefixed with a
# launcher (Zephyr auto-enables ccache when present).
cxx_path = next((t for t in shlex.split(command) if t.endswith("++")), None)
if cxx_path is None:
raise RuntimeError(f"no C++ compiler in compile command: {command}")
return {
"includes": {
"toolchain": get_builtin_include_paths(cxx_path),
"build": set(),
"build": extract_include_paths(command),
},
"defines": set(),
"defines": extract_defines(command),
"cxx_path": cxx_path,
"cxx_flags": set(),
"cxx_flags": extract_cxx_flags(command),
}
for entry in compile_commands:
command = entry["command"]
exec = command.split()[0]
if exec != cxx_path:
continue
idedata["includes"]["build"].update(extract_include_paths(command))
idedata["defines"].update(extract_defines(command))
idedata["cxx_flags"].update(extract_cxx_flags(command))
# Convert sets to lists for JSON serialization
idedata["includes"]["build"] = list(idedata["includes"]["build"])
idedata["defines"] = list(idedata["defines"])
idedata["cxx_flags"] = list(idedata["cxx_flags"])
return idedata
compile_commands = json.loads(
Path(
build_dir / ".pio" / "build" / build_environment / "compile_commands.json"
).read_text(encoding="utf-8")
)
compile_commands = json.loads(compile_commands_path.read_text(encoding="utf-8"))
return transform_to_idedata_format(compile_commands)
+1 -1
View File
@@ -1,5 +1,5 @@
{
"target_module": "esphome.__main__",
"margin_pct": 15,
"margin_pct": 20,
"cumulative_us": 91000
}
+1 -1
View File
@@ -139,7 +139,7 @@ def main():
print()
print("Running pyupgrade...")
print()
PYUPGRADE_TARGET = "--py311-plus"
PYUPGRADE_TARGET = "--py312-plus"
for files in filesets:
cmd = ["pyupgrade", PYUPGRADE_TARGET] + files
log = get_err(*cmd)
+13 -2
View File
@@ -4,7 +4,12 @@
set -e
cd "$(dirname "$0")/.."
if [ ! -n "$VIRTUAL_ENV" ]; then
if [ -n "$VIRTUAL_ENV" ]; then
# A virtual environment is already active (e.g. the devcontainer's pre-provisioned
# esphome-venv). Install into it rather than creating a ./venv in the workspace.
created_venv=false
else
created_venv=true
if [ -x "$(command -v uv)" ]; then
uv venv --seed venv
else
@@ -26,4 +31,10 @@ mkdir -p .temp
echo
echo
echo "Virtual environment created. Run 'source venv/bin/activate' to use it."
if [ "$created_venv" = true ]; then
echo "Virtual environment created at ./venv. Run 'source venv/bin/activate' to use it."
else
echo "Dependencies installed into the active virtual environment:"
echo " $VIRTUAL_ENV"
echo "It is already active in this shell, so no 'source venv/bin/activate' is needed."
fi
+30 -8
View File
@@ -40,6 +40,7 @@ from script.analyze_component_buses import (
uses_local_file_references,
)
from script.helpers import (
components_conflict,
get_component_test_files,
is_validate_only_file,
parse_test_filename,
@@ -788,14 +789,35 @@ def run_grouped_component_tests(
if plat == platform and sig != NO_BUSES_SIGNATURE
]
if platform_groups:
# Distribute no_buses components round-robin across existing groups
for i, comp in enumerate(no_buses_comps):
sig, _ = platform_groups[i % len(platform_groups)]
grouped_components[(platform, sig)].append(comp)
else:
# No other groups for this platform - keep no_buses components together
grouped_components[(platform, NO_BUSES_SIGNATURE)] = no_buses_comps
# Distribute no_buses components round-robin across existing groups,
# but never place a component into a group it conflicts with. Conflict
# splitting (split_conflicting_groups) may have created sibling groups
# like "no_buses__conflict1" precisely to keep incompatible components
# apart (e.g. on nRF52, network pulls in openthread which zigbee
# conflicts with); redistribution must not silently undo that split.
leftover: list[str] = []
for i, comp in enumerate(no_buses_comps):
placed = False
# Try groups starting at the round-robin offset to keep the spread.
for offset in range(len(platform_groups)):
sig, comps = platform_groups[(i + offset) % len(platform_groups)]
if any(components_conflict(comp, other, platform) for other in comps):
continue
# comps is the same list object stored in grouped_components, so
# this also extends the group in grouped_components.
comps.append(comp)
placed = True
break
if not placed:
leftover.append(comp)
if leftover:
# Components that conflict with every existing group stay together in
# their own no_buses group (they were grouped before, so they don't
# conflict with each other).
grouped_components.setdefault((platform, NO_BUSES_SIGNATURE), []).extend(
leftover
)
groups_to_test = []
individual_tests = set() # Use set to avoid duplicates