feat(tools): mcp-server added monitor-device tool

Add a `monitor_device` MCP tool that lets an AI agent run a scripted,
non-interactive `esp-idf-monitor` session against a flashed device and
get back a short status plus a log file path, instead of raw serial
output inline.

Under the hood:

- The agent supplies a plain-text command body (expect/send/sleep/reset/
  exit/comments). `assemble_monitor_script_from_agent_commands()` frames
  it into a script the monitor's non-interactive command mode can
  consume via stdin: it appends `exit` if the agent didn't already end
  with one, and rewrites every bare `expect <regex>` into `expect
  --timeout <timeout_sec> <regex>` via `_monitor_normalize_expect_line()`
  (an already-bounded `expect --timeout ...` line is left untouched so
  the monitor itself reports a bad value). A leading `reset` is not
  prepended - the monitor already resets the chip when it opens the
  port - and any `reset` the agent wrote is left in place.
  `_monitor_parse_sleep_duration()` extracts each `sleep <n>` duration.
  The effective timeout is the sum of every bounded expect duration
  plus every sleep duration. Scripts whose sum exceeds
  `MONITOR_MAX_SCRIPT_SEC` are rejected. If the script has neither
  expect nor sleep (for example only `send`), `timeout_sec` is used so
  the process still has a kill bound.

- `monitor_device()` runs `python -m esp_idf_monitor` via
  `subprocess.run(..., input=script, timeout=2 * effective_timeout)`.
  `no_reset` is forwarded as `--no-reset` so the connection reset can be
  skipped; an explicit `-p` is forwarded when a port is given. Extra
  arguments match `idf.py monitor` where a build exists: baud (`baud`
  tool arg, else `monitor_baud` from `project_description.json`),
  toolchain prefix, `--target`/`--revision`, coredump/panic decode, and
  ELF files with the app ELF first. The 2x hard timeout is a safety net
  independent of the script's own `expect --timeout`/`exit` logic; on
  `TimeoutExpired` the process is killed but any output already captured
  is preserved and logged. `decode_stream()` normalizes that captured
  output, which can be `bytes` on the timeout path even though the
  process otherwise runs in text mode.

- The monitor's exit code drives the reported status via
  `_monitor_status()`, using `EXIT_EXPECT_TIMEOUT` and
  `EXIT_SCRIPT_ERROR` from `esp_idf_monitor.base.constants`: 0 is
  success, 110 means an `expect` pattern never showed up before its
  `--timeout` elapsed, 2 means the monitor rejected the script (bad
  syntax/timeout/regex), anything else is reported generically.

- Serial output and the monitor's own messages share one pipe
  (`stderr=STDOUT`) so decoded panic backtraces stay next to the lines
  that triggered them. `_save_monitor_output()` writes the full merge to
  `<tempdir>/esp_idf_mcp_log/action_monitor/monitor_<timestamp>.log` and
  reports a dedicated `Log file:` line. On non-zero exit or process
  kill, a short tail of that same merge is also returned inline so the
  agent has some failure context without a second file read. If the log
  file can't be written, it falls back to inlining a truncated tail.

Closes https://github.com/espressif/esp-idf/issues/18757
Closes https://github.com/espressif/esp-idf/pull/18385

Co-authored-by: Cursor <cursoragent@cursor.com>
This commit is contained in:
Marek Fiala
2026-09-11 15:20:45 +02:00
co-authored by Cursor
parent 7e5235a7a7
commit 3f005c8a4d
4 changed files with 824 additions and 23 deletions
+401 -2
View File
@@ -11,7 +11,9 @@ import importlib
import importlib.util
import json
import os
import subprocess
import sys
import tempfile
import types
from collections.abc import Callable
from pathlib import Path
@@ -106,6 +108,7 @@ def mcp_ext(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> tuple[types.Modu
tools_mod.PropertyDict = dict # type: ignore[attr-defined]
tools_mod.get_target = mock.Mock(return_value='esp32') # type: ignore[attr-defined]
tools_mod.idf_version = mock.Mock(return_value='5.4.0') # type: ignore[attr-defined]
tools_mod.get_sdkconfig_value = mock.Mock(return_value=None) # type: ignore[attr-defined]
idf_py_actions_pkg.errors = errors_mod # type: ignore[attr-defined]
idf_py_actions_pkg.tools = tools_mod # type: ignore[attr-defined]
@@ -115,6 +118,13 @@ def mcp_ext(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> tuple[types.Modu
mcp_server_pkg = types.ModuleType('mcp.server')
mcp_server_pkg.MCPServer = lambda name: mock_mcp_instance # type: ignore[attr-defined]
# Stub monitor exit codes (same values as esp_idf_monitor.base.constants).
esp_idf_monitor_pkg = types.ModuleType('esp_idf_monitor')
esp_idf_monitor_base = types.ModuleType('esp_idf_monitor.base')
esp_idf_monitor_constants = types.ModuleType('esp_idf_monitor.base.constants')
esp_idf_monitor_constants.EXIT_EXPECT_TIMEOUT = 110 # type: ignore[attr-defined]
esp_idf_monitor_constants.EXIT_SCRIPT_ERROR = 2 # type: ignore[attr-defined]
stubs = {
'rich_click': rich_click,
'idf_py_actions': idf_py_actions_pkg,
@@ -122,6 +132,9 @@ def mcp_ext(monkeypatch: pytest.MonkeyPatch, tmp_path: Path) -> tuple[types.Modu
'idf_py_actions.tools': tools_mod,
'mcp': mcp_pkg,
'mcp.server': mcp_server_pkg,
'esp_idf_monitor': esp_idf_monitor_pkg,
'esp_idf_monitor.base': esp_idf_monitor_base,
'esp_idf_monitor.base.constants': esp_idf_monitor_constants,
}
for name, stub_mod in stubs.items():
monkeypatch.setitem(sys.modules, name, stub_mod)
@@ -176,13 +189,17 @@ class TestIsValidProjectDir:
mod, _ = mcp_ext
assert mod._is_valid_project_dir(str(tmp_path / 'nonexistent')) is False
def test_directory_without_cmakelists(self, tmp_path: Path, mcp_ext: tuple[types.ModuleType, _MockMCPServer]) -> None:
def test_directory_without_cmakelists(
self, tmp_path: Path, mcp_ext: tuple[types.ModuleType, _MockMCPServer]
) -> None:
mod, _ = mcp_ext
d = tmp_path / 'no_cmake'
d.mkdir()
assert mod._is_valid_project_dir(str(d)) is False
def test_cmakelists_without_idf_line(self, tmp_path: Path, mcp_ext: tuple[types.ModuleType, _MockMCPServer]) -> None:
def test_cmakelists_without_idf_line(
self, tmp_path: Path, mcp_ext: tuple[types.ModuleType, _MockMCPServer]
) -> None:
mod, _ = mcp_ext
proj = _make_invalid_project(tmp_path / 'plain')
assert mod._is_valid_project_dir(str(proj)) is False
@@ -481,6 +498,387 @@ class TestCleanProject:
assert cmd[cmd.index('-C') + 1] == str(proj)
# ---------------------------------------------------------------------------
# Tests: monitor script building
# ---------------------------------------------------------------------------
class TestBuildMonitorScript:
"""Covers only what the MCP server itself is responsible for: framing the
agent's command body (exit/comments) and computing the effective
timeout. Validity of esp-idf-monitor's own 'expect --timeout' syntax is
esp-idf-monitor's job, not duplicated here."""
def test_commands_without_an_actual_command_are_rejected(
self, mcp_ext: tuple[types.ModuleType, _MockMCPServer]
) -> None:
mod, _ = mcp_ext
with pytest.raises(ValueError, match='No monitor commands given'):
mod.assemble_monitor_script_from_agent_commands(' \n# just a note\n')
def test_bare_expect_gets_default_timeout_and_framing(
self, mcp_ext: tuple[types.ModuleType, _MockMCPServer]
) -> None:
mod, _ = mcp_ext
script, effective = mod.assemble_monitor_script_from_agent_commands('expect Hello world!')
assert script == 'expect --timeout 20 Hello world!\nexit\n'
assert effective == 20.0
@pytest.mark.parametrize(
'commands, timeout_sec, expected_effective, expected_snippet',
[
('expect --timeout 45 ALL TESTS PASSED', 20.0, 45.0, 'expect --timeout 45 ALL TESTS PASSED'),
(
'expect --timeout 5 first\nexpect --timeout 90 second\nexpect third',
30.0,
125.0,
'expect --timeout 30 third',
),
('sleep 120\nexpect ready\nsleep 30', 20.0, 170.0, 'expect --timeout 20 ready'),
('expect --timeout abc pattern', 20.0, 20.0, 'expect --timeout abc pattern'),
('send hello', 20.0, 20.0, 'send hello'),
],
ids=[
'keeps_explicit_timeout',
'expect_timeouts_are_summed_and_default_is_injected',
'sleeps_are_added_to_the_sum',
'malformed_timeout_left_to_the_monitor',
'send_only_uses_timeout_sec_as_kill_bound',
],
)
def test_timeout_handling(
self,
mcp_ext: tuple[types.ModuleType, _MockMCPServer],
commands: str,
timeout_sec: float,
expected_effective: float,
expected_snippet: str,
) -> None:
mod, _ = mcp_ext
script, effective = mod.assemble_monitor_script_from_agent_commands(commands, timeout_sec=timeout_sec)
assert expected_snippet in script
assert effective == expected_effective
def test_script_at_max_duration_is_accepted(self, mcp_ext: tuple[types.ModuleType, _MockMCPServer]) -> None:
mod, _ = mcp_ext
_, effective = mod.assemble_monitor_script_from_agent_commands(
'sleep 580\nexpect --timeout 20 ready', timeout_sec=20.0
)
assert effective == 600.0
@pytest.mark.parametrize(
'commands, timeout_sec',
[
('sleep 99999\nexpect ready', 20.0),
('expect --timeout 601 ready', 20.0),
('expect --timeout 400 a\nexpect --timeout 250 b', 20.0),
],
ids=['long_sleep', 'long_expect', 'sum_of_expects'],
)
def test_script_over_max_duration_is_rejected(
self, mcp_ext: tuple[types.ModuleType, _MockMCPServer], commands: str, timeout_sec: float
) -> None:
mod, _ = mcp_ext
with pytest.raises(ValueError, match='The limit is'):
mod.assemble_monitor_script_from_agent_commands(commands, timeout_sec=timeout_sec)
@pytest.mark.parametrize(
'commands, expected_script',
[
('expect uptime', 'expect --timeout 20 uptime\nexit\n'),
('reset\nexpect done\nexit\n', 'reset\nexpect --timeout 20 done\nexit\n'),
(
'expect first\nreset\nexpect second',
'expect --timeout 20 first\nreset\nexpect --timeout 20 second\nexit\n',
),
],
ids=['exit_appended_and_reset_not_prepended', 'agent_reset_and_exit_kept', 'mid_script_reset_kept'],
)
def test_script_framing(
self, mcp_ext: tuple[types.ModuleType, _MockMCPServer], commands: str, expected_script: str
) -> None:
mod, _ = mcp_ext
script, _ = mod.assemble_monitor_script_from_agent_commands(commands)
assert script == expected_script
@pytest.mark.parametrize('command', ['flash', 'app-flash', 'log', 'output'])
def test_disallowed_commands_are_rejected(
self, mcp_ext: tuple[types.ModuleType, _MockMCPServer], command: str
) -> None:
mod, _ = mcp_ext
with pytest.raises(ValueError, match=rf'Unsupported monitor command {command!r}'):
mod.assemble_monitor_script_from_agent_commands(f'{command}\nexpect ready')
# ---------------------------------------------------------------------------
# Tests: monitor_device tool
# ---------------------------------------------------------------------------
@pytest.fixture()
def monitor_tools(
mcp_ext: tuple[types.ModuleType, _MockMCPServer],
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> dict[str, Callable[..., Any]]:
"""Registered tools with monitor logs redirected into tmp_path."""
_, mock_mcp = mcp_ext
monkeypatch.setattr(tempfile, 'tempdir', str(tmp_path))
tools, _ = _start_server(mcp_ext, mock_mcp, str(tmp_path))
return tools
def _log_path_from(result: str) -> Path:
"""Extract the log file path the tool reported."""
for line in result.splitlines():
if line.startswith('Log file: '):
return Path(line.split(': ', 1)[1])
raise AssertionError(f'no serial log path in result:\n{result}')
class TestMonitorDevice:
"""Covers what the MCP tool itself is responsible for: turning a script and
exit code into a status message, wiring subprocess/log-file plumbing, and
forwarding its own arguments. Behaviour of esp-idf-monitor's non-interactive
command mode (e.g. what makes an 'expect --timeout' line valid) is exercised
by esp-idf-monitor's own tests, not duplicated here."""
@pytest.mark.parametrize(
'commands, timeout_sec, expected_snippet',
[
(' ', 20.0, 'No monitor commands given'),
('expect ready', 0, 'timeout_sec must be a finite number greater than 0'),
('flash\nexpect ready', 20.0, "Unsupported monitor command 'flash'"),
],
ids=['empty_commands', 'invalid_timeout_sec', 'disallowed_command'],
)
def test_guard_clauses_return_error_without_running(
self, monitor_tools: dict[str, Callable[..., Any]], commands: str, timeout_sec: float, expected_snippet: str
) -> None:
with mock.patch('subprocess.run') as mock_run:
result = monitor_tools['monitor_device'](commands=commands, timeout_sec=timeout_sec)
assert expected_snippet in result
mock_run.assert_not_called()
@pytest.mark.parametrize(
'returncode, output, expected_status',
[
(0, 'Hello world!\n', 'completed successfully (exit code 0)'),
(110, "boot\nExpect pattern 'nope' timed out after 20.0s", 'not seen before its --timeout elapsed'),
(2, 'Invalid expect timeout value: (must be a finite number > 0)', 'rejected the script'),
(1, 'could not open port', 'exited with code 1'),
],
ids=['success', 'expect_timeout', 'script_error', 'unexpected_code'],
)
def test_status_message_and_log_reflect_exit_code(
self,
monitor_tools: dict[str, Callable[..., Any]],
returncode: int,
output: str,
expected_status: str,
) -> None:
with mock.patch('subprocess.run') as mock_run:
mock_run.return_value = mock.Mock(returncode=returncode, stdout=output, stderr=None)
result = monitor_tools['monitor_device'](commands='expect ready')
assert expected_status in result
if returncode != 0:
# Why the monitor stopped is its last output, so the tail carries it.
assert output in result
log_file = _log_path_from(result)
assert log_file.is_absolute()
assert log_file.read_text(encoding='utf-8') == output
@pytest.mark.parametrize(
'call_kwargs, expected_in_cmd, expected_not_in_cmd',
[
({'port': '/dev/ttyUSB0'}, ['-p', '/dev/ttyUSB0'], []),
({'no_reset': True}, ['--no-reset'], []),
],
ids=['port_forwarded', 'no_reset_passed_to_the_monitor'],
)
def test_cli_arguments_are_forwarded(
self,
monitor_tools: dict[str, Callable[..., Any]],
call_kwargs: dict[str, Any],
expected_in_cmd: list[str],
expected_not_in_cmd: list[str],
) -> None:
with mock.patch('subprocess.run') as mock_run:
mock_run.return_value = mock.Mock(returncode=0, stdout='', stderr='')
monitor_tools['monitor_device'](commands='expect ready', **call_kwargs)
cmd = mock_run.call_args[0][0]
assert cmd[:3] == [sys.executable, '-m', 'esp_idf_monitor']
for item in expected_in_cmd:
assert item in cmd
for item in expected_not_in_cmd:
assert item not in cmd
def test_hard_timeout_is_twice_the_effective_timeout(self, monitor_tools: dict[str, Callable[..., Any]]) -> None:
"""The effective-timeout math itself (sum of waits plus sleeps) is covered
by TestBuildMonitorScript; this only checks the tool wires 2x it in."""
with mock.patch('subprocess.run') as mock_run:
mock_run.return_value = mock.Mock(returncode=0, stdout='', stderr='')
monitor_tools['monitor_device'](commands='sleep 60\nexpect --timeout 20 ready')
assert mock_run.call_args[1]['timeout'] == 2 * 80.0
def test_process_timeout_keeps_partial_output(self, monitor_tools: dict[str, Callable[..., Any]]) -> None:
with mock.patch('subprocess.run') as mock_run:
mock_run.side_effect = subprocess.TimeoutExpired(
cmd=['python', '-m', 'esp_idf_monitor'],
timeout=60.0,
output=b'partial serial output\nExpect still waiting\n',
)
result = monitor_tools['monitor_device'](commands='expect ready')
assert 'was killed after 40 seconds' in result
assert 'Expect still waiting' in result
logged = _log_path_from(result).read_text(encoding='utf-8')
assert logged == 'partial serial output\nExpect still waiting\n'
def test_launch_failure_is_reported(self, monitor_tools: dict[str, Callable[..., Any]]) -> None:
with mock.patch('subprocess.run') as mock_run:
mock_run.side_effect = OSError('no such interpreter')
result = monitor_tools['monitor_device'](commands='expect ready')
assert 'Failed to run the monitor' in result
assert 'no such interpreter' in result
def test_log_write_failure_falls_back_to_inline_tail(
self,
monitor_tools: dict[str, Callable[..., Any]],
monkeypatch: pytest.MonkeyPatch,
) -> None:
monkeypatch.setattr(Path, 'mkdir', mock.Mock(side_effect=OSError('disk full')))
with mock.patch('subprocess.run') as mock_run:
mock_run.return_value = mock.Mock(returncode=0, stdout='important tail\n', stderr='')
result = monitor_tools['monitor_device'](commands='expect ready')
assert 'Log file could not be written (disk full)' in result
assert 'important tail' in result
def test_long_output_is_not_inlined(self, monitor_tools: dict[str, Callable[..., Any]]) -> None:
"""The whole point of the log file: a huge serial dump must not come back
through the tool result."""
big_output = 'x' * 200000
with mock.patch('subprocess.run') as mock_run:
mock_run.return_value = mock.Mock(returncode=0, stdout=big_output, stderr='')
result = monitor_tools['monitor_device'](commands='expect ready')
assert len(result) < 500
assert _log_path_from(result).read_text(encoding='utf-8') == big_output
def test_both_streams_are_captured_through_one_pipe(self, monitor_tools: dict[str, Callable[..., Any]]) -> None:
"""esp-idf-monitor prints decoded panic backtraces on stderr, so a log
built from stdout alone would lose exactly what the ELF files enable."""
with mock.patch('subprocess.run') as mock_run:
mock_run.return_value = mock.Mock(returncode=0, stdout='', stderr=None)
monitor_tools['monitor_device'](commands='expect ready')
assert mock_run.call_args[1]['stderr'] == subprocess.STDOUT
def test_no_baud_elf_or_metadata_when_description_absent(
self, monitor_tools: dict[str, Callable[..., Any]]
) -> None:
with mock.patch('subprocess.run') as mock_run:
mock_run.return_value = mock.Mock(returncode=0, stdout='', stderr='')
monitor_tools['monitor_device'](commands='expect ready')
cmd = mock_run.call_args[0][0]
assert '-b' not in cmd
assert '--toolchain-prefix' not in cmd
assert '--target' not in cmd
assert '--revision' not in cmd
assert '--decode-coredumps' not in cmd
assert '--decode-panic' not in cmd
assert not any(str(arg).endswith('.elf') for arg in cmd)
def test_forwards_baud_elf_order_and_metadata_when_description_has_elfs(
self,
mcp_ext: tuple[types.ModuleType, _MockMCPServer],
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
mod, mock_mcp = mcp_ext
monkeypatch.setattr(tempfile, 'tempdir', str(tmp_path))
monkeypatch.setenv('IDF_MCP_WORKSPACE_FOLDER', '')
proj = _make_valid_project(tmp_path / 'proj')
build_dir = proj / 'build'
boot_dir = build_dir / 'bootloader'
boot_dir.mkdir(parents=True)
app_elf = build_dir / 'hello_world.elf'
boot_elf = boot_dir / 'bootloader.elf'
app_elf.write_bytes(b'')
boot_elf.write_bytes(b'')
(build_dir / 'project_description.json').write_text(
json.dumps(
{
'app_elf': 'hello_world.elf',
'monitor_baud': '115200',
'monitor_toolprefix': 'xtensa-esp32-elf-',
'target': 'esp32',
'min_rev': '3',
'config_file': str(proj / 'sdkconfig'),
}
),
encoding='utf-8',
)
monkeypatch.setattr(
mod,
'get_sdkconfig_value',
lambda _path, key: {
'CONFIG_ESP_COREDUMP_DECODE': 'info',
'CONFIG_IDF_TARGET_ARCH_RISCV': 'y',
}.get(key),
)
tools, _ = _start_server(mcp_ext, mock_mcp, str(proj))
with mock.patch('subprocess.run') as mock_run:
mock_run.return_value = mock.Mock(returncode=0, stdout='', stderr='')
tools['monitor_device'](commands='expect ready')
cmd = mock_run.call_args[0][0]
assert cmd[cmd.index('-b') + 1] == '115200'
elf_args = [arg for arg in cmd if str(arg).endswith('.elf')]
assert elf_args[0] == str(app_elf)
assert str(boot_elf) in elf_args
assert cmd[cmd.index('--toolchain-prefix') + 1] == 'xtensa-esp32-elf-'
assert cmd[cmd.index('--target') + 1] == 'esp32'
assert cmd[cmd.index('--revision') + 1] == '3'
assert cmd[cmd.index('--decode-coredumps') + 1] == 'info'
assert cmd[cmd.index('--decode-panic') + 1] == 'backtrace'
def test_explicit_baud_overrides_description_monitor_baud(
self,
mcp_ext: tuple[types.ModuleType, _MockMCPServer],
tmp_path: Path,
monkeypatch: pytest.MonkeyPatch,
) -> None:
_, mock_mcp = mcp_ext
monkeypatch.setattr(tempfile, 'tempdir', str(tmp_path))
monkeypatch.setenv('IDF_MCP_WORKSPACE_FOLDER', '')
proj = _make_valid_project(tmp_path / 'proj')
build_dir = proj / 'build'
build_dir.mkdir()
(build_dir / 'project_description.json').write_text(
json.dumps({'monitor_baud': '115200', 'app_elf': 'hello_world.elf'}),
encoding='utf-8',
)
tools, _ = _start_server(mcp_ext, mock_mcp, str(proj))
with mock.patch('subprocess.run') as mock_run:
mock_run.return_value = mock.Mock(returncode=0, stdout='', stderr='')
tools['monitor_device'](commands='expect ready', baud='9600')
cmd = mock_run.call_args[0][0]
assert cmd[cmd.index('-b') + 1] == '9600'
assert '115200' not in cmd
# ---------------------------------------------------------------------------
# Test: server starts without error when project_path is not valid
# ---------------------------------------------------------------------------
@@ -517,6 +915,7 @@ class TestServerStartsOutsideProject:
assert 'set_target' in tools
assert 'flash_project' in tools
assert 'clean_project' in tools
assert 'monitor_device' in tools
# ---------------------------------------------------------------------------