From f103e79d9c28ea0677198a4a8a2bd3eac1f60156 Mon Sep 17 00:00:00 2001 From: Saumya-R Date: Thu, 16 Jul 2026 22:44:46 +0530 Subject: [PATCH 1/9] added test cases for conditional launching Copyright fixes added test case to reject unknown conditions fixed the readme added launch manager processes adding additional test cases for lifecycle removing the changes in pyproject and requirements txt added review comment fixes: - cleared the non-required. - removed the dead LIFECYCLE_TESTS_SUMMARY.md doc reference. - dropped the class-level 13-req partially_verifies blanket claim; each requirement is now tagged on the specific test that actually exercises it (add_test_properties moved onto individual methods) - deleted test_config_defines_startup_retry_policy Adding review comments added the following fixes : - The scenario now actually polls and checks each condition instead of printing it - Replaced plain std::cout text with the same structured JSON log shape the Rust tracing subscriber emits - TestConditionalLaunchingScenario now creates a real flag file, sets a real env var, and spawns a real sleep process - Added TestConditionalLaunchingScenarioTimesOutOnUnmetConditions, a negative test where none of the conditions are ever satisfied - removed test_startup_launches_supervised_apps and test_dependency_gates_rust_startup from test_process_launching_with_daemon.py - TestConditionalLaunchingBlocksOnMissingDependency spins up its own launch_manager with cpp withheld to prove real gating added TestConditionalLaunchingScenarioRejectsUnsupportedPrefix Added timing assertion to TestConditionalLaunchingScenarioRejectsUnsupportedPrefix changed the decorators removed duplicate parametrization added patches/lifecycle/001-forward-visibility-to-config-combiner.patch updated known_goods.json for patch review comments addressed: - Split lifecycle tests into a standalone fit_lifecycle_daemon bazel target with no longer routed by language marker - strengthened test to assert the other app's pid is untouched, proving retry recovery - Added signal_process() helper - Retagged to launch_support updated read me added the issue for persistency build added real retry exhaustion test case added a fix for fit_cpp_orch , added test cases for parallel launching scorebug issue resolved removed the manual marker assuming CI always has resources removed the organizational marker daemon and marker manual as watchdog detection should be part of every CI added daemon invocation review comment address resolving the merge issue --- feature_integration_tests/README.md | 93 +++ feature_integration_tests/configs/BUILD | 4 + .../configs/lifecycle_daemon_config.json | 103 +++ ...fecycle_daemon_parallel_launch_config.json | 100 +++ ...ifecycle_daemon_retry_exhausts_config.json | 81 +++ ...ifecycle_daemon_retry_recovers_config.json | 76 +++ feature_integration_tests/test_cases/BUILD | 171 ++++- .../test_cases/conftest.py | 54 +- .../test_cases/daemon_helpers.py | 619 ++++++++++++++++++ .../test_cases/lifecycle_scenario.py | 50 ++ .../support_apps/flaky_startup_app/BUILD | 25 + .../support_apps/flaky_startup_app/main.cpp | 100 +++ .../test_cases/tests/basic/conftest.py | 33 + .../lifecycle/test_conditional_launching.py | 146 +++++ .../test_conditional_launching_scenario.py | 407 ++++++++++++ .../test_process_launching_with_daemon.py | 558 ++++++++++++++++ .../tests/lifecycle/test_retry_exhaustion.py | 148 +++++ .../cpp/src/internals/log_helpers.h | 135 ++++ .../lifecycle/conditional_launching.cpp | 275 ++++++++ .../lifecycle/conditional_launching.h | 17 + .../test_scenarios/cpp/src/scenarios/mod.cpp | 13 +- .../test_scenarios/rust/src/main.rs | 1 + .../lifecycle/conditional_launching.rs | 158 +++++ .../rust/src/scenarios/lifecycle/mod.rs | 25 + .../test_scenarios/rust/src/scenarios/mod.rs | 9 +- ...orward-visibility-to-config-combiner.patch | 10 + 26 files changed, 3396 insertions(+), 15 deletions(-) create mode 100644 feature_integration_tests/configs/lifecycle_daemon_config.json create mode 100644 feature_integration_tests/configs/lifecycle_daemon_parallel_launch_config.json create mode 100644 feature_integration_tests/configs/lifecycle_daemon_retry_exhausts_config.json create mode 100644 feature_integration_tests/configs/lifecycle_daemon_retry_recovers_config.json create mode 100644 feature_integration_tests/test_cases/daemon_helpers.py create mode 100644 feature_integration_tests/test_cases/lifecycle_scenario.py create mode 100644 feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD create mode 100644 feature_integration_tests/test_cases/support_apps/flaky_startup_app/main.cpp create mode 100644 feature_integration_tests/test_cases/tests/basic/conftest.py create mode 100644 feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py create mode 100644 feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching_scenario.py create mode 100644 feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py create mode 100644 feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py create mode 100644 feature_integration_tests/test_scenarios/cpp/src/internals/log_helpers.h create mode 100644 feature_integration_tests/test_scenarios/cpp/src/scenarios/lifecycle/conditional_launching.cpp create mode 100644 feature_integration_tests/test_scenarios/cpp/src/scenarios/lifecycle/conditional_launching.h create mode 100644 feature_integration_tests/test_scenarios/rust/src/scenarios/lifecycle/conditional_launching.rs create mode 100644 feature_integration_tests/test_scenarios/rust/src/scenarios/lifecycle/mod.rs create mode 100644 patches/lifecycle/001-forward-visibility-to-config-combiner.patch diff --git a/feature_integration_tests/README.md b/feature_integration_tests/README.md index 6f9ced5daf9..88255555f24 100644 --- a/feature_integration_tests/README.md +++ b/feature_integration_tests/README.md @@ -48,6 +48,98 @@ bazel run //feature_integration_tests/test_scenarios/rust:rust_test_scenarios -- bazel test --config=linux-x86_64 //feature_integration_tests/test_cases:fit --test_output=streamed ``` +To run the lifecycle tests directly with `pytest` and build the scenario binaries on demand: + +```sh +python3 -m pytest feature_integration_tests/test_cases/tests/lifecycle/ \ + --build-scenarios \ + -m rust \ + --rust-target-name=//feature_integration_tests/test_scenarios/rust:rust_test_scenarios \ + -q -v + +python3 -m pytest feature_integration_tests/test_cases/tests/lifecycle/ \ + --build-scenarios \ + -m cpp \ + -q -v +``` + +The Rust override is required because plain `--build-scenarios` defaults to +`//feature_integration_tests/test_scenarios/rust:rust_test_scenarios`, while the +lifecycle tests need the reduced lifecycle-only Rust target. + +#### Sandbox uid/gid and scheduling-policy tests + +Some lifecycle daemon tests (`test_launched_process_uid_gid_matches_config_when_applied`, +`test_launched_process_scheduling_matches_config_when_applied`) verify that `launch_manager` +applies the sandbox `uid`/`gid` and scheduling policy from +`feature_integration_tests/configs/lifecycle_daemon_config.json`. This requires granting +`launch_manager` the `cap_setuid,cap_setgid,cap_sys_nice` file capabilities via `setcap`, which +in turn requires `CAP_SETFCAP` — not available to a non-root test runner by default, so these +tests opt in via the `FIT_ENABLE_SETCAP` env var (backed by a passwordless sudoers rule scoped +to the `setcap` binary, e.g. ` ALL=(root) NOPASSWD: /usr/sbin/setcap`, with no trailing +arguments pinned — the target path is a fresh `tmp_path` on every run) and skip otherwise. + +Under `bazel test`, undeclared env vars like `FIT_ENABLE_SETCAP` only reach the test process when +passed via `--test_env` (not `--action_env`, which only affects build actions). Two variants of +the full suite are relevant: + +```sh +# Default: matches CI/CD exactly (sandboxed, no FIT_ENABLE_SETCAP) — the two capability tests skip. +bazel test --config=linux-x86_64 --nocache_test_results //feature_integration_tests/test_cases:fit \ + --test_output=all --test_arg=-rs --test_verbose_timeout_warnings + +# Local verification: also exercises the uid/gid and scheduling-policy grants instead of skipping. +bazel test --config=linux-x86_64 --nocache_test_results //feature_integration_tests/test_cases:fit \ + --spawn_strategy=local --test_env=FIT_ENABLE_SETCAP=1 \ + --test_output=all --test_arg=-rs --test_verbose_timeout_warnings +``` + +Flag rationale (shared by both commands unless noted): + +- `--nocache_test_results`: forces re-execution instead of replaying a cached PASS/SKIP, so a + fresh `setcap` attempt is made every time. +- `--test_output=all`: prints full stdout/stderr for every test, not just failures, so the + `sandbox_privileged_reason` and pytest skip-reason diagnostics are visible. +- `--test_arg=-rs`: forwards pytest's `-rs` flag, which prints the reason for every `SKIPPED` + test instead of just `SKIPPED` with no context. +- `--test_verbose_timeout_warnings`: warns when a test's actual runtime is far from its declared + `timeout`/`size`, useful for right-sizing `fit_lifecycle_daemon`'s `timeout = "long"`. +- `--spawn_strategy=local` (local-verification command only): runs the test action directly on + the host instead of inside Bazel's `linux-sandbox`. The sandbox sets `PR_SET_NO_NEW_PRIVS`, + which makes `setuid` (`sudo`) and file capabilities (`setcap`) inert at exec time even with a + correctly configured host — the grant is applied but silently dropped when the supervised + binary later executes. Only unsandboxed execution lets the grant persist. +- `--test_env=FIT_ENABLE_SETCAP=1` (local-verification command only): opts the test process into + the `sudo -n setcap` attempt; without it these tests always take the plain, non-sudo `setcap` + path and skip on a non-root runner. `bazel run` inherits the shell environment directly, so + `export FIT_ENABLE_SETCAP=1` beforehand is sufficient there instead of `--test_env`. + +#### Tests skipped in CI/CD + +The GitHub Actions runners (`ubuntu-latest`, see `.github/workflows/build_and_test_linux.yml`) run +`bazel test` sandboxed (default `linux-sandbox` strategy) and do not set `FIT_ENABLE_SETCAP` or +provision a passwordless `sudo setcap` rule. As a result, the following subtests in +`fit_lifecycle_daemon` always skip in CI, for both the `rust` and `cpp` supervised-app variants: + +- `test_process_launching_with_daemon.py::TestProcessLaunchingWithDaemon::test_launched_process_uid_gid_matches_config_when_applied[rust|cpp]` +- `test_process_launching_with_daemon.py::TestProcessLaunchingWithDaemon::test_launched_process_scheduling_matches_config_when_applied[rust|cpp]` + +Reason: both depend on `launch_manager` successfully gaining `cap_setuid,cap_setgid,cap_sys_nice` +via `setcap` (see `daemon_helpers._grant_sandbox_capabilities`), which fails in CI for two +independent reasons, either sufficient on its own: + +1. **Sandboxed execution**: `linux-sandbox` sets `PR_SET_NO_NEW_PRIVS`, making any `setuid`/file-capability + escalation inert at exec time, so even a successful `setcap` call has no effect on the process + that actually runs. +2. **No opt-in / no sudoers rule**: `FIT_ENABLE_SETCAP` is not set in the CI workflow, so the tests + never attempt the `sudo -n setcap` path; and the CI runner has no passwordless sudoers entry for + `setcap` regardless. + +This is by design: `_grant_sandbox_capabilities` degrades gracefully (never raises) and the two +capability-dependent subtests self-skip with a diagnostic reason instead of failing the build. All +other subtests in `fit_lifecycle_daemon` only check same-uid process behavior and require no +privilege escalation, so they run and pass normally in CI. + ### ITF Tests (QEMU-based) ITF tests run on a QEMU target and require the `itf-qnx-x86_64` config: @@ -62,6 +154,7 @@ Test scenarios can be listed and run directly for debugging: ```sh bazel run //feature_integration_tests/test_scenarios/rust:rust_test_scenarios -- --list-scenarios +bazel run //feature_integration_tests/test_scenarios/rust:rust_lifecycle_test_scenarios -- --list-scenarios bazel run --config=linux-x86_64 //feature_integration_tests/test_scenarios/cpp:cpp_test_scenarios -- --list-scenarios ``` diff --git a/feature_integration_tests/configs/BUILD b/feature_integration_tests/configs/BUILD index dce9a78284e..b81d0f5bb4f 100644 --- a/feature_integration_tests/configs/BUILD +++ b/feature_integration_tests/configs/BUILD @@ -15,6 +15,10 @@ exports_files( "dlt_config_qnx_x86_64.json", "dlt_config_x86_64.json", "qemu_bridge_config.json", + "lifecycle_daemon_config.json", + "lifecycle_daemon_parallel_launch_config.json", + "lifecycle_daemon_retry_recovers_config.json", + "lifecycle_daemon_retry_exhausts_config.json", ], ) diff --git a/feature_integration_tests/configs/lifecycle_daemon_config.json b/feature_integration_tests/configs/lifecycle_daemon_config.json new file mode 100644 index 00000000000..b8760188bc5 --- /dev/null +++ b/feature_integration_tests/configs/lifecycle_daemon_config.json @@ -0,0 +1,103 @@ +{ + "schema_version": 1, + "defaults": { + "deployment_config": { + "bin_dir": "__FIT_RUNTIME_ROOT__/bin", + "ready_timeout": 2.0, + "shutdown_timeout": 2.0, + "ready_recovery_action": { + "restart": { + "number_of_attempts": 2 + } + }, + "recovery_action": { + "switch_run_target": { + "run_target": "fallback_run_target" + } + }, + "sandbox": { + "uid": 1001, + "gid": 1001, + "scheduling_policy": "SCHED_OTHER", + "scheduling_priority": 0 + } + }, + "component_properties": { + "application_profile": { + "application_type": "Reporting", + "is_self_terminating": false, + "alive_supervision": { + "reporting_cycle": 0.1, + "min_indications": 1, + "max_indications": 3, + "failed_cycles_tolerance": 1 + } + }, + "ready_condition": { + "process_state": "Running" + } + } + }, + "components": { + "cpp_supervised_app": { + "component_properties": { + "binary_name": "cpp_supervised_app", + "application_profile": { + "application_type": "Reporting_And_Supervised" + }, + "process_arguments": [ + "-d50" + ] + }, + "deployment_config": { + "environmental_variables": { + "PROCESSIDENTIFIER": "cpp_supervised_app", + "IDENTIFIER": "cpp_supervised_app" + } + } + }, + "rust_supervised_app": { + "component_properties": { + "binary_name": "rust_supervised_app", + "depends_on": [ + "cpp_supervised_app" + ], + "application_profile": { + "application_type": "Reporting_And_Supervised" + }, + "process_arguments": [ + "-d50" + ] + }, + "deployment_config": { + "environmental_variables": { + "PROCESSIDENTIFIER": "rust_supervised_app", + "IDENTIFIER": "rust_supervised_app" + } + } + } + }, + "run_targets": { + "Startup": { + "depends_on": [ + "cpp_supervised_app", + "rust_supervised_app" + ], + "recovery_action": { + "switch_run_target": { + "run_target": "fallback_run_target" + } + } + } + }, + "initial_run_target": "Startup", + "alive_supervision": { + "evaluation_cycle": 0.05 + }, + "fallback_run_target": { + "depends_on": [ + "cpp_supervised_app", + "rust_supervised_app" + ] + } +} diff --git a/feature_integration_tests/configs/lifecycle_daemon_parallel_launch_config.json b/feature_integration_tests/configs/lifecycle_daemon_parallel_launch_config.json new file mode 100644 index 00000000000..c2aca25e498 --- /dev/null +++ b/feature_integration_tests/configs/lifecycle_daemon_parallel_launch_config.json @@ -0,0 +1,100 @@ +{ + "schema_version": 1, + "defaults": { + "deployment_config": { + "bin_dir": "__FIT_RUNTIME_ROOT__/bin", + "ready_timeout": 2.0, + "shutdown_timeout": 2.0, + "ready_recovery_action": { + "restart": { + "number_of_attempts": 2 + } + }, + "recovery_action": { + "switch_run_target": { + "run_target": "fallback_run_target" + } + }, + "sandbox": { + "uid": 1001, + "gid": 1001, + "scheduling_policy": "SCHED_OTHER", + "scheduling_priority": 0 + } + }, + "component_properties": { + "application_profile": { + "application_type": "Reporting", + "is_self_terminating": false, + "alive_supervision": { + "reporting_cycle": 0.1, + "min_indications": 1, + "max_indications": 3, + "failed_cycles_tolerance": 1 + } + }, + "ready_condition": { + "process_state": "Running" + } + } + }, + "components": { + "cpp_supervised_app": { + "component_properties": { + "binary_name": "cpp_supervised_app", + "application_profile": { + "application_type": "Reporting_And_Supervised" + }, + "process_arguments": [ + "-d50" + ] + }, + "deployment_config": { + "environmental_variables": { + "PROCESSIDENTIFIER": "cpp_supervised_app", + "IDENTIFIER": "cpp_supervised_app" + } + } + }, + "rust_supervised_app": { + "component_properties": { + "binary_name": "rust_supervised_app", + "application_profile": { + "application_type": "Reporting_And_Supervised" + }, + "process_arguments": [ + "-d50" + ] + }, + "deployment_config": { + "environmental_variables": { + "PROCESSIDENTIFIER": "rust_supervised_app", + "IDENTIFIER": "rust_supervised_app" + } + } + } + }, + "run_targets": { + "Startup": { + "depends_on": [ + "cpp_supervised_app", + "rust_supervised_app" + ], + "recovery_action": { + "switch_run_target": { + "run_target": "fallback_run_target" + } + } + } + }, + "initial_run_target": "Startup", + "alive_supervision": { + "evaluation_cycle": 0.05 + }, + "fallback_run_target": { + "depends_on": [ + "cpp_supervised_app", + "rust_supervised_app" + ] + } +} diff --git a/feature_integration_tests/configs/lifecycle_daemon_retry_exhausts_config.json b/feature_integration_tests/configs/lifecycle_daemon_retry_exhausts_config.json new file mode 100644 index 00000000000..e161ed7f234 --- /dev/null +++ b/feature_integration_tests/configs/lifecycle_daemon_retry_exhausts_config.json @@ -0,0 +1,81 @@ +{ + "schema_version": 1, + "defaults": { + "deployment_config": { + "bin_dir": "__FIT_RUNTIME_ROOT__/bin", + "ready_timeout": 2.0, + "shutdown_timeout": 2.0, + "ready_recovery_action": { + "restart": { + "number_of_attempts": 0 + } + }, + "recovery_action": { + "switch_run_target": { + "run_target": "fallback_run_target" + } + }, + "sandbox": { + "uid": 1001, + "gid": 1001, + "scheduling_policy": "SCHED_OTHER", + "scheduling_priority": 0 + } + }, + "component_properties": { + "application_profile": { + "application_type": "Reporting", + "is_self_terminating": false, + "alive_supervision": { + "reporting_cycle": 0.1, + "min_indications": 1, + "max_indications": 3, + "failed_cycles_tolerance": 1 + } + }, + "ready_condition": { + "process_state": "Running" + } + } + }, + "components": { + "flaky_startup_app": { + "component_properties": { + "binary_name": "flaky_startup_app", + "process_arguments": [ + "__FIT_RUNTIME_ROOT__/flaky_startup_app.counter", + "999" + ] + }, + "deployment_config": { + "ready_recovery_action": { + "restart": { + "number_of_attempts": 2 + } + }, + "environmental_variables": { + "PROCESSIDENTIFIER": "flaky_startup_app" + } + } + } + }, + "run_targets": { + "Startup": { + "depends_on": [ + "flaky_startup_app" + ], + "recovery_action": { + "switch_run_target": { + "run_target": "fallback_run_target" + } + } + } + }, + "initial_run_target": "Startup", + "alive_supervision": { + "evaluation_cycle": 0.05 + }, + "fallback_run_target": { + "depends_on": [] + } +} diff --git a/feature_integration_tests/configs/lifecycle_daemon_retry_recovers_config.json b/feature_integration_tests/configs/lifecycle_daemon_retry_recovers_config.json new file mode 100644 index 00000000000..207e84c667d --- /dev/null +++ b/feature_integration_tests/configs/lifecycle_daemon_retry_recovers_config.json @@ -0,0 +1,76 @@ +{ + "schema_version": 1, + "defaults": { + "deployment_config": { + "bin_dir": "__FIT_RUNTIME_ROOT__/bin", + "ready_timeout": 2.0, + "shutdown_timeout": 2.0, + "ready_recovery_action": { + "restart": { + "number_of_attempts": 0 + } + }, + "recovery_action": { + "switch_run_target": { + "run_target": "fallback_run_target" + } + }, + "sandbox": { + "uid": 1001, + "gid": 1001, + "scheduling_policy": "SCHED_OTHER", + "scheduling_priority": 0 + } + }, + "component_properties": { + "application_profile": { + "application_type": "Reporting", + "is_self_terminating": false, + "alive_supervision": { + "reporting_cycle": 0.1, + "min_indications": 1, + "max_indications": 3, + "failed_cycles_tolerance": 1 + } + }, + "ready_condition": { + "process_state": "Running" + } + } + }, + "components": { + "flaky_startup_app": { + "component_properties": { + "binary_name": "flaky_startup_app", + "process_arguments": [ + "__FIT_RUNTIME_ROOT__/flaky_startup_app.counter", + "2" + ] + }, + "deployment_config": { + "ready_recovery_action": { + "restart": { + "number_of_attempts": 2 + } + }, + "environmental_variables": { + "PROCESSIDENTIFIER": "flaky_startup_app" + } + } + } + }, + "run_targets": { + "Startup": { + "depends_on": [ + "flaky_startup_app" + ] + } + }, + "initial_run_target": "Startup", + "alive_supervision": { + "evaluation_cycle": 0.05 + }, + "fallback_run_target": { + "depends_on": [] + } +} diff --git a/feature_integration_tests/test_cases/BUILD b/feature_integration_tests/test_cases/BUILD index 18d0b24212f..66e8c79e091 100644 --- a/feature_integration_tests/test_cases/BUILD +++ b/feature_integration_tests/test_cases/BUILD @@ -37,9 +37,11 @@ compile_pip_requirements( ) # Tests targets + score_py_pytest( - name = "fit_rust", - srcs = glob(["tests/**/*.py"]), + name = "fit_rust_persistency", + timeout = "long", + srcs = glob(["tests/persistency/**/*.py"]), args = [ "-m rust", "--traces=all", @@ -60,8 +62,40 @@ score_py_pytest( ) score_py_pytest( - name = "fit_cpp", - srcs = glob(["tests/**/*.py"]), + name = "fit_rust_scenario_lifecycle", + timeout = "long", + srcs = ["tests/lifecycle/test_conditional_launching_scenario.py"], + args = [ + "-m rust", + "--traces=all", + "--rust-target-path=$(rootpath //feature_integration_tests/test_scenarios/rust:rust_test_scenarios)", + ], + data = [ + "conftest.py", + "fit_scenario.py", + "lifecycle_scenario.py", + "test_properties.py", + "//feature_integration_tests/test_scenarios/rust:rust_test_scenarios", + ], + env = { + "RUST_BACKTRACE": "1", + }, + pytest_config = "//:pyproject.toml", + deps = all_requirements, +) + +test_suite( + name = "fit_rust", + tests = [ + ":fit_rust_persistency", + ":fit_rust_scenario_lifecycle", + ], +) + +score_py_pytest( + name = "fit_cpp_persistency", + timeout = "long", + srcs = glob(["tests/persistency/**/*.py"]), args = [ "-m cpp", "--traces=all", @@ -78,10 +112,139 @@ score_py_pytest( deps = all_requirements, ) +score_py_pytest( + name = "fit_cpp_scenario_lifecycle", + timeout = "long", + srcs = ["tests/lifecycle/test_conditional_launching_scenario.py"], + args = [ + "-m cpp", + "--traces=all", + "--cpp-target-path=$(rootpath //feature_integration_tests/test_scenarios/cpp:cpp_test_scenarios)", + ], + data = [ + "conftest.py", + "fit_scenario.py", + "lifecycle_scenario.py", + "test_properties.py", + "//feature_integration_tests/test_scenarios/cpp:cpp_test_scenarios", + ], + pytest_config = "//:pyproject.toml", + deps = all_requirements, +) + +test_suite( + name = "fit_cpp", + tests = [ + ":fit_cpp_persistency", + ":fit_cpp_scenario_lifecycle", + ], +) + +# Daemon-driven lifecycle tests (test_conditional_launching.py, test_process_launching_with_daemon.py) +# exercise launch_manager against both supervised apps together and don't depend on which +# scenario-binary language variant is under test elsewhere, so they run once here rather than +# being routed - and silently deselected - by the rust/cpp scenario-binary language marker. +score_py_pytest( + name = "fit_lifecycle_daemon", + timeout = "long", + srcs = [ + "tests/lifecycle/test_conditional_launching.py", + "tests/lifecycle/test_process_launching_with_daemon.py", + ], + args = [ + "--traces=all", + ], + data = [ + "conftest.py", + "daemon_helpers.py", + "test_properties.py", + "//feature_integration_tests/configs:lifecycle_daemon_config.json", + "//feature_integration_tests/configs:lifecycle_daemon_parallel_launch_config.json", + "@flatbuffers//:flatc", + "@score_lifecycle_health//examples/control_application:control_daemon", + "@score_lifecycle_health//examples/control_application:lmcontrol", + "@score_lifecycle_health//examples/cpp_supervised_app", + "@score_lifecycle_health//examples/rust_supervised_app", + "@score_lifecycle_health//score/launch_manager", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hm_flatcfg.fbs", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hmcore_flatcfg.fbs", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:lm_flatcfg.fbs", + "@score_lifecycle_health//scripts/config_mapping:lifecycle_config", + ], + env = { + "FIT_CPP_SUPERVISED_APP_PATH": "$(rootpath @score_lifecycle_health//examples/cpp_supervised_app)", + "FIT_LAUNCH_MANAGER_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager)", + "FIT_FLATC_PATH": "$(rootpath @flatbuffers//:flatc)", + "FIT_LIFECYCLE_CONFIG_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json)", + "FIT_LIFECYCLE_CONFIG_TOOL_PATH": "$(rootpath @score_lifecycle_health//scripts/config_mapping:lifecycle_config)", + "FIT_LIFECYCLE_DAEMON_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_config.json)", + "FIT_LIFECYCLE_HMCORE_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hmcore_flatcfg.fbs)", + "FIT_LIFECYCLE_HM_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hm_flatcfg.fbs)", + "FIT_LIFECYCLE_LM_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:lm_flatcfg.fbs)", + "FIT_LIFECYCLE_PARALLEL_LAUNCH_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_parallel_launch_config.json)", + "FIT_RUST_SUPERVISED_APP_PATH": "$(rootpath @score_lifecycle_health//examples/rust_supervised_app)", + "RUST_BACKTRACE": "1", + }, + env_inherit = ["FIT_ENABLE_SETCAP"], + pytest_config = "//:pyproject.toml", + # Runs sandboxed: PR_SET_NO_NEW_PRIVS blocks setuid (sudo) / file-capability (setcap) escalation + # at exec time, so sandbox_privileged is always False here and the uid/gid/scheduling tests + # skip themselves accordingly (see daemon_helpers._grant_sandbox_capabilities). Everything + # else only signals same-uid processes and needs no privilege escalation. FIT_ENABLE_SETCAP=1 + # is for local/manual runs outside the sandbox where the sestcap grant can actually take effect. + deps = all_requirements, +) + +# Dedicated, isolated coverage for ready_recovery_action.restart.number_of_attempts +# (feat_req__lifecycle__retries_configurable). Each daemon invocation generates its own +# single-component config and runtime tree under TEST_TMPDIR. +score_py_pytest( + name = "fit_lifecycle_retries", + timeout = "long", + srcs = [ + "tests/lifecycle/test_retry_exhaustion.py", + ], + args = [ + "--traces=all", + ], + data = [ + "conftest.py", + "daemon_helpers.py", + "test_properties.py", + "//feature_integration_tests/configs:lifecycle_daemon_retry_exhausts_config.json", + "//feature_integration_tests/configs:lifecycle_daemon_retry_recovers_config.json", + "//feature_integration_tests/test_cases/support_apps/flaky_startup_app", + "@flatbuffers//:flatc", + "@score_lifecycle_health//score/launch_manager", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hm_flatcfg.fbs", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hmcore_flatcfg.fbs", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:lm_flatcfg.fbs", + "@score_lifecycle_health//scripts/config_mapping:lifecycle_config", + ], + env = { + "FIT_FLAKY_STARTUP_APP_PATH": "$(rootpath //feature_integration_tests/test_cases/support_apps/flaky_startup_app)", + "FIT_FLATC_PATH": "$(rootpath @flatbuffers//:flatc)", + "FIT_LAUNCH_MANAGER_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager)", + "FIT_LIFECYCLE_CONFIG_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json)", + "FIT_LIFECYCLE_CONFIG_TOOL_PATH": "$(rootpath @score_lifecycle_health//scripts/config_mapping:lifecycle_config)", + "FIT_LIFECYCLE_HMCORE_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hmcore_flatcfg.fbs)", + "FIT_LIFECYCLE_HM_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hm_flatcfg.fbs)", + "FIT_LIFECYCLE_LM_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:lm_flatcfg.fbs)", + "FIT_LIFECYCLE_RETRY_EXHAUSTS_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_retry_exhausts_config.json)", + "FIT_LIFECYCLE_RETRY_RECOVERS_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_retry_recovers_config.json)", + }, + pytest_config = "//:pyproject.toml", + deps = all_requirements, +) + test_suite( name = "fit", tests = [ ":fit_cpp", + ":fit_lifecycle_daemon", + ":fit_lifecycle_retries", ":fit_rust", ], ) diff --git a/feature_integration_tests/test_cases/conftest.py b/feature_integration_tests/test_cases/conftest.py index 662b7210943..36713ae4219 100644 --- a/feature_integration_tests/test_cases/conftest.py +++ b/feature_integration_tests/test_cases/conftest.py @@ -15,6 +15,39 @@ import pytest from testing_utils import BazelTools +try: + # Private API - not guaranteed stable across pytest versions. + from _pytest.mark.expression import Expression +except ImportError: + Expression = None + +_DEFAULT_RUST_TARGET = "//feature_integration_tests/test_scenarios/rust:rust_test_scenarios" + + +def _selected_versions(session: pytest.Session) -> set[str]: + """Return the scenario variants explicitly requested by the mark expression. + + Uses pytest's own marker expression evaluator so that logical operators and + negations are respected. For example, ``-m "not rust"`` must *not* select the + Rust build, while a plain substring check would incorrectly match it. + Falls back to all variants when no expression is given or parsing fails. + """ + mark_expression = session.config.option.markexpr or "" + if not mark_expression or Expression is None: + return {"rust", "cpp"} + try: + expr = Expression.compile(mark_expression) + # pytest 9's MatcherNameAdapter.__call__ forwards **kwargs to the matcher (needed for + # registered markers with args, e.g. `test_properties(x=1)`), so a matcher that only + # accepts `name` raises TypeError for an expression like `"rust and test_properties(x=1)"`. + # Keep evaluate() inside this try so that also falls back to all variants. + selected_versions = { + version for version in ("rust", "cpp") if expr.evaluate(lambda name, **_kwargs: name == version) + } + except Exception: # noqa: BLE001 – malformed expression; fall back to all variants + return {"rust", "cpp"} + return selected_versions or {"rust", "cpp"} + # Cmdline options def pytest_addoption(parser): @@ -31,7 +64,7 @@ def pytest_addoption(parser): parser.addoption( "--rust-target-name", type=str, - default="//feature_integration_tests/test_scenarios/rust:rust_test_scenarios", + default=_DEFAULT_RUST_TARGET, help="Rust test scenario executable target.", ) parser.addoption( @@ -88,18 +121,21 @@ def pytest_sessionstart(session): # Build scenarios. if session.config.getoption("--build-scenarios"): build_timeout = session.config.getoption("--build-scenarios-timeout") + selected_versions = _selected_versions(session) # Build Rust test scenarios. - print("Building Rust test scenarios executable...") - rust_tools = BazelTools(option_prefix="rust", build_timeout=build_timeout) - rust_target_name = session.config.getoption("--rust-target-name") - rust_tools.build(rust_target_name) + if "rust" in selected_versions: + print("Building Rust test scenarios executable...") + rust_tools = BazelTools(option_prefix="rust", build_timeout=build_timeout) + rust_target_name = session.config.getoption("--rust-target-name") + rust_tools.build(rust_target_name) # Build C++ test scenarios. - print("Building C++ test scenarios executable...") - cpp_tools = BazelTools(option_prefix="cpp", build_timeout=build_timeout) - cpp_target_name = session.config.getoption("--cpp-target-name") - cpp_tools.build(cpp_target_name) + if "cpp" in selected_versions: + print("Building C++ test scenarios executable...") + cpp_tools = BazelTools(option_prefix="cpp", build_timeout=build_timeout) + cpp_target_name = session.config.getoption("--cpp-target-name") + cpp_tools.build(cpp_target_name) except Exception as e: pytest.exit(str(e), returncode=1) diff --git a/feature_integration_tests/test_cases/daemon_helpers.py b/feature_integration_tests/test_cases/daemon_helpers.py new file mode 100644 index 00000000000..2f409a51114 --- /dev/null +++ b/feature_integration_tests/test_cases/daemon_helpers.py @@ -0,0 +1,619 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Daemon helpers for lifecycle behavior tests against real Launch Manager.""" + +from __future__ import annotations + +import json +import os +import re +import shutil +import signal +import subprocess +import tempfile +import threading +import time +from dataclasses import dataclass +from pathlib import Path +from typing import Any + +import pytest + + +_TARGET_ENV_MAP = { + "@score_lifecycle_health//score/launch_manager:launch_manager": "FIT_LAUNCH_MANAGER_PATH", + "@score_lifecycle_health//examples/rust_supervised_app:rust_supervised_app": "FIT_RUST_SUPERVISED_APP_PATH", + "@score_lifecycle_health//examples/cpp_supervised_app:cpp_supervised_app": "FIT_CPP_SUPERVISED_APP_PATH", + "//feature_integration_tests/configs:lifecycle_daemon_config.json": "FIT_LIFECYCLE_DAEMON_CONFIG_PATH", + "//feature_integration_tests/configs:lifecycle_daemon_parallel_launch_config.json": ( + "FIT_LIFECYCLE_PARALLEL_LAUNCH_CONFIG_PATH" + ), + "//feature_integration_tests/test_cases/support_apps/flaky_startup_app:flaky_startup_app": ( + "FIT_FLAKY_STARTUP_APP_PATH" + ), + "//feature_integration_tests/configs:lifecycle_daemon_retry_recovers_config.json": ( + "FIT_LIFECYCLE_RETRY_RECOVERS_CONFIG_PATH" + ), + "//feature_integration_tests/configs:lifecycle_daemon_retry_exhausts_config.json": ( + "FIT_LIFECYCLE_RETRY_EXHAUSTS_CONFIG_PATH" + ), + "@score_lifecycle_health//scripts/config_mapping:lifecycle_config": "FIT_LIFECYCLE_CONFIG_TOOL_PATH", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json": "FIT_LIFECYCLE_CONFIG_SCHEMA_PATH", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:lm_flatcfg.fbs": "FIT_LIFECYCLE_LM_SCHEMA_PATH", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hm_flatcfg.fbs": "FIT_LIFECYCLE_HM_SCHEMA_PATH", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hmcore_flatcfg.fbs": "FIT_LIFECYCLE_HMCORE_SCHEMA_PATH", + "@flatbuffers//:flatc": "FIT_FLATC_PATH", +} + + +def _repo_root() -> Path: + return Path(__file__).resolve().parents[2] + + +def _run(cmd: list[str]) -> str: + completed = subprocess.run( + cmd, + cwd=_repo_root(), + capture_output=True, + text=True, + check=True, + ) + return completed.stdout.strip() + + +def _resolve_from_env(target: str) -> Path | None: + """Resolve a target path from Bazel-provided runfile environment variables.""" + env_var = _TARGET_ENV_MAP.get(target) + if env_var is None: + return None + + raw_path = os.environ.get(env_var) + if not raw_path: + return None + + candidate = Path(raw_path) + search_roots = [Path.cwd()] + + test_srcdir = os.environ.get("TEST_SRCDIR") + test_workspace = os.environ.get("TEST_WORKSPACE") + if test_srcdir and test_workspace: + search_roots.append(Path(test_srcdir) / test_workspace) + if test_srcdir: + search_roots.append(Path(test_srcdir)) + + for root in search_roots: + resolved = candidate if candidate.is_absolute() else (root / candidate) + if resolved.exists(): + return resolved.resolve() + + return None + + +def _resolve_target_path(target: str) -> Path: + """Resolve an executable/file path from a bazel target label.""" + env_resolved = _resolve_from_env(target) + if env_resolved is not None: + return env_resolved + + _run(["bazel", "build", target]) + output = _run(["bazel", "cquery", "--output=files", target]) + candidates = [line.strip() for line in output.splitlines() if line.strip()] + if not candidates: + raise RuntimeError(f"No files produced by target: {target}") + + execution_root = Path(_run(["bazel", "info", "execution_root"])) + for item in candidates: + candidate = Path(item) + if not candidate.is_absolute(): + candidate = execution_root / candidate + if candidate.exists(): + return candidate + + raise RuntimeError(f"No existing artifact found for target: {target}. Candidates: {candidates!r}") + + +def get_binary_path(target: str) -> Path: + """Compatibility helper used by daemon tests for bazel labels.""" + return _resolve_target_path(target) + + +def pgrep_cmdline_pattern(binary_path: str) -> str: + """Build POSIX ERE pattern matching binary with optional arguments.""" + return rf"^{re.escape(binary_path)}([[:space:]]|$)" + + +def is_running(binary_path: str | Path) -> bool: + result = subprocess.run( + ["pgrep", "-f", pgrep_cmdline_pattern(str(binary_path))], + capture_output=True, + text=True, + check=False, + ) + return result.returncode == 0 + + +def first_pid(binary_path: str | Path) -> str | None: + result = subprocess.run( + ["pgrep", "-f", pgrep_cmdline_pattern(str(binary_path))], + capture_output=True, + text=True, + check=False, + ) + if result.returncode != 0: + return None + lines = [line for line in result.stdout.splitlines() if line] + return lines[0] if lines else None + + +def wait_until(predicate, timeout_s: float, interval_s: float = 0.2) -> bool: + deadline = time.time() + timeout_s + while time.time() < deadline: + if predicate(): + return True + time.sleep(interval_s) + return False + + +_SETCAP_CAPS = "cap_setuid,cap_setgid,cap_sys_nice+ep" + + +def _mount_nosuid(path: Path) -> bool: + """Best-effort check whether `path` lives on a filesystem mounted `nosuid`. + + A `nosuid` mount silently strips file capabilities at exec time even when `setcap` + itself reports success, which otherwise looks identical to "grant never happened" + from the caller's point of view. + """ + try: + findmnt = shutil.which("findmnt") + if findmnt is None: + return False + result = subprocess.run( + [findmnt, "-n", "-o", "OPTIONS", "-T", str(path)], + capture_output=True, + text=True, + check=False, + ) + return result.returncode == 0 and "nosuid" in result.stdout + except OSError: + return False + + +def _grant_sandbox_capabilities(binary_path: Path) -> tuple[bool, str]: + """Best-effort grant of the capabilities launch_manager needs to apply sandbox uid/gid + and scheduling policy without running as root. Returns `(granted, reason)`: `granted` + is a *verified* result (re-read via `getcap`, not just the setcap exit code) so tests + can key off a real, established precondition instead of assuming root; `reason` is a + human-readable diagnostic that is safe to surface directly in a pytest.skip() message. + + Requires CAP_SETFCAP to write the capability xattr, which a non-root test runner does not + have by default. Set FIT_ENABLE_SETCAP=1 to opt into a `sudo -n setcap` attempt, backed by + a passwordless sudoers rule scoped to the setcap binary (e.g. ` ALL=(root) NOPASSWD: + /usr/sbin/setcap`, with NO trailing arguments pinned — the target path is a fresh tmp_path + on every test run, so a rule that also pins the argument list will never match). Without + the flag, only a plain (non-sudo) setcap is tried, which only succeeds if the runner is + already root. + + Under `bazel test`, undeclared env vars (like FIT_ENABLE_SETCAP) do not reach the test + process unless passed via `--test_env=FIT_ENABLE_SETCAP=1` (NOT `--action_env`, which only + affects build actions). `bazel run` inherits the invoking shell's environment directly, so + `--action_env` is a no-op for this variable there; it is only needed to force a rebuild + when it affects action inputs, which it does not here. + """ + if shutil.which("setcap") is None: + return False, "setcap binary not found on PATH" + + setcap_enabled = os.environ.get("FIT_ENABLE_SETCAP") == "1" + attempts: list[tuple[list[str], str]] = [ + (["setcap", _SETCAP_CAPS, str(binary_path)], "plain setcap (requires running as root)") + ] + if setcap_enabled: + if shutil.which("sudo") is None: + attempts.append(([], "FIT_ENABLE_SETCAP=1 set but 'sudo' not found on PATH")) + else: + attempts.insert( + 0, + (["sudo", "-n", "setcap", _SETCAP_CAPS, str(binary_path)], "sudo -n setcap"), + ) + else: + attempts.append(([], "FIT_ENABLE_SETCAP not set to '1'; skipping sudo setcap attempt")) + + failures: list[str] = [] + for cmd, label in attempts: + if not cmd: + failures.append(label) + continue + result = subprocess.run(cmd, capture_output=True, text=True, check=False) + if result.returncode != 0: + failures.append( + f"{label} failed (rc={result.returncode}): " + f"{result.stderr.strip() or result.stdout.strip() or ''}" + ) + continue + + # setcap can report success while the kernel still drops the capability at exec + # time (e.g. the binary lives on a filesystem mounted `nosuid`). Verify by reading + # the xattr back instead of trusting the exit code. + getcap = shutil.which("getcap") + if getcap is not None: + verify = subprocess.run([getcap, str(binary_path)], capture_output=True, text=True, check=False) + if "cap_setuid" not in verify.stdout or "cap_setgid" not in verify.stdout: + nosuid_hint = " (path is on a 'nosuid' mount)" if _mount_nosuid(binary_path) else "" + failures.append( + f"{label} reported success but getcap did not confirm the capabilities" + f"{nosuid_hint}: {verify.stdout.strip() or ''}" + ) + continue + + return True, f"granted via {label}" + + return False, "; ".join(failures) if failures else "no grant attempt produced a result" + + +def signal_process(pid: str, sig: str, *, sandbox_privileged: bool) -> tuple[bool, str]: + """Send `sig` (e.g. "-9", "-STOP", "-CONT") to `pid`, escalating via sudo if needed. + + Under sandbox capabilities, supervised apps run as the configured sandbox uid/gid, + not the runner's own uid, so a plain `kill` fails. Falls back to `sudo -n kill` when + `FIT_ENABLE_SETCAP=1` (same sudoers scope as `_grant_sandbox_capabilities`). + """ + attempts: list[list[str]] = [["kill", sig, pid]] + if sandbox_privileged and os.environ.get("FIT_ENABLE_SETCAP") == "1" and shutil.which("sudo") is not None: + attempts.append(["sudo", "-n", "kill", sig, pid]) + + failures: list[str] = [] + for cmd in attempts: + result = subprocess.run(cmd, capture_output=True, text=True, check=False) + if result.returncode == 0: + return True, f"sent via {' '.join(cmd)}" + failures.append(f"{' '.join(cmd)} failed (rc={result.returncode}): {result.stderr.strip() or ''}") + + return False, "; ".join(failures) + + +def _wait_for_apps(apps: dict[str, Path], timeout_s: float = 8.0, interval_s: float = 0.2) -> bool: + return wait_until(lambda: all(is_running(path) for path in apps.values()), timeout_s, interval_s) + + +def _tmpdir_root() -> Path: + """Return the writable temp root for the current test invocation. + + Bazel sets `TEST_TMPDIR` to a fresh directory per test. Outside Bazel, fall + back to the system temp dir; callers create unique children in either case. + """ + value = os.environ.get("TEST_TMPDIR") + if value: + return Path(value) + return Path(tempfile.gettempdir()) + + +@dataclass +class ManagedDaemon: + """A subprocess wrapper with line-buffered output collection.""" + + process: subprocess.Popen[str] + _lines: list[str] + _thread: threading.Thread + + def is_running(self) -> bool: + return self.process.poll() is None + + def pid(self) -> int: + return self.process.pid + + def stop(self) -> None: + if self.is_running(): + os.killpg(os.getpgid(self.process.pid), signal.SIGTERM) + deadline = time.time() + 5.0 + while self.is_running() and time.time() < deadline: + time.sleep(0.1) + if self.is_running(): + os.killpg(os.getpgid(self.process.pid), signal.SIGKILL) + self.process.wait(timeout=5) + self._thread.join(timeout=1) + + def get_logs(self) -> str: + return "\n".join(self._lines) + + +def _cleanup_runtime_root(runtime_root: Path) -> None: + """Remove a daemon's uniquely allocated runtime directory.""" + shutil.rmtree(runtime_root, ignore_errors=True) + + +def _generate_runtime_config(config_template: str, runtime_root: Path, etc_dir: Path) -> None: + """Render and serialize an isolated launch-manager config for one daemon.""" + config = json.loads(_resolve_target_path(config_template).read_text(encoding="utf-8")) + config["defaults"]["deployment_config"]["bin_dir"] = str(runtime_root / "bin") + + for component in config["components"].values(): + arguments = component["component_properties"].get("process_arguments", []) + component["component_properties"]["process_arguments"] = [ + str(runtime_root / "flaky_startup_app.counter") + if argument == "__FIT_RUNTIME_ROOT__/flaky_startup_app.counter" + else argument + for argument in arguments + ] + + rendered_config = etc_dir / "lifecycle_config.json" + rendered_config.write_text(json.dumps(config), encoding="utf-8") + generated_dir = etc_dir / "generated" + generated_dir.mkdir() + config_tool = _resolve_target_path("@score_lifecycle_health//scripts/config_mapping:lifecycle_config") + config_schema = _resolve_target_path( + "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json" + ) + subprocess.run( + [str(config_tool), str(rendered_config), "--schema", str(config_schema), "-o", str(generated_dir)], + capture_output=True, + text=True, + check=True, + ) + + flatc = _resolve_target_path("@flatbuffers//:flatc") + buffers = ( + ( + "lm_demo", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:lm_flatcfg.fbs", + ), + ("hm_demo", "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hm_flatcfg.fbs"), + ( + "hmcore", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hmcore_flatcfg.fbs", + ), + ) + for name, schema_target in buffers: + subprocess.run( + [ + str(flatc), + "--binary", + "--strict-json", + "-o", + str(etc_dir), + str(_resolve_target_path(schema_target)), + str(generated_dir / f"{name}.json"), + ], + capture_output=True, + text=True, + check=True, + ) + + +def start_launch_manager_daemon( + tmp_path_factory: pytest.TempPathFactory, + blocked_apps: frozenset[str] = frozenset(), + wait_for_apps: bool = True, + config_template: str = "//feature_integration_tests/configs:lifecycle_daemon_config.json", +) -> dict[str, Any]: + """Start a real launch_manager process with generated flatbuffer config. + + `blocked_apps` names ("rust"/"cpp") are copied into place but left + non-executable, so launch_manager cannot start them until the caller + chmod's them back to 0o755. Used to exercise the dependency-gating + negative path: assert the dependent app stays down while its + dependency is withheld, then unblock and assert it starts - and, with + an independent config (no depends_on between the two apps), the inverse: + assert the other app starts anyway, proving it isn't gated at all. + + Each invocation receives its own directory beneath `TEST_TMPDIR`, so it can + run concurrently with the class-scoped fixture or another Bazel test process. + """ + + runtime_root = Path(tempfile.mkdtemp(prefix="lifecycle_fit-", dir=_tmpdir_root())) + try: + work_dir = tmp_path_factory.mktemp("lm-daemon") + etc_dir = work_dir / "etc" + etc_dir.mkdir(parents=True, exist_ok=True) + + bin_dir = runtime_root / "bin" + bin_dir.mkdir(parents=True, exist_ok=True) + + launch_manager = _resolve_target_path("@score_lifecycle_health//score/launch_manager:launch_manager") + rust_supervised = _resolve_target_path( + "@score_lifecycle_health//examples/rust_supervised_app:rust_supervised_app" + ) + cpp_supervised = _resolve_target_path("@score_lifecycle_health//examples/cpp_supervised_app:cpp_supervised_app") + + lm_dst = work_dir / "launch_manager" + shutil.copy2(launch_manager, lm_dst) + lm_dst.chmod(0o755) + sandbox_privileged, sandbox_privileged_reason = _grant_sandbox_capabilities(lm_dst) + + for key, src in (("rust", rust_supervised), ("cpp", cpp_supervised)): + dst = bin_dir / src.name + shutil.copy2(src, dst) + dst.chmod(0o000 if key in blocked_apps else 0o755) + + _generate_runtime_config(config_template, runtime_root, etc_dir) + + env = os.environ.copy() + env.setdefault("ECUCFG_ENV_VAR_ROOTFOLDER", str(etc_dir)) + + lines: list[str] = [] + process = subprocess.Popen( + [str(lm_dst)], + cwd=work_dir, + env=env, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + text=True, + start_new_session=True, + ) + + def _collect_output() -> None: + assert process.stdout is not None + for line in process.stdout: + line = line.rstrip("\n") + if line: + lines.append(line) + + thread = threading.Thread(target=_collect_output, daemon=True) + thread.start() + + daemon = ManagedDaemon(process=process, _lines=lines, _thread=thread) + + # Give startup a chance to complete and fail early if config is broken. + time.sleep(1.0) + if not daemon.is_running(): + logs = daemon.get_logs() + pytest.skip(f"launch_manager failed to start in this environment. Logs:\n{logs}") + + apps = { + "rust": bin_dir / "rust_supervised_app", + "cpp": bin_dir / "cpp_supervised_app", + } + if wait_for_apps and not _wait_for_apps({k: v for k, v in apps.items() if k not in blocked_apps}): + process_snapshot = _run(["ps", "-eo", "pid,args"]) + daemon.stop() + _cleanup_runtime_root(runtime_root) + pytest.fail( + "Launch Manager did not bring supervised apps to running state within timeout.\n" + f"Expected apps: {apps}\n" + f"Daemon logs:\n{daemon.get_logs()}\n" + f"Process snapshot:\n{process_snapshot}" + ) + except BaseException: + _cleanup_runtime_root(runtime_root) + raise + + return { + "daemon": daemon, + "work_dir": work_dir, + "bin_dir": bin_dir, + "apps": apps, + "sandbox_privileged": sandbox_privileged, + "sandbox_privileged_reason": sandbox_privileged_reason, + "runtime_root": runtime_root, + } + + +def start_flaky_retry_daemon( + tmp_path_factory: pytest.TempPathFactory, + config_template: str, + crashes_before_success: int, +) -> dict[str, Any]: + """Start launch_manager against a single-component retry config. + + Drives `flaky_startup_app` (see support_apps/flaky_startup_app/main.cpp), which + aborts on its first `crashes_before_success` startup attempts and stays running + from then on, so `ready_recovery_action.restart.number_of_attempts` can be + exercised deterministically instead of relying on a real, racy startup failure. + Does not wait for the app to reach Running: whether it ever does is exactly + what the calling test is checking. + """ + runtime_root = Path(tempfile.mkdtemp(prefix="lifecycle_fit_retries-", dir=_tmpdir_root())) + try: + work_dir = tmp_path_factory.mktemp("lm-retry-daemon") + etc_dir = work_dir / "etc" + etc_dir.mkdir(parents=True, exist_ok=True) + + bin_dir = runtime_root / "bin" + bin_dir.mkdir(parents=True, exist_ok=True) + + launch_manager = _resolve_target_path("@score_lifecycle_health//score/launch_manager:launch_manager") + flaky_app = _resolve_target_path( + "//feature_integration_tests/test_cases/support_apps/flaky_startup_app:flaky_startup_app" + ) + lm_dst = work_dir / "launch_manager" + shutil.copy2(launch_manager, lm_dst) + lm_dst.chmod(0o755) + + app_dst = bin_dir / "flaky_startup_app" + shutil.copy2(flaky_app, app_dst) + app_dst.chmod(0o755) + + counter_path = runtime_root / "flaky_startup_app.counter" + if counter_path.exists(): + counter_path.unlink() + + _generate_runtime_config(config_template, runtime_root, etc_dir) + + env = os.environ.copy() + env.setdefault("ECUCFG_ENV_VAR_ROOTFOLDER", str(etc_dir)) + + lines: list[str] = [] + process = subprocess.Popen( + [str(lm_dst)], + cwd=work_dir, + env=env, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + text=True, + start_new_session=True, + ) + + def _collect_output() -> None: + assert process.stdout is not None + for line in process.stdout: + line = line.rstrip("\n") + if line: + lines.append(line) + + thread = threading.Thread(target=_collect_output, daemon=True) + thread.start() + + daemon = ManagedDaemon(process=process, _lines=lines, _thread=thread) + + time.sleep(1.0) + if not daemon.is_running(): + logs = daemon.get_logs() + pytest.skip(f"launch_manager failed to start in this environment. Logs:\n{logs}") + except BaseException: + _cleanup_runtime_root(runtime_root) + raise + + return { + "daemon": daemon, + "work_dir": work_dir, + "bin_dir": bin_dir, + "app_path": app_dst, + "counter_path": counter_path, + "crashes_before_success": crashes_before_success, + "runtime_root": runtime_root, + } + + +def stop_flaky_retry_daemon(daemon_info: dict[str, Any]) -> None: + """Tear down a daemon started by `start_flaky_retry_daemon`.""" + daemon_info["daemon"].stop() + subprocess.run( + ["pkill", "-f", pgrep_cmdline_pattern(str(daemon_info["app_path"]))], + capture_output=True, + text=True, + check=False, + ) + _cleanup_runtime_root(daemon_info["runtime_root"]) + + +def read_retry_attempt_count(counter_path: Path) -> int: + """Read flaky_startup_app's persisted attempt counter; 0 if it hasn't run yet.""" + try: + return int(counter_path.read_text().strip()) + except (FileNotFoundError, ValueError): + return 0 + + +def stop_launch_manager_daemon(daemon_info: dict[str, Any]) -> None: + """Tear down a daemon started by `start_launch_manager_daemon`.""" + daemon_info["daemon"].stop() + _cleanup_runtime_root(daemon_info["runtime_root"]) + + +@pytest.fixture(scope="class") +def launch_manager_daemon(tmp_path_factory: pytest.TempPathFactory) -> dict[str, Any]: + """Start a real launch_manager process with generated flatbuffer config.""" + daemon_info = start_launch_manager_daemon(tmp_path_factory) + try: + yield daemon_info + finally: + stop_launch_manager_daemon(daemon_info) diff --git a/feature_integration_tests/test_cases/lifecycle_scenario.py b/feature_integration_tests/test_cases/lifecycle_scenario.py new file mode 100644 index 00000000000..29753b0efc5 --- /dev/null +++ b/feature_integration_tests/test_cases/lifecycle_scenario.py @@ -0,0 +1,50 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +""" +Helpers and base scenario class for lifecycle feature integration tests. + +``LifecycleScenario`` is a ``FitScenario`` subclass that supplies the shared +``temp_dir`` fixture so individual test classes do not have to duplicate it. +""" + +from collections.abc import Generator +from pathlib import Path + +import pytest +from fit_scenario import FitScenario, temp_dir_common + + +class LifecycleScenario(FitScenario): + """ + Base class for lifecycle feature integration tests. + + Provides the ``temp_dir`` fixture shared by all lifecycle test classes. + """ + + @pytest.fixture(scope="class") + def temp_dir( + self, + tmp_path_factory: pytest.TempPathFactory, + version: str, + ) -> Generator[Path, None, None]: + """ + Provide a temporary working directory for the lifecycle tests. + + Parameters + ---------- + tmp_path_factory : pytest.TempPathFactory + Built-in pytest factory for temporary directories. + version : str + Parametrized scenario version (``"rust"`` or ``"cpp"``). + """ + yield from temp_dir_common(tmp_path_factory, self.__class__.__name__, version) diff --git a/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD b/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD new file mode 100644 index 00000000000..e887c051c65 --- /dev/null +++ b/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD @@ -0,0 +1,25 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +load("@rules_cc//cc:defs.bzl", "cc_binary") + +# Plain "Native" supervised app (no launch_manager lifecycle API integration) used +# only to deterministically drive ready_recovery_action.restart in +# lifecycle_daemon_retries_config.json. See main.cpp for behavior. +cc_binary( + name = "flaky_startup_app", + srcs = ["main.cpp"], + visibility = ["//feature_integration_tests:__subpackages__"], + deps = [ + "@score_lifecycle_health//score/launch_manager:lifecycle_cc", + ], +) diff --git a/feature_integration_tests/test_cases/support_apps/flaky_startup_app/main.cpp b/feature_integration_tests/test_cases/support_apps/flaky_startup_app/main.cpp new file mode 100644 index 00000000000..f20e6b0a6e0 --- /dev/null +++ b/feature_integration_tests/test_cases/support_apps/flaky_startup_app/main.cpp @@ -0,0 +1,100 @@ +/******************************************************************************** + * Copyright (c) 2026 Contributors to the Eclipse Foundation + * + * See the NOTICE file(s) distributed with this work for additional + * information regarding copyright ownership. + * + * This program and the accompanying materials are made available under the + * terms of the Apache License Version 2.0 which is available at + * https://www.apache.org/licenses/LICENSE-2.0 + * + * SPDX-License-Identifier: Apache-2.0 + ********************************************************************************/ + +// A "Reporting" launch_manager component that deterministically aborts on its +// first N startup attempts, then reports Running from attempt N+1 onward. The +// attempt count is persisted in a counter file so it survives across the +// process restarts that launch_manager performs in place, letting FITs +// exercise `ready_recovery_action.restart.number_of_attempts` (retry, and +// retry exhaustion) without relying on a real, racy startup failure. +// +// Must call report_running(): launch_manager's restart-in-place accounting is +// driven by the control-client channel that report_running() sets up. A plain +// "Native" process (no lifecycle API integration) never establishes that +// channel, so its startup failures go straight to the process group's +// recovery_action instead of being retried per `number_of_attempts`. + +#include +#include +#include +#include +#include +#include +#include +#include +#include + +#include + +namespace { + +std::atomic exit_requested{false}; + +void signal_handler(int signal) +{ + if (signal == SIGINT || signal == SIGTERM) + { + exit_requested = true; + } +} + +int read_attempt_count(const std::filesystem::path& counter_path) +{ + std::ifstream in(counter_path); + int count = 0; + if (in) + { + in >> count; + } + return count; +} + +void write_attempt_count(const std::filesystem::path& counter_path, int count) +{ + std::ofstream out(counter_path, std::ios::trunc); + out << count; +} + +} // namespace + +int main(int argc, char** argv) +{ + if (argc < 3) + { + std::cerr << "usage: flaky_startup_app " << std::endl; + return EXIT_FAILURE; + } + + const std::filesystem::path counter_path{argv[1]}; + const int crashes_before_success = std::atoi(argv[2]); + + const int attempt = read_attempt_count(counter_path); + write_attempt_count(counter_path, attempt + 1); + + if (attempt < crashes_before_success) + { + std::cerr << "flaky_startup_app: simulating crash on attempt " << attempt << std::endl; + std::abort(); + } + + std::cerr << "flaky_startup_app: starting successfully on attempt " << attempt << std::endl; + score::mw::lifecycle::report_running(); + + signal(SIGINT, signal_handler); + signal(SIGTERM, signal_handler); + while (!exit_requested) + { + std::this_thread::sleep_for(std::chrono::milliseconds(50)); + } + return EXIT_SUCCESS; +} diff --git a/feature_integration_tests/test_cases/tests/basic/conftest.py b/feature_integration_tests/test_cases/tests/basic/conftest.py new file mode 100644 index 00000000000..da0b01b28c9 --- /dev/null +++ b/feature_integration_tests/test_cases/tests/basic/conftest.py @@ -0,0 +1,33 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +import pytest + + +def pytest_sessionfinish(session: pytest.Session, exitstatus: int) -> None: + """Tolerate an empty ``-m cpp`` selection under tests/basic/ instead of failing the build. + + fit_cpp_orch runs ``-m cpp`` here, but every test under tests/basic/ is currently + @pytest.mark.rust-only, so today's selection is legitimately empty. Without this, + pytest's NO_TESTS_COLLECTED exit code would fail fit_cpp_orch permanently until a cpp + test exists. Once a cpp-marked test is added under tests/basic/, testscollected > 0 + and this hook no longer applies - the target then runs (and can fail) normally. + """ + if exitstatus == pytest.ExitCode.NO_TESTS_COLLECTED and session.testscollected == 0: + markexpr = session.config.getoption("markexpr", "") + if "cpp" in markexpr: + print( + "fit_cpp_orch: no @pytest.mark.cpp tests exist under tests/basic/ yet - " + "treating the empty selection as a pass. Add one and this target will " + "start actually running it." + ) + session.exitstatus = pytest.ExitCode.OK diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py b/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py new file mode 100644 index 00000000000..85cb8898f05 --- /dev/null +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py @@ -0,0 +1,146 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +""" +Feature integration tests for conditional launching against a real Launch Manager. + +Unlike scenario-stub checks, these tests validate behavior from an actual +launch_manager process started with lifecycle daemon configuration. +""" + +import json +from pathlib import Path +from typing import Any + +import pytest +from daemon_helpers import ( + is_running, + launch_manager_daemon, + start_launch_manager_daemon, + stop_launch_manager_daemon, + wait_until, +) +from test_properties import add_test_properties + + +@pytest.mark.parametrize("version", ["rust", "cpp"], scope="class") +class TestConditionalLaunchingWithDaemon: + """Verify dependency-based conditional launching with real daemon behavior.""" + + @add_test_properties( + partially_verifies=["feat_req__lifecycle__launch_support"], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + def test_startup_launches_conditioned_processes(self, launch_manager_daemon: dict[str, Any], version: str) -> None: + """Verify supervised processes are launched as part of conditional startup.""" + daemon_info = launch_manager_daemon + app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" + app_path = str(daemon_info["apps"][version]) + + started = wait_until(lambda: is_running(app_path), timeout_s=8.0) + assert started, f"{app_name} was not launched in conditional startup" + + +class TestConditionalLaunchingDependencyOrdering: + """Verify cpp-before-rust ordering is declared in config. + + Not parametrized on `version`: inspects the static config only, independent of + the scenario variant under test elsewhere. + + Real ordering/gating evidence lives in + TestConditionalLaunchingBlocksOnMissingDependency below - a start-tick + comparison used to live here but was near-vacuous (both processes launch + within the same ~10ms tick regardless of ordering) and was removed. + """ + + def test_dependency_is_declared_in_lifecycle_config(self) -> None: + """Verify runtime configuration defines rust conditional dependency on cpp.""" + config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" + config = json.loads(config_path.read_text(encoding="utf-8")) + + rust_component = config["components"]["rust_supervised_app"]["component_properties"] + depends_on = rust_component.get("depends_on", []) + assert "cpp_supervised_app" in depends_on, ( + "Expected rust_supervised_app to depend on cpp_supervised_app in lifecycle daemon config" + ) + + +class TestConditionalLaunchingBlocksOnMissingDependency: + """Verify rust startup is actually gated on cpp, not merely correlated with it. + + Runs its own launch_manager instance (rather than the shared class-scoped + `launch_manager_daemon` fixture) with cpp_supervised_app withheld, so it can + observe the negative case: rust must not start while its dependency cannot. + It uses a unique runtime root and generated configuration beneath + `TEST_TMPDIR`, so it is independent of other lifecycle daemon instances. + + Not parametrized on `version`: dependency gating is independent of which + scenario variant is under test elsewhere, so this runs exactly once. + """ + + @add_test_properties( + partially_verifies=[ + "feat_req__lifecycle__waitfor_support", + "feat_req__lifecycle__dependency_check", + "feat_req__lifecycle__cond_process_start", + "feat_req__lifecycle__process_ordering", + "feat_req__lifecycle__define_swc_dependencies", + ], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + def test_rust_stays_down_until_cpp_dependency_becomes_available( + self, tmp_path_factory: pytest.TempPathFactory + ) -> None: + """Verify rust does not start while cpp is withheld, and does once cpp is unblocked.""" + daemon_info = start_launch_manager_daemon( + tmp_path_factory, + blocked_apps=frozenset({"cpp"}), + wait_for_apps=False, + ) + try: + cpp_path = daemon_info["apps"]["cpp"] + rust_path = str(daemon_info["apps"]["rust"]) + + # cpp cannot execute (mode 0o000): rust must not appear while it is withheld. + rust_started_early = wait_until(lambda: is_running(rust_path), timeout_s=4.0) + assert not rust_started_early, ( + "rust_supervised_app started even though its cpp_supervised_app dependency " + "was withheld (non-executable); dependency gating was not enforced" + ) + + # Repeated, path-specific launch failures demonstrate the daemon keeps + # processing the unavailable dependency rather than aborting once. + cpp_launch_failure = f"File does not exist or is not executable: {cpp_path}" + failures_observed = wait_until( + lambda: daemon_info["daemon"].get_logs().count(cpp_launch_failure) >= 2, + 4.0, + ) + assert failures_observed, ( + "Expected repeated cpp launch failures while it was withheld; matching daemon logs:\n" + + "\n".join( + line for line in daemon_info["daemon"].get_logs().splitlines() if cpp_launch_failure in line + ) + ) + + # Once cpp becomes executable and reaches Running, its dependent rust app + # should be released as well. + cpp_path.chmod(0o755) + cpp_started = wait_until(lambda: is_running(cpp_path), timeout_s=8.0) + assert cpp_started, "cpp_supervised_app did not start after becoming executable" + rust_started = wait_until(lambda: is_running(rust_path), timeout_s=8.0) + assert rust_started, ( + "rust_supervised_app did not start after its cpp_supervised_app dependency became available" + ) + finally: + stop_launch_manager_daemon(daemon_info) diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching_scenario.py b/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching_scenario.py new file mode 100644 index 00000000000..6f9343213f3 --- /dev/null +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching_scenario.py @@ -0,0 +1,407 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Scenario-level smoke tests for the conditional-launching test-scenario binary. + +These exercise the bespoke wait-condition poller in test_scenarios/{rust,cpp}/.../lifecycle/ +conditional_launching.{rs,cpp} directly: preconditions (a path, an env var, a running process) +are really established or really withheld, so the assertions verify that *this stub* observes +and enforces them, not merely that it echoes back what was configured. + +This is a fact about the FIT's own test code, not about launch_manager - no test here starts +or drives an actual launch_manager instance, so none of them verify a `feat_req__lifecycle__*` +requirement of the lifecycle module. That verification belongs to the daemon-driven tests in +test_conditional_launching.py / test_process_launching_with_daemon.py, or a future test that +exercises this same wait-condition logic through launch_manager's real config. Hence no add +`@add_test_properties(partially_verifies=[...])` claims have been added to classes in this file. +""" + +import os +import subprocess +import sys +import time +from collections.abc import Generator +from pathlib import Path +from typing import Any + +import pytest +from fit_scenario import ResultCode +from lifecycle_scenario import LifecycleScenario +from testing_utils import ScenarioResult + +pytestmark = [pytest.mark.parametrize("version", ["rust", "cpp"], scope="class")] + +_CONDITION_ENV_VAR = "LM_CONDITION_READY" +_CONDITION_PROCESS_NAME = "sleep" + + +class TestConditionalLaunchingScenario(LifecycleScenario): + """Verify the scenario actually waits for and detects satisfied conditions.""" + + @pytest.fixture(scope="class") + def scenario_name(self) -> str: + return "lifecycle.conditional_launching" + + @pytest.fixture(scope="class") + def flag_path(self, temp_dir: Path) -> Path: + return temp_dir / "lifecycle_launch_ready.flag" + + @pytest.fixture(scope="class", autouse=True) + def satisfied_preconditions(self, flag_path: Path) -> Generator[None, None, None]: + """Really establish the preconditions the scenario is told to wait for. + + The flag file is created up front (path condition already met), the env var is + set in this process (inherited by the scenario subprocess), and a real `sleep` + process is kept alive for the duration of the scenario run (process condition). + Torn down afterwards so this class does not leak state into later tests. + """ + flag_path.write_text("ready", encoding="utf-8") + os.environ[_CONDITION_ENV_VAR] = "1" + process = subprocess.Popen([_CONDITION_PROCESS_NAME, "30"]) + try: + yield + finally: + process.kill() + process.wait() + del os.environ[_CONDITION_ENV_VAR] + flag_path.unlink(missing_ok=True) + + @pytest.fixture(scope="class") + def test_config(self, flag_path: Path, satisfied_preconditions: None) -> dict[str, Any]: + # Depends on `satisfied_preconditions` explicitly (rather than relying on autouse + # ordering) so preconditions are guaranteed established before `results` executes. + return { + "test": { + "wait_conditions": [ + f"path:{flag_path}", + f"env:{_CONDITION_ENV_VAR}", + f"process:{_CONDITION_PROCESS_NAME}", + ], + "polling_interval_ms": 50, + "timeout_ms": 2000, + }, + } + + def test_conditions_already_satisfied_allow_immediate_success( + self, + results: ScenarioResult, + version: str, + ) -> None: + """Verify the scenario succeeds once path/env/process conditions are all really met.""" + assert results.return_code == ResultCode.SUCCESS, ( + f"Expected success with satisfied preconditions, got: {results}" + ) + + def test_each_condition_is_individually_confirmed_satisfied( + self, + logs_info_level: Any, + flag_path: Path, + version: str, + ) -> None: + """Verify the scenario reports each condition as satisfied, not just configured.""" + expected_messages = [ + f"Condition satisfied: path:{flag_path}", + f"Condition satisfied: env:{_CONDITION_ENV_VAR}", + f"Condition satisfied: process:{_CONDITION_PROCESS_NAME}", + "All dependencies satisfied", + ] + for expected in expected_messages: + log = logs_info_level.find_log("message", value=expected) + assert log is not None, f"Expected scenario to log: {expected}" + + def test_timeout_and_polling_interval_are_honored( + self, + logs_info_level: Any, + version: str, + ) -> None: + """Verify the scenario logs the configured wait timing values.""" + assert logs_info_level.find_log("message", value="Polling interval: 50ms") is not None + assert logs_info_level.find_log("message", value="Condition timeout: 2000ms") is not None + + +class TestConditionalLaunchingScenarioTimesOutOnUnmetConditions(LifecycleScenario): + """Verify the scenario fails when its wait conditions are never satisfied. + + Without this, an implementation that always reports success regardless of whether + a path exists, an env var is set, or a process is running would still pass the + happy-path test above. + """ + + @pytest.fixture(scope="class") + def scenario_name(self) -> str: + return "lifecycle.conditional_launching" + + @pytest.fixture(scope="class") + def test_config(self, temp_dir: Path) -> dict[str, Any]: + missing_path = temp_dir / "never_created.flag" + return { + "test": { + "wait_conditions": [ + f"path:{missing_path}", + "env:LM_CONDITION_NEVER_SET", + "process:process_that_does_not_exist_anywhere", + ], + "polling_interval_ms": 20, + "timeout_ms": 200, + }, + } + + def expect_command_failure(self) -> bool: + return True + + def capture_stderr(self) -> bool: + return True + + def test_scenario_fails_when_conditions_stay_unmet(self, results: ScenarioResult, version: str) -> None: + """Verify the scenario reports failure - and specifically a wait-condition timeout, + not merely any nonzero exit - when conditions are never satisfied.""" + assert results.return_code != ResultCode.SUCCESS, ( + f"Expected failure when wait conditions are never satisfied, got: {results}" + ) + assert results.stderr is not None + assert "Timed out" in results.stderr and "condition" in results.stderr, ( + f"Expected a wait-condition timeout error on stderr, got: {results.stderr}" + ) + + +class TestConditionalLaunchingScenarioDetectsConditionArrivingLate(LifecycleScenario): + """Verify the scenario is actually re-checking the condition over time (real polling), + rather than only ever observing the condition's state at process start (t=0) or its + absence at the very end (timeout). + + Without this, an implementation that checks the condition exactly once - either at + the very start or only right before giving up - would still pass both the + already-satisfied test and the never-satisfied timeout test above. + """ + + _DELAY_BEFORE_CONDITION_MET_S = 0.5 + _TIMEOUT_MS = 3000 + _POLLING_INTERVAL_MS = 100 + + @pytest.fixture(scope="class") + def scenario_name(self) -> str: + return "lifecycle.conditional_launching" + + @pytest.fixture(scope="class") + def flag_path(self, temp_dir: Path) -> Path: + return temp_dir / "lifecycle_launch_ready_late.flag" + + @pytest.fixture(scope="class") + def test_config(self, flag_path: Path) -> dict[str, Any]: + return { + "test": { + "wait_conditions": [f"path:{flag_path}"], + "polling_interval_ms": self._POLLING_INTERVAL_MS, + "timeout_ms": self._TIMEOUT_MS, + }, + } + + @pytest.fixture(scope="class") + def results( + self, + command: list[str], + execution_timeout: float, + flag_path: Path, + ) -> Generator[ScenarioResult, None, None]: + # Overrides the base class-scoped `results` fixture so the condition is armed before + # the command runs. The base fixture is also pulled in by the autouse `print_to_report` + # fixture ahead of the test body, so starting the trigger inside the test method itself + # is too late: the base fixture would already have run the command to completion and + # timed out before the test method ever executes. + # + # Use a separate subprocess to write the flag after the configured delay instead of a + # `threading.Timer`. Under heavy Bazel load, thread creation and scheduling can itself + # consume a meaningful slice of the delay budget, and the test is specifically checking + # that a condition becomes true mid-wait, not that a Python timer thread can fire before + # the OS reschedules it. A tiny helper subprocess makes the delay deterministic and + # independent of the test runner's thread scheduler while still exercising the real + # polling logic in the scenario under test. + start = time.monotonic() + trigger = subprocess.Popen( + [ + sys.executable, + "-c", + ( + "import pathlib, sys, time; " + "time.sleep(float(sys.argv[1])); " + "pathlib.Path(sys.argv[2]).write_text('ready', encoding='utf-8')" + ), + str(self._DELAY_BEFORE_CONDITION_MET_S), + str(flag_path), + ] + ) + try: + result = self._run_command(command, execution_timeout) + # A pytest fixture's `self` and a test method's `self` are different instances of + # the test class, so state can't be handed off via an instance attribute; stash it + # on the class object instead, which both share. + type(self)._elapsed_s = time.monotonic() - start + finally: + trigger.terminate() + try: + trigger.wait(timeout=1) + except subprocess.TimeoutExpired: + trigger.kill() + trigger.wait(timeout=1) + yield result + flag_path.unlink(missing_ok=True) + + def test_condition_satisfied_partway_through_the_wait_is_detected_promptly( + self, + results: ScenarioResult, + version: str, + ) -> None: + """Verify the scenario succeeds shortly after the condition becomes true mid-wait, + not merely at t=0 or by coincidentally still being true once the full timeout + elapses.""" + result = results + elapsed_s = self._elapsed_s + + assert result.return_code == ResultCode.SUCCESS, f"Expected success once the condition became true: {result}" + assert elapsed_s >= self._DELAY_BEFORE_CONDITION_MET_S, ( + f"Scenario reported success ({elapsed_s:.2f}s) before the condition could possibly have been " + f"true ({self._DELAY_BEFORE_CONDITION_MET_S}s) - it isn't actually observing the real condition." + ) + margin_s = self._TIMEOUT_MS / 1000 - self._DELAY_BEFORE_CONDITION_MET_S + assert elapsed_s < self._DELAY_BEFORE_CONDITION_MET_S + margin_s / 2, ( + f"Scenario took {elapsed_s:.2f}s to detect a condition that became true after " + f"{self._DELAY_BEFORE_CONDITION_MET_S}s - this is close to the full {self._TIMEOUT_MS}ms timeout, " + "suggesting it isn't re-checking at the configured polling interval." + ) + + +class TestConditionalLaunchingScenarioRejectsUnsupportedPrefix(LifecycleScenario): + """Verify an unsupported wait-condition prefix is rejected as invalid configuration, + distinct from a legitimate condition that simply times out unmet.""" + + @pytest.fixture(scope="class") + def scenario_name(self) -> str: + return "lifecycle.conditional_launching" + + _TIMEOUT_MS = 2000 + + @pytest.fixture(scope="class") + def test_config(self) -> dict[str, Any]: + return { + "test": { + "wait_conditions": ["badprefix:value"], + "polling_interval_ms": 20, + "timeout_ms": self._TIMEOUT_MS, + }, + } + + def expect_command_failure(self) -> bool: + return True + + def capture_stderr(self) -> bool: + return True + + @pytest.fixture(scope="class") + def results(self, command: list[str], execution_timeout: float) -> ScenarioResult: + # Overrides the base class-scoped `results` fixture to time the single command + # invocation. Without this override, the test body's own `self._run_command(...)` + # call would run the scenario binary a *second* time on top of the one the autouse + # `print_to_report` -> `logs` -> `results` chain already ran ahead of the test. + start = time.monotonic() + result = self._run_command(command, execution_timeout) + # A pytest fixture's `self` and a test method's `self` are different instances of + # the test class, so state can't be handed off via an instance attribute; stash it + # on the class object instead, which both share. + type(self)._elapsed_s = time.monotonic() - start + return result + + def test_unsupported_prefix_is_rejected_immediately( + self, + results: ScenarioResult, + version: str, + ) -> None: + """Verify validation rejects the condition outright, before entering the wait loop, + rather than only surfacing the same error after waiting out the full timeout.""" + result = results + elapsed_s = self._elapsed_s + + assert result.return_code != ResultCode.SUCCESS, ( + f"Expected failure for an unsupported wait-condition prefix, got: {result}" + ) + assert result.stderr is not None + assert "Unsupported wait condition prefix" in result.stderr, ( + f"Expected an unsupported-prefix validation error on stderr, got: {result.stderr}" + ) + assert elapsed_s < (self._TIMEOUT_MS / 1000) / 2, ( + f"Rejection took {elapsed_s:.2f}s, close to the full {self._TIMEOUT_MS}ms timeout - " + "the prefix should be validated up front, not discovered by waiting it out." + ) + + +class TestConditionalLaunchingScenarioRejectsEmptyConditions(LifecycleScenario): + """Verify an empty wait_conditions list is rejected as invalid configuration up front, + distinct from a legitimate condition that simply times out unmet. + + Without this, a future refactor of parse_wait_conditions/parse_string_array_field could + silently start treating an empty list as "nothing to wait for, immediate success" instead + of a configuration error - a regression this test would otherwise be the only thing to + catch, since the happy-path and unsupported-prefix tests never exercise this branch. + """ + + @pytest.fixture(scope="class") + def scenario_name(self) -> str: + return "lifecycle.conditional_launching" + + _TIMEOUT_MS = 2000 + + @pytest.fixture(scope="class") + def test_config(self) -> dict[str, Any]: + return { + "test": { + "wait_conditions": [], + "polling_interval_ms": 20, + "timeout_ms": self._TIMEOUT_MS, + }, + } + + def expect_command_failure(self) -> bool: + return True + + def capture_stderr(self) -> bool: + return True + + @pytest.fixture(scope="class") + def results(self, command: list[str], execution_timeout: float) -> ScenarioResult: + # See TestConditionalLaunchingScenarioRejectsUnsupportedPrefix.results: overrides the + # base class-scoped `results` fixture so the test body doesn't run the scenario binary + # a second time on top of the one the autouse fixture chain already ran. + start = time.monotonic() + result = self._run_command(command, execution_timeout) + type(self)._elapsed_s = time.monotonic() - start + return result + + def test_empty_conditions_are_rejected_immediately( + self, + results: ScenarioResult, + version: str, + ) -> None: + """Verify validation rejects an empty wait_conditions list outright, before entering + the wait loop, rather than only surfacing the same error after the full timeout.""" + result = results + elapsed_s = self._elapsed_s + + assert result.return_code != ResultCode.SUCCESS, ( + f"Expected failure for an empty wait_conditions list, got: {result}" + ) + assert result.stderr is not None + assert "Wait conditions were not provided" in result.stderr, ( + f"Expected a missing/empty wait_conditions validation error on stderr, got: {result.stderr}" + ) + assert elapsed_s < (self._TIMEOUT_MS / 1000) / 2, ( + f"Rejection took {elapsed_s:.2f}s, close to the full {self._TIMEOUT_MS}ms timeout - " + "an empty condition list should be validated up front, not discovered by waiting it out." + ) diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py b/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py new file mode 100644 index 00000000000..e064597374e --- /dev/null +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py @@ -0,0 +1,558 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +""" +Feature integration tests for lifecycle with running Launch Manager daemon. + +These tests validate actual supervision and lifecycle management behavior +by running test applications under a real Launch Manager daemon instance. + +To run these tests: + + # Run both Rust and C++ variants + pytest feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py -v + + # Run only Rust variant + pytest feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py -v -k rust + + # Run only C++ variant + pytest feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py -v -k cpp +""" + +import json +import re +import subprocess +import time +import os +from pathlib import Path +from typing import Any + +import pytest +from daemon_helpers import ( + first_pid, + is_running, + launch_manager_daemon, + pgrep_cmdline_pattern, + signal_process, + start_launch_manager_daemon, + stop_launch_manager_daemon, + wait_until, +) +from test_properties import add_test_properties + +pytestmark = [ + pytest.mark.parametrize("version", ["rust", "cpp"], scope="class"), +] + + +class TestProcessLaunchingWithDaemon: + """ + Verify lifecycle management with running Launch Manager daemon. + + These tests demonstrate end-to-end integration including: + - Process launching under supervision + - Execution state reporting to the daemon + - Process monitoring and health checks + - Recovery actions on failure + """ + + @staticmethod + def _proc_cmdline(pid: str) -> list[str]: + """Read process cmdline from /proc and split NUL-separated arguments.""" + raw = Path(f"/proc/{pid}/cmdline").read_bytes() + return [arg.decode("utf-8") for arg in raw.split(b"\0") if arg] + + @staticmethod + def _proc_environ(pid: str) -> dict[str, str]: + """Read process environment from /proc as a key/value mapping.""" + raw = Path(f"/proc/{pid}/environ").read_bytes() + env: dict[str, str] = {} + for item in raw.split(b"\0"): + if not item: + continue + key, sep, value = item.partition(b"=") + if not sep: + continue + env[key.decode("utf-8")] = value.decode("utf-8") + return env + + @staticmethod + def _proc_status_ids(pid: str) -> tuple[int, int] | None: + """Read effective uid/gid from /proc status for a process.""" + try: + lines = Path(f"/proc/{pid}/status").read_text(encoding="utf-8").splitlines() + except OSError: + return None + uid_line = next((line for line in lines if line.startswith("Uid:")), None) + gid_line = next((line for line in lines if line.startswith("Gid:")), None) + if uid_line is None or gid_line is None: + return None + try: + uid_parts = uid_line.split()[1:] + gid_parts = gid_line.split()[1:] + # /proc status format: real effective saved filesystem + return int(uid_parts[1]), int(gid_parts[1]) + except (IndexError, ValueError): + return None + + @staticmethod + def _proc_sched_policy_and_priority(pid: str) -> tuple[str, int] | None: + """Read scheduler policy and RT priority from chrt output for a process.""" + result = subprocess.run(["chrt", "-p", pid], capture_output=True, text=True, check=False) + if result.returncode != 0: + return None + + policy = None + priority = None + for line in result.stdout.splitlines(): + lower = line.lower().strip() + # e.g. "pid 400132's current scheduling policy: SCHED_OTHER" + if "scheduling policy" in lower: + policy = line.split(":", 1)[1].strip() + elif "scheduling priority" in lower: + try: + priority = int(line.split(":", 1)[1].strip()) + except ValueError: + return None + + if policy is None or priority is None: + return None + return policy, priority + + # Dependency-gating coverage (rust-on-cpp startup order) lives in + # test_conditional_launching.py; not duplicated here. + + @add_test_properties( + partially_verifies=["feat_req__lifecycle__launch_support"], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + def test_startup_declares_and_launches_multiple_processes( + self, + launch_manager_daemon: dict[str, Any], + version: str, + ) -> None: + """Verify startup run target includes multiple processes and both are launched.""" + config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" + config = json.loads(config_path.read_text(encoding="utf-8")) + startup_deps = config["run_targets"]["Startup"]["depends_on"] + + assert isinstance(startup_deps, list), "Startup depends_on should be a list" + assert len(startup_deps) >= 2, "Startup run target should define multiple process dependencies" + assert "cpp_supervised_app" in startup_deps, "cpp_supervised_app missing in Startup depends_on" + assert "rust_supervised_app" in startup_deps, "rust_supervised_app missing in Startup depends_on" + + daemon_info = launch_manager_daemon + daemon = daemon_info["daemon"] + cpp_path = str(daemon_info["apps"]["cpp"]) + rust_path = str(daemon_info["apps"]["rust"]) + both_running = wait_until( + lambda: is_running(cpp_path) and is_running(rust_path), + timeout_s=8.0, + ) + assert both_running, "Startup should launch all configured supervised processes" + assert daemon.is_running(), "Launch Manager daemon stopped unexpectedly" + + @add_test_properties( + partially_verifies=["feat_req__lifecycle__process_launch_args"], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + def test_launch_process_arguments_are_applied( + self, + launch_manager_daemon: dict[str, Any], + version: str, + ) -> None: + """Verify launched process cmdline includes every configured lifecycle argument. + + Expected args are read from the config itself, not hardcoded, so the test + tracks config drift. + """ + daemon_info = launch_manager_daemon + app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" + app_path = str(daemon_info["apps"][version]) + + config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" + config = json.loads(config_path.read_text(encoding="utf-8")) + configured_args = config["components"][app_name]["component_properties"]["process_arguments"] + assert configured_args, f"{app_name} does not configure any process_arguments to verify against" + + started = wait_until(lambda: is_running(app_path), timeout_s=8.0) + assert started, f"{app_name} was not launched before argument verification" + + pid = first_pid(app_path) + assert pid is not None, f"Could not resolve PID for {app_name}" + cmdline = self._proc_cmdline(pid) + + assert cmdline, f"Could not read command line arguments for {app_name} pid={pid}" + for configured_arg in configured_args: + assert configured_arg in cmdline, ( + f"Configured launch argument {configured_arg!r} missing in {app_name} cmdline: {cmdline}" + ) + + @add_test_properties( + partially_verifies=["feat_req__lifecycle__process_launch_args"], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + def test_launch_process_environment_is_applied( + self, + launch_manager_daemon: dict[str, Any], + version: str, + ) -> None: + """Verify launched process environment matches every configured environment variable. + + Expected variables are read from the config itself, not hardcoded, so the + test tracks config drift. + """ + daemon_info = launch_manager_daemon + app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" + app_path = str(daemon_info["apps"][version]) + + config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" + config = json.loads(config_path.read_text(encoding="utf-8")) + configured_env = config["components"][app_name]["deployment_config"]["environmental_variables"] + assert configured_env, f"{app_name} does not configure any environmental_variables to verify against" + + started = wait_until(lambda: is_running(app_path), timeout_s=8.0) + assert started, f"{app_name} was not launched before environment verification" + + pid = first_pid(app_path) + assert pid is not None, f"Could not resolve PID for {app_name}" + proc_env = self._proc_environ(pid) + + for key, expected_value in configured_env.items(): + assert proc_env.get(key) == expected_value, ( + f"{key} mismatch for {app_name}: expected {expected_value!r}, got {proc_env.get(key)!r}" + ) + + def test_config_defines_uid_gid_scheduling_and_priority(self, version: str) -> None: + """Sanity-check the lifecycle config's shape for launch user/group and scheduling defaults. + + No requirement tag here: this only confirms the config file is well-formed, not + that launch_manager applies it - that's covered by + test_launched_process_uid_gid_matches_config_when_applied and + test_launched_process_scheduling_matches_config_when_applied below, which inspect + the real launched process. `version` is unused but required by the class-scope + parametrize on this class. + """ + config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" + config = json.loads(config_path.read_text(encoding="utf-8")) + + sandbox = config["defaults"]["deployment_config"]["sandbox"] + assert isinstance(sandbox.get("uid"), int), "Expected integer uid in sandbox defaults" + assert isinstance(sandbox.get("gid"), int), "Expected integer gid in sandbox defaults" + assert isinstance(sandbox.get("scheduling_priority"), int), "Expected integer scheduling priority" + assert isinstance(sandbox.get("scheduling_policy"), str), "Expected scheduling policy string" + + @add_test_properties( + partially_verifies=["feat_req__lifecycle__uid_gid_support"], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + # Skipped in CI/CD (both rust/cpp): requires launch_manager to gain cap_setuid/cap_setgid via + # setcap, which needs both FIT_ENABLE_SETCAP=1 (unset in the GitHub Actions workflow) and + # unsandboxed execution (linux-sandbox's PR_SET_NO_NEW_PRIVS makes the grant inert at exec + # time even if setcap itself succeeds). See feature_integration_tests/README.md for details. + def test_launched_process_uid_gid_matches_config_when_applied( + self, + launch_manager_daemon: dict[str, Any], + version: str, + ) -> None: + """Verify launched process runs with configured effective uid/gid when runtime applies sandbox identity.""" + daemon_info = launch_manager_daemon + if not daemon_info["sandbox_privileged"]: + pytest.skip( + "launch_manager was not granted cap_setuid/cap_setgid in this environment; " + f"sandbox uid/gid cannot be applied. Reason: {daemon_info['sandbox_privileged_reason']}" + ) + + app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" + app_path = str(daemon_info["apps"][version]) + + config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" + config = json.loads(config_path.read_text(encoding="utf-8")) + sandbox = config["defaults"]["deployment_config"]["sandbox"] + expected_uid = int(sandbox["uid"]) + expected_gid = int(sandbox["gid"]) + + started = wait_until(lambda: is_running(app_path), timeout_s=8.0) + assert started, f"{app_name} was not launched before uid/gid verification" + + pid = first_pid(app_path) + assert pid is not None, f"Could not resolve PID for {app_name}" + proc_ids = self._proc_status_ids(pid) + assert proc_ids is not None, f"Could not read /proc status uid/gid for {app_name} pid={pid}" + + effective_uid, effective_gid = proc_ids + assert effective_uid == expected_uid, ( + f"Effective uid mismatch for {app_name}: expected {expected_uid}, got {effective_uid}" + ) + assert effective_gid == expected_gid, ( + f"Effective gid mismatch for {app_name}: expected {expected_gid}, got {effective_gid}" + ) + + @add_test_properties( + partially_verifies=[ + "feat_req__lifecycle__launch_priority_support", + "feat_req__lifecycle__scheduling_policy", + ], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + # Skipped in CI/CD (both rust/cpp): requires launch_manager to gain cap_sys_nice via setcap, + # which needs both FIT_ENABLE_SETCAP=1 (unset in the GitHub Actions workflow) and unsandboxed + # execution (linux-sandbox's PR_SET_NO_NEW_PRIVS makes the grant inert at exec time even if + # setcap itself succeeds). See feature_integration_tests/README.md for details. + def test_launched_process_scheduling_matches_config_when_applied( + self, + launch_manager_daemon: dict[str, Any], + version: str, + ) -> None: + """Verify launched process uses configured scheduler policy and priority when applied.""" + daemon_info = launch_manager_daemon + if not daemon_info["sandbox_privileged"]: + pytest.skip( + "launch_manager was not granted cap_sys_nice in this environment; " + f"scheduling policy cannot be applied. Reason: {daemon_info['sandbox_privileged_reason']}" + ) + + app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" + app_path = str(daemon_info["apps"][version]) + + config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" + config = json.loads(config_path.read_text(encoding="utf-8")) + sandbox = config["defaults"]["deployment_config"]["sandbox"] + configured_policy = sandbox["scheduling_policy"] + configured_priority = int(sandbox["scheduling_priority"]) + + started = wait_until(lambda: is_running(app_path), timeout_s=8.0) + assert started, f"{app_name} was not launched before scheduling verification" + + # Pid can go stale between resolution and the chrt call if the app restarts; + # retry against a fresh pid rather than failing on that race. + sched = None + pid = None + for _ in range(20): + pid = first_pid(app_path) + if pid is None: + time.sleep(0.1) + continue + sched = self._proc_sched_policy_and_priority(pid) + if sched is not None: + break + time.sleep(0.1) + assert pid is not None, f"Could not resolve PID for {app_name}" + assert sched is not None, f"Could not read scheduling metadata via chrt for {app_name} pid={pid}" + + policy, rt_priority = sched + expected_policy = configured_policy.upper() + assert policy.upper() == expected_policy, ( + f"Scheduling policy mismatch for {app_name}: expected {expected_policy}, got {policy}" + ) + assert rt_priority == configured_priority, ( + f"Scheduling priority mismatch for {app_name}: expected {configured_priority}, got {rt_priority}" + ) + + @add_test_properties( + partially_verifies=["feat_req__lifecycle__secpol_non_root"], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + def test_launch_manager_and_apps_are_not_running_as_root( + self, + launch_manager_daemon: dict[str, Any], + version: str, + ) -> None: + """Verify launch setup executes without root privileges in this integration setup.""" + daemon_info = launch_manager_daemon + daemon = daemon_info["daemon"] + app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" + app_path = str(daemon_info["apps"][version]) + + assert os.geteuid() != 0, "Test environment unexpectedly runs as root" + assert daemon.pid() > 0, "Launch Manager daemon pid should be available" + + started = wait_until(lambda: is_running(app_path), timeout_s=8.0) + assert started, f"{app_name} was not launched before non-root verification" + + pid = first_pid(app_path) + assert pid is not None, f"Could not resolve PID for {app_name}" + proc_ids = self._proc_status_ids(pid) + assert proc_ids is not None, f"Could not read /proc status uid/gid for {app_name} pid={pid}" + effective_uid, _ = proc_ids + assert effective_uid != 0, f"{app_name} is unexpectedly running as root" + + @add_test_properties( + partially_verifies=[ + "feat_req__lifecycle__monitor_abnormal_term", + "feat_req__lifecycle__recovery_action_support", + ], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + def test_supervised_app_recovery( + self, + tmp_path_factory: pytest.TempPathFactory, + version: str, + ) -> None: + """Verify daemon restarts a killed supervised app in place per the retry policy. + + Also confirms the other supervised app is left untouched, proving recovery + went through `ready_recovery_action.restart` rather than a run-target switch. + + Does not claim `feat_req__lifecycle__retries_configurable`: this only exercises a + single restart within the configured attempt budget, it never varies or exhausts + `number_of_attempts`, so the "configurable" half of that requirement is unverified. + """ + daemon_info = start_launch_manager_daemon(tmp_path_factory) + try: + daemon = daemon_info["daemon"] + app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" + app_path = str(daemon_info["apps"][version]) + other_version = "cpp" if version == "rust" else "rust" + other_app_path = str(daemon_info["apps"][other_version]) + + started = wait_until(lambda: is_running(app_path), timeout_s=8.0) + assert started, f"{app_name} was not running before recovery test" + + old_pid = first_pid(app_path) + assert old_pid is not None, f"Could not resolve PID for {app_name}" + other_old_pid = first_pid(other_app_path) + assert other_old_pid is not None, "Could not resolve PID for the other supervised app" + + sent, reason = signal_process(old_pid, "-9", sandbox_privileged=daemon_info["sandbox_privileged"]) + assert sent, f"Could not signal {app_name} (pid={old_pid}): {reason}" + + restarted = wait_until( + lambda: (new_pid := first_pid(app_path)) is not None and new_pid != old_pid, + timeout_s=12.0, + ) + assert restarted, f"{app_name} was not restarted after forced termination" + + assert daemon.is_running(), "Launch Manager daemon should still be running after recovery" + + other_new_pid = first_pid(other_app_path) + assert other_new_pid == other_old_pid, ( + "The other supervised app was relaunched too, indicating recovery switched the " + "whole run target instead of retrying only the failed app per the configured " + "restart policy" + ) + finally: + stop_launch_manager_daemon(daemon_info) + + +class TestParallelLaunch: + """Verify genuinely parallel launch of independent components. + + Runs its own launch_manager instance (rather than the shared class-scoped + `launch_manager_daemon` fixture used by TestProcessLaunchingWithDaemon), for two reasons: + + 1. `lifecycle_daemon_parallel_launch_config.json` has no depends_on between the + two apps, unlike the shared fixture's config - that's the whole point. + 2. Every daemon receives an independent runtime directory and generated config + beneath `TEST_TMPDIR`, so it cannot interfere with the shared fixture. + + Parametrized on `version` only because the module-level `pytestmark` applies it + to every class in this file; parallel launch itself is independent of which + scenario variant is under test elsewhere, so `version` is unused here. + """ + + @add_test_properties( + partially_verifies=["feat_req__lifecycle__parallel_launch_support"], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + def test_independent_processes_launch_without_waiting_on_each_other( + self, + tmp_path_factory: pytest.TempPathFactory, + version: str, + ) -> None: + """Verify two independent components launch in parallel, not one-after-the-other. + + `lifecycle_daemon_config.json` has rust_supervised_app depend on + cpp_supervised_app, so it cannot demonstrate parallel launch - both apps + eventually running there is equally consistent with strict serialization. + + Runs against `lifecycle_daemon_parallel_launch_config.json`, where neither + app depends on the other, and withholds one app's binary (non-executable) at + a time. If launch order were still serialized (e.g. alphabetically or by + declaration order), withholding the first-launched app would also block the + second. The other app reaching Running regardless of which one is withheld + shows launch does not wait on the withheld one, i.e. genuine parallel launch. + + `version` is unused but required by the module-scope parametrize. + """ + for blocked, other in (("cpp", "rust"), ("rust", "cpp")): + daemon_info = start_launch_manager_daemon( + tmp_path_factory, + blocked_apps=frozenset({blocked}), + wait_for_apps=False, + config_template="//feature_integration_tests/configs:lifecycle_daemon_parallel_launch_config.json", + ) + try: + other_path = str(daemon_info["apps"][other]) + other_started = wait_until(lambda: is_running(other_path), timeout_s=8.0) + assert other_started, ( + f"{other}_supervised_app did not start while {blocked}_supervised_app was " + "withheld, even though neither depends on the other - launch is not parallel" + ) + finally: + stop_launch_manager_daemon(daemon_info) + + +class TestHealthMonitoringWithDaemon: + """Health monitoring / watchdog tests with daemon.""" + + @add_test_properties( + partially_verifies=[ + "feat_req__lifecycle__liveliness_detection", + "feat_req__lifecycle__smart_watchdog_config", + ], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + def test_watchdog_detection(self, launch_manager_daemon: dict[str, Any], version: str) -> None: + """Verify watchdog detects an unresponsive app (stopped, not reporting health) and reacts.""" + daemon = launch_manager_daemon["daemon"] + app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" + + # Stop the supervised process to emulate a non-reporting workload. + app_path = str(launch_manager_daemon["apps"][version]) + result = subprocess.run( + ["pgrep", "-f", pgrep_cmdline_pattern(app_path)], + capture_output=True, + text=True, + check=False, + ) + if result.returncode != 0: + pytest.skip(f"{app_name} not active; activate Running run target before watchdog check") + + pid = result.stdout.strip().split("\n")[0] + sandbox_privileged = launch_manager_daemon["sandbox_privileged"] + sent, reason = signal_process(pid, "-STOP", sandbox_privileged=sandbox_privileged) + assert sent, f"Could not signal {app_name} (pid={pid}): {reason}" + try: + # Allow supervision/watchdog loop to detect stalled process. + time.sleep(4.0) + logs = daemon.get_logs() + watchdog_patterns = [ + rf"Got kRunning timeout for process.*\(\s*{re.escape(app_name)}\s*\)", + rf"unexpected termination of process.*\(\s*{re.escape(app_name)}\s*\)", + rf"Alive Supervision \(\s*{re.escape(app_name)}_alive_supervision\s*\) switched to FAILED", + rf"Alive Supervision \(\s*{re.escape(app_name)}_alive_supervision\s*\) switched to EXPIRED", + ] + assert any(re.search(pattern, logs) for pattern in watchdog_patterns), ( + f"No target-specific watchdog diagnostics found for {app_name}.\nDaemon logs:\n{logs}" + ) + finally: + signal_process(pid, "-CONT", sandbox_privileged=sandbox_privileged) diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py b/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py new file mode 100644 index 00000000000..f0fc82bcc58 --- /dev/null +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py @@ -0,0 +1,148 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Real retry-exhaustion coverage for `feat_req__lifecycle__retries_configurable`. + +Drives a dedicated `flaky_startup_app` (see support_apps/flaky_startup_app) whose +crash-before-success count is fixed in the launch_manager config, so both sides of +`ready_recovery_action.restart.number_of_attempts` are exercised deterministically +instead of relying on a real, racy startup failure: + +- `TestRetrySucceedsWithinConfiguredAttempts`: the component crashes fewer times + than the configured attempts allow, so it must recover and reach Running. +- `TestRetryExhaustionTriggersRecovery`: the component always crashes, so the + daemon must give up after exactly the configured attempts and execute the run + target's `recovery_action` (switch to `fallback_run_target`) instead of + restarting forever. +""" + +from __future__ import annotations + +from typing import Any + +import pytest +from daemon_helpers import ( + is_running, + read_retry_attempt_count, + start_flaky_retry_daemon, + stop_flaky_retry_daemon, + wait_until, +) +from test_properties import add_test_properties + +# Must match "number_of_attempts" in both lifecycle_daemon_retry_*_config.json. +_NUMBER_OF_ATTEMPTS = 2 + + +@pytest.fixture(scope="class") +def recovers_daemon(tmp_path_factory: pytest.TempPathFactory): + daemon_info = start_flaky_retry_daemon( + tmp_path_factory, + "//feature_integration_tests/configs:lifecycle_daemon_retry_recovers_config.json", + crashes_before_success=2, + ) + try: + yield daemon_info + finally: + stop_flaky_retry_daemon(daemon_info) + + +@pytest.fixture(scope="class") +def exhausts_daemon(tmp_path_factory: pytest.TempPathFactory): + daemon_info = start_flaky_retry_daemon( + tmp_path_factory, + "//feature_integration_tests/configs:lifecycle_daemon_retry_exhausts_config.json", + crashes_before_success=999, + ) + try: + yield daemon_info + finally: + stop_flaky_retry_daemon(daemon_info) + + +class TestRetrySucceedsWithinConfiguredAttempts: + """The component crashes fewer times than `number_of_attempts` allows.""" + + @add_test_properties( + partially_verifies=["feat_req__lifecycle__retries_configurable"], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + def test_component_recovers_within_configured_attempts(self, recovers_daemon: dict[str, Any]) -> None: + """Daemon retries a failing component up to `number_of_attempts` and lets + it reach Running once it stops crashing. + """ + app_path = recovers_daemon["app_path"] + counter_path = recovers_daemon["counter_path"] + expected_attempts = recovers_daemon["crashes_before_success"] + 1 + + # Check the attempt counter before is_running(): a crashing attempt is still + # technically "running" for the microseconds before it aborts, so polling + # is_running() first can catch that transient window rather than the + # eventual successful attempt. + reached = wait_until(lambda: read_retry_attempt_count(counter_path) >= expected_attempts, timeout_s=8.0) + assert reached, "flaky_startup_app never reached the expected number of launch attempts" + + attempts = read_retry_attempt_count(counter_path) + assert attempts == expected_attempts, ( + f"Expected exactly {expected_attempts} launch attempts (crashes_before_success + 1 success), got {attempts}" + ) + + started = wait_until(lambda: is_running(app_path), timeout_s=2.0) + assert started, "flaky_startup_app never reached Running after its last launch attempt" + + # Once healthy it should stay up: no further restarts. + relaunched = wait_until(lambda: read_retry_attempt_count(counter_path) != attempts, timeout_s=1.5) + assert not relaunched, "Component was relaunched again after it was already Running" + assert is_running(app_path), "flaky_startup_app stopped running after recovering" + + +class TestRetryExhaustionTriggersRecovery: + """The component always crashes, exceeding `number_of_attempts`.""" + + @add_test_properties( + partially_verifies=["feat_req__lifecycle__retries_configurable"], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + def test_daemon_gives_up_after_configured_attempts(self, exhausts_daemon: dict[str, Any]) -> None: + """Daemon stops restarting a component once `number_of_attempts` is + exhausted, instead of retrying forever, and executes the run target's + `recovery_action` (switch to `fallback_run_target`). + """ + app_path = exhausts_daemon["app_path"] + counter_path = exhausts_daemon["counter_path"] + + settled = wait_until( + lambda: read_retry_attempt_count(counter_path) >= _NUMBER_OF_ATTEMPTS + 1, + timeout_s=8.0, + ) + assert settled, "flaky_startup_app never reached the configured number of launch attempts" + + # Give the daemon a chance to keep retrying, if it hasn't actually given up. + attempts_after_exhaustion = read_retry_attempt_count(counter_path) + kept_retrying = wait_until( + lambda: read_retry_attempt_count(counter_path) != attempts_after_exhaustion, + timeout_s=2.0, + ) + assert not kept_retrying, ( + f"Daemon kept restarting the component past the configured number_of_attempts={_NUMBER_OF_ATTEMPTS}" + ) + assert attempts_after_exhaustion == _NUMBER_OF_ATTEMPTS + 1, ( + f"Expected exactly {_NUMBER_OF_ATTEMPTS + 1} launch attempts before giving up, got " + f"{attempts_after_exhaustion}" + ) + assert not is_running(app_path), ( + "flaky_startup_app is still running after exhausting retries; recovery_action " + "(switch_run_target -> fallback_run_target) should have stopped further attempts" + ) + assert exhausts_daemon["daemon"].is_running(), "Launch Manager daemon crashed instead of switching run target" diff --git a/feature_integration_tests/test_scenarios/cpp/src/internals/log_helpers.h b/feature_integration_tests/test_scenarios/cpp/src/internals/log_helpers.h new file mode 100644 index 00000000000..5df9f60b11d --- /dev/null +++ b/feature_integration_tests/test_scenarios/cpp/src/internals/log_helpers.h @@ -0,0 +1,135 @@ +/******************************************************************************** + * Copyright (c) 2026 Contributors to the Eclipse Foundation + * + * See the NOTICE file(s) distributed with this work for additional + * information regarding copyright ownership. + * + * This program and the accompanying materials are made available under the + * terms of the Apache License Version 2.0 which is available at + * https://www.apache.org/licenses/LICENSE-2.0 + * + * SPDX-License-Identifier: Apache-2.0 + ********************************************************************************/ + +#ifndef INTERNALS_LOG_HELPERS_H_ +#define INTERNALS_LOG_HELPERS_H_ + +#include +#include +#include +#include +#include +#include + +namespace log_helpers { + +/** + * @brief Return the current UNIX timestamp as a decimal string (seconds). + * + * Used to populate the "timestamp" field in structured JSON log lines so that + * the C++ output matches the Rust tracing JSON shape expected by the FIT log + * filters. + * + * @return String containing the number of seconds since the UNIX epoch. + */ +inline std::string unix_seconds_string() { + const auto now = std::chrono::system_clock::now(); + const auto secs = + std::chrono::duration_cast(now.time_since_epoch()).count(); + return std::to_string(secs); +} + +/** + * @brief Escape a string for embedding as a JSON string value. + * + * Escapes '"', '\\', and control characters. Without this, interpolating a raw + * value (e.g. a filesystem path containing '"' or '\\') straight into a JSON + * string produces malformed JSON that downstream JSON-based log parsing (e.g. + * Python's FIT LogContainer) cannot read back. + * + * @param value Raw string to escape. + * @return JSON-escaped string, without surrounding quotes. + */ +inline std::string json_escape(const std::string& value) { + std::string escaped; + escaped.reserve(value.size()); + for (const char c : value) { + switch (c) { + case '"': + escaped += "\\\""; + break; + case '\\': + escaped += "\\\\"; + break; + case '\n': + escaped += "\\n"; + break; + case '\r': + escaped += "\\r"; + break; + case '\t': + escaped += "\\t"; + break; + default: + if (static_cast(c) < 0x20) { + std::ostringstream oss; + oss << "\\u" << std::hex << std::setfill('0') << std::setw(4) + << static_cast(static_cast(c)); + escaped += oss.str(); + } else { + escaped += c; + } + } + } + return escaped; +} + +/** + * @brief Emit a structured JSON INFO log line to stdout. + * + * Matches the Rust tracing JSON format expected by the FIT LogContainer so + * that Python test assertions can use find_log() uniformly for both Rust and + * C++ scenarios. + * + * Example output: + * @code + * {"timestamp":"1234567890","level":"INFO","fields":{"key":"my_key","value":42.0}, + * "target":"cpp_test_scenarios::scenarios::persistency::my_module","threadId":"ThreadId(1)"} + * @endcode + * + * @param fields JSON fragment for the "fields" object, e.g. @c "\"key\":\"x\",\"value\":1.0" + * Caller is responsible for escaping any string values embedded here. + * @param target Module target string embedded in the log line. + */ +inline void log_info(const std::string& fields, const std::string& target) { + std::cout << "{\"timestamp\":\"" << unix_seconds_string() + << "\",\"level\":\"INFO\",\"fields\":{" << fields + << "},\"target\":\"" << json_escape(target) + << "\",\"threadId\":\"ThreadId(1)\"}\n"; +} + +/** + * @brief Format a double value to match Python's str(float) representation. + * + * For whole-number values (e.g. 42.0, 200.0) this appends ".0" so that the + * resulting string matches what Python's f-string interpolation produces. + * Non-integer values (e.g. 3.14) are printed as-is by the default stream. + * + * @param v Double value to format. + * @return String representation matching Python float str(). + */ +inline std::string format_double_python(double v) { + std::ostringstream oss; + oss.imbue(std::locale::classic()); // Ensure '.' decimal separator regardless of process locale. + oss << v; + std::string s = oss.str(); + if (s.find('.') == std::string::npos && s.find('e') == std::string::npos && + s.find('E') == std::string::npos) { + s += ".0"; + } + return s; +} + +} // namespace log_helpers + +#endif // INTERNALS_LOG_HELPERS_H_ diff --git a/feature_integration_tests/test_scenarios/cpp/src/scenarios/lifecycle/conditional_launching.cpp b/feature_integration_tests/test_scenarios/cpp/src/scenarios/lifecycle/conditional_launching.cpp new file mode 100644 index 00000000000..3ba73ab2dd9 --- /dev/null +++ b/feature_integration_tests/test_scenarios/cpp/src/scenarios/lifecycle/conditional_launching.cpp @@ -0,0 +1,275 @@ +/******************************************************************************** + * Copyright (c) 2026 Contributors to the Eclipse Foundation + * + * See the NOTICE file(s) distributed with this work for additional + * information regarding copyright ownership. + * + * This program and the accompanying materials are made available under the + * terms of the Apache License Version 2.0 which is available at + * https://www.apache.org/licenses/LICENSE-2.0 + * + * SPDX-License-Identifier: Apache-2.0 + ********************************************************************************/ + +#include "conditional_launching.h" + +#include "internals/log_helpers.h" +#include "score/json/json_parser.h" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +namespace { + +constexpr const char* kTarget = "cpp_test_scenarios::scenarios::lifecycle::conditional_launching"; + +void log_info(const std::string& message) { + log_helpers::log_info("\"message\":\"" + log_helpers::json_escape(message) + "\"", kTarget); +} + +bool path_condition_met(const std::string& path) { + std::error_code ec; + return std::filesystem::exists(path, ec) && !ec; +} + +bool env_condition_met(const std::string& name) { + return std::getenv(name.c_str()) != nullptr; +} + +// Best-effort check whether a process matching `process_name` is currently running, by scanning +// /proc//comm (the kernel-truncated 15-char command name) and /proc//cmdline (the full +// argv[0], which covers names comm truncates). +bool process_condition_met(const std::string& process_name) { + // Iterating /proc races with processes exiting mid-scan (ENOENT on a just-vanished pid's + // subdirectory); std::filesystem surfaces that as filesystem_error even with the + // non-throwing error_code constructor, since only construction/increment on the top-level + // directory is covered, not opening files underneath. Treat it as "not found this pass" + // rather than letting a race abort the whole wait loop. + try { + std::error_code ec; + for (const auto& entry : std::filesystem::directory_iterator( + "/proc", std::filesystem::directory_options::skip_permission_denied, ec)) { + const std::string pid = entry.path().filename().string(); + if (pid.empty() || + !std::all_of(pid.begin(), pid.end(), [](unsigned char c) { return std::isdigit(c); })) { + continue; + } + + std::ifstream comm(entry.path() / "comm"); + std::string comm_value; + if (comm && std::getline(comm, comm_value) && comm_value == process_name) { + return true; + } + + std::ifstream cmdline(entry.path() / "cmdline"); + std::stringstream cmdline_buffer; + cmdline_buffer << cmdline.rdbuf(); + const std::string argv0 = cmdline_buffer.str(); + if (!argv0.empty()) { + const auto argv0_end = argv0.find('\0'); + const std::string first_arg = argv0.substr(0, argv0_end); + // Compare the basename only (portion after the last '/'), not a raw suffix of + // the full path: a plain suffix match would also accept e.g. "/usr/bin/oversleep" + // as satisfying process_name="sleep". + const auto slash_pos = first_arg.find_last_of('/'); + const std::string basename = + slash_pos == std::string::npos ? first_arg : first_arg.substr(slash_pos + 1); + if (basename == process_name) { + return true; + } + } + } + } catch (const std::filesystem::filesystem_error&) { + return false; + } + return false; +} + +template +std::vector parse_string_array_field(const std::string& input, + const std::string& field_name, + Converter convert) { + std::vector values; + + const score::json::JsonParser parser; + const auto root_any_res = parser.FromBuffer(input); + if (!root_any_res.has_value()) { + return values; + } + + const auto root_object_res = root_any_res.value().As(); + if (!root_object_res.has_value()) { + return values; + } + + const auto& root = root_object_res.value().get(); + const auto test_it = root.find("test"); + if (test_it == root.end()) { + return values; + } + + const auto test_object_res = test_it->second.As(); + if (!test_object_res.has_value()) { + return values; + } + + const auto& test = test_object_res.value().get(); + const auto field_it = test.find(field_name); + if (field_it == test.end()) { + return values; + } + + const auto array_res = field_it->second.As(); + if (!array_res.has_value()) { + return values; + } + + for (const auto& element : array_res.value().get()) { + const auto converted = convert(element); + if (!converted.has_value()) { + throw std::invalid_argument("Wait condition entries must be strings"); + } + values.push_back(*converted); + } + + return values; +} + +std::vector parse_wait_conditions(const std::string& input) { + return parse_string_array_field(input, "wait_conditions", [](const score::json::Any& element) { + const auto value = element.As(); + if (!value.has_value()) { + return std::optional{}; + } + return std::optional{value.value()}; + }); +} + +class ConditionalLaunching : public Scenario { +public: + std::string name() const override { return "conditional_launching"; } + + void run(const std::string& input) const override { + const score::json::JsonParser parser; + const auto root_any_res = parser.FromBuffer(input); + if (!root_any_res.has_value()) { + throw std::invalid_argument("Failed to parse scenario input JSON"); + } + + uint64_t polling_interval = 50; + uint64_t timeout = 5000; + const auto wait_conditions = parse_wait_conditions(input); + + const auto root_object_res = root_any_res.value().As(); + if (root_object_res.has_value()) { + const auto& root = root_object_res.value().get(); + const auto test_it = root.find("test"); + if (test_it != root.end()) { + const auto test_object_res = test_it->second.As(); + if (test_object_res.has_value()) { + const auto& test = test_object_res.value().get(); + + const auto polling_it = test.find("polling_interval_ms"); + if (polling_it != test.end()) { + const auto polling_res = polling_it->second.As(); + if (polling_res.has_value()) { + polling_interval = polling_res.value(); + } + } + + const auto timeout_it = test.find("timeout_ms"); + if (timeout_it != test.end()) { + const auto timeout_res = timeout_it->second.As(); + if (timeout_res.has_value()) { + timeout = timeout_res.value(); + } + } + } + } + } + + if (wait_conditions.empty()) { + throw std::runtime_error( + "Wait conditions were not provided: missing or empty 'test.wait_conditions' in scenario input"); + } + + log_info("Testing conditional launching"); + + for (const auto& condition : wait_conditions) { + if (condition.rfind("path:", 0) != 0U && condition.rfind("env:", 0) != 0U && + condition.rfind("process:", 0) != 0U) { + throw std::runtime_error("Unsupported wait condition prefix: " + condition); + } + } + + log_info("Polling interval: " + std::to_string(polling_interval) + "ms"); + log_info("Condition timeout: " + std::to_string(timeout) + "ms"); + + const auto deadline = std::chrono::steady_clock::now() + std::chrono::milliseconds(timeout); + std::vector satisfied(wait_conditions.size(), false); + + while (true) { + bool all_satisfied = true; + for (std::size_t i = 0; i < wait_conditions.size(); ++i) { + if (satisfied[i]) { + continue; + } + const auto& condition = wait_conditions[i]; + bool met = false; + if (condition.rfind("path:", 0) == 0U) { + met = path_condition_met(condition.substr(5)); + } else if (condition.rfind("env:", 0) == 0U) { + met = env_condition_met(condition.substr(4)); + } else { + met = process_condition_met(condition.substr(8)); + } + + if (met) { + satisfied[i] = true; + log_info("Condition satisfied: " + condition); + } else { + all_satisfied = false; + } + } + + if (all_satisfied) { + break; + } + + if (std::chrono::steady_clock::now() >= deadline) { + std::string unmet; + for (std::size_t i = 0; i < wait_conditions.size(); ++i) { + if (!satisfied[i]) { + if (!unmet.empty()) { + unmet += ", "; + } + unmet += wait_conditions[i]; + } + } + throw std::runtime_error("Timed out after " + std::to_string(timeout) + + "ms waiting for condition(s): " + unmet); + } + + std::this_thread::sleep_for(std::chrono::milliseconds(polling_interval)); + } + + log_info("All dependencies satisfied"); + } +}; + +} // namespace + +Scenario::Ptr make_conditional_launching_scenario() { + return std::make_shared(); +} diff --git a/feature_integration_tests/test_scenarios/cpp/src/scenarios/lifecycle/conditional_launching.h b/feature_integration_tests/test_scenarios/cpp/src/scenarios/lifecycle/conditional_launching.h new file mode 100644 index 00000000000..95a4a6a1797 --- /dev/null +++ b/feature_integration_tests/test_scenarios/cpp/src/scenarios/lifecycle/conditional_launching.h @@ -0,0 +1,17 @@ +/******************************************************************************** + * Copyright (c) 2026 Contributors to the Eclipse Foundation + * + * See the NOTICE file(s) distributed with this work for additional + * information regarding copyright ownership. + * + * This program and the accompanying materials are made available under the + * terms of the Apache License Version 2.0 which is available at + * https://www.apache.org/licenses/LICENSE-2.0 + * + * SPDX-License-Identifier: Apache-2.0 + ********************************************************************************/ +#pragma once + +#include + +Scenario::Ptr make_conditional_launching_scenario(); diff --git a/feature_integration_tests/test_scenarios/cpp/src/scenarios/mod.cpp b/feature_integration_tests/test_scenarios/cpp/src/scenarios/mod.cpp index 83a32e5af8e..67bac4d8a2e 100644 --- a/feature_integration_tests/test_scenarios/cpp/src/scenarios/mod.cpp +++ b/feature_integration_tests/test_scenarios/cpp/src/scenarios/mod.cpp @@ -13,6 +13,8 @@ #include +#include "scenarios/lifecycle/conditional_launching.h" + #include Scenario::Ptr make_multiple_kvs_per_app_scenario(); @@ -38,9 +40,18 @@ ScenarioGroup::Ptr persistency_scenario_group() { std::vector{supported_datatypes_group(), default_values_group()}); } +ScenarioGroup::Ptr lifecycle_scenario_group() { + return std::make_shared( + "lifecycle", + std::vector{ + make_conditional_launching_scenario(), + }, + std::vector{}); +} + ScenarioGroup::Ptr root_scenario_group() { return std::make_shared( "root", std::vector{}, - std::vector{persistency_scenario_group()}); + std::vector{persistency_scenario_group(), lifecycle_scenario_group()}); } diff --git a/feature_integration_tests/test_scenarios/rust/src/main.rs b/feature_integration_tests/test_scenarios/rust/src/main.rs index 024b09a2555..aedd088a3fc 100644 --- a/feature_integration_tests/test_scenarios/rust/src/main.rs +++ b/feature_integration_tests/test_scenarios/rust/src/main.rs @@ -22,6 +22,7 @@ use std::time::{SystemTime, UNIX_EPOCH}; use tracing::Level; use tracing_subscriber::fmt::time::FormatTime; use tracing_subscriber::FmtSubscriber; + struct NumericUnixTime; impl FormatTime for NumericUnixTime { diff --git a/feature_integration_tests/test_scenarios/rust/src/scenarios/lifecycle/conditional_launching.rs b/feature_integration_tests/test_scenarios/rust/src/scenarios/lifecycle/conditional_launching.rs new file mode 100644 index 00000000000..2481b027379 --- /dev/null +++ b/feature_integration_tests/test_scenarios/rust/src/scenarios/lifecycle/conditional_launching.rs @@ -0,0 +1,158 @@ +// ******************************************************************************* +// Copyright (c) 2026 Contributors to the Eclipse Foundation +// +// See the NOTICE file(s) distributed with this work for additional +// information regarding copyright ownership. +// +// This program and the accompanying materials are made available under the +// terms of the Apache License Version 2.0 which is available at +// +// +// SPDX-License-Identifier: Apache-2.0 +// ******************************************************************************* + +use serde_json::Value; +use std::fs; +use std::path::Path; +use std::time::{Duration, Instant}; +use test_scenarios_rust::scenario::Scenario; +use tracing::info; + +pub struct ConditionalLaunching; + +fn path_condition_met(path: &str) -> bool { + Path::new(path).exists() +} + +fn env_condition_met(name: &str) -> bool { + std::env::var_os(name).is_some() +} + +/// Best-effort check whether a process matching `process_name` is currently running, by +/// scanning /proc//comm (kernel-truncated to 15 chars) and /proc//cmdline (full +/// argv[0], which covers names `comm` truncates). +fn process_condition_met(process_name: &str) -> bool { + let Ok(entries) = fs::read_dir("/proc") else { + return false; + }; + + for entry in entries.flatten() { + let pid = entry.file_name(); + let Some(pid) = pid.to_str() else { continue }; + if !pid.chars().all(|c| c.is_ascii_digit()) { + continue; + } + + if let Ok(comm) = fs::read_to_string(entry.path().join("comm")) { + if comm.trim_end() == process_name { + return true; + } + } + + if let Ok(cmdline) = fs::read(entry.path().join("cmdline")) { + let argv0 = cmdline.split(|&b| b == 0).next().unwrap_or(&[]); + if let Ok(argv0) = std::str::from_utf8(argv0) { + // Compare the basename only: a raw suffix match on the full path would also + // accept e.g. "/usr/bin/oversleep" as satisfying process_name="sleep". + let basename = argv0.rsplit('/').next().unwrap_or(argv0); + if basename == process_name { + return true; + } + } + } + } + false +} + +impl Scenario for ConditionalLaunching { + fn name(&self) -> &str { + "conditional_launching" + } + + fn run(&self, input: &str) -> Result<(), String> { + let value: Value = serde_json::from_str(input).map_err(|error| format!("Parse error: {error}"))?; + let test = value + .get("test") + .ok_or_else(|| "Missing 'test' field in scenario input".to_string())?; + + let polling_interval = test.get("polling_interval_ms").and_then(Value::as_u64).unwrap_or(50); + let timeout = test.get("timeout_ms").and_then(Value::as_u64).unwrap_or(5000); + let conditions = test.get("wait_conditions").and_then(Value::as_array).ok_or_else(|| { + "Wait conditions were not provided: missing 'test.wait_conditions' in scenario input".to_string() + })?; + + if conditions.is_empty() { + return Err( + "Wait conditions were not provided: empty 'test.wait_conditions' in scenario input".to_string(), + ); + } + + info!("Testing conditional launching"); + + let conditions: Vec<&str> = conditions + .iter() + .map(|condition| { + condition + .as_str() + .ok_or_else(|| "Wait condition entries must be strings".to_string()) + }) + .collect::>()?; + + for condition in &conditions { + if !condition.starts_with("path:") && !condition.starts_with("env:") && !condition.starts_with("process:") { + return Err(format!("Unsupported wait condition prefix: {condition}")); + } + } + + info!("Polling interval: {polling_interval}ms"); + info!("Condition timeout: {timeout}ms"); + + let deadline = Instant::now() + Duration::from_millis(timeout); + let mut satisfied = vec![false; conditions.len()]; + + loop { + let mut all_satisfied = true; + for (index, condition) in conditions.iter().enumerate() { + if satisfied[index] { + continue; + } + + let met = if let Some(path) = condition.strip_prefix("path:") { + path_condition_met(path) + } else if let Some(name) = condition.strip_prefix("env:") { + env_condition_met(name) + } else { + process_condition_met(condition.strip_prefix("process:").expect("checked above")) + }; + + if met { + satisfied[index] = true; + info!("Condition satisfied: {condition}"); + } else { + all_satisfied = false; + } + } + + if all_satisfied { + break; + } + + if Instant::now() >= deadline { + let unmet = conditions + .iter() + .zip(&satisfied) + .filter(|(_, met)| !**met) + .map(|(condition, _)| *condition) + .collect::>() + .join(", "); + return Err(format!("Timed out after {timeout}ms waiting for condition(s): {unmet}")); + } + + std::thread::sleep(Duration::from_millis(polling_interval)); + } + + info!("All dependencies satisfied"); + + Ok(()) + } +} diff --git a/feature_integration_tests/test_scenarios/rust/src/scenarios/lifecycle/mod.rs b/feature_integration_tests/test_scenarios/rust/src/scenarios/lifecycle/mod.rs new file mode 100644 index 00000000000..2c180f72b89 --- /dev/null +++ b/feature_integration_tests/test_scenarios/rust/src/scenarios/lifecycle/mod.rs @@ -0,0 +1,25 @@ +// ******************************************************************************* +// Copyright (c) 2026 Contributors to the Eclipse Foundation +// +// See the NOTICE file(s) distributed with this work for additional +// information regarding copyright ownership. +// +// This program and the accompanying materials are made available under the +// terms of the Apache License Version 2.0 which is available at +// +// +// SPDX-License-Identifier: Apache-2.0 +// ******************************************************************************* + +mod conditional_launching; + +use conditional_launching::ConditionalLaunching; +use test_scenarios_rust::scenario::{ScenarioGroup, ScenarioGroupImpl}; + +pub fn lifecycle_group() -> Box { + Box::new(ScenarioGroupImpl::new( + "lifecycle", + vec![Box::new(ConditionalLaunching)], + vec![], + )) +} diff --git a/feature_integration_tests/test_scenarios/rust/src/scenarios/mod.rs b/feature_integration_tests/test_scenarios/rust/src/scenarios/mod.rs index 5c7013138f6..d4a774fe6be 100644 --- a/feature_integration_tests/test_scenarios/rust/src/scenarios/mod.rs +++ b/feature_integration_tests/test_scenarios/rust/src/scenarios/mod.rs @@ -12,10 +12,17 @@ // ******************************************************************************* use test_scenarios_rust::scenario::{ScenarioGroup, ScenarioGroupImpl}; + +mod lifecycle; mod persistency; +use lifecycle::lifecycle_group; use persistency::persistency_group; pub fn root_scenario_group() -> Box { - Box::new(ScenarioGroupImpl::new("root", vec![], vec![persistency_group()])) + Box::new(ScenarioGroupImpl::new( + "root", + vec![], + vec![lifecycle_group(), persistency_group()], + )) } diff --git a/patches/lifecycle/001-forward-visibility-to-config-combiner.patch b/patches/lifecycle/001-forward-visibility-to-config-combiner.patch new file mode 100644 index 00000000000..512d319bbe0 --- /dev/null +++ b/patches/lifecycle/001-forward-visibility-to-config-combiner.patch @@ -0,0 +1,10 @@ +diff --git a/scripts/config_mapping/config.bzl b/scripts/config_mapping/config.bzl +index 0a07e26d..ccdde084 100644 +--- a/scripts/config_mapping/config.bzl ++++ b/scripts/config_mapping/config.bzl +@@ -233,4 +233,5 @@ def launch_manager_config( + ], + }), + dir_name = flatbuffer_out_dir, ++ visibility = kwargs.get("visibility"), + ) From 8204a3358a24ce6174dc33cf67917e1a6181939b Mon Sep 17 00:00:00 2001 From: Saumya-R Date: Tue, 1 Sep 2026 23:09:50 +0530 Subject: [PATCH 2/9] formatting fixes --- .../test_scenarios/rust/src/scenarios/mod.rs | 1 - 1 file changed, 1 deletion(-) diff --git a/feature_integration_tests/test_scenarios/rust/src/scenarios/mod.rs b/feature_integration_tests/test_scenarios/rust/src/scenarios/mod.rs index d4a774fe6be..2ab4d9c956c 100644 --- a/feature_integration_tests/test_scenarios/rust/src/scenarios/mod.rs +++ b/feature_integration_tests/test_scenarios/rust/src/scenarios/mod.rs @@ -12,7 +12,6 @@ // ******************************************************************************* use test_scenarios_rust::scenario::{ScenarioGroup, ScenarioGroupImpl}; - mod lifecycle; mod persistency; From f7094707af9c15675d1ef313812f80bf7f11c45c Mon Sep 17 00:00:00 2001 From: Saumya-R Date: Wed, 16 Sep 2026 13:08:06 +0530 Subject: [PATCH 3/9] fixed the issue due to removal of config folder from launch manager --- feature_integration_tests/test_cases/BUILD | 52 ++++++++----------- .../test_cases/daemon_helpers.py | 50 ++++++++---------- .../support_apps/flaky_startup_app/BUILD | 2 +- .../test_process_launching_with_daemon.py | 4 +- 4 files changed, 47 insertions(+), 61 deletions(-) diff --git a/feature_integration_tests/test_cases/BUILD b/feature_integration_tests/test_cases/BUILD index 66e8c79e091..38950a062d5 100644 --- a/feature_integration_tests/test_cases/BUILD +++ b/feature_integration_tests/test_cases/BUILD @@ -161,29 +161,25 @@ score_py_pytest( "//feature_integration_tests/configs:lifecycle_daemon_config.json", "//feature_integration_tests/configs:lifecycle_daemon_parallel_launch_config.json", "@flatbuffers//:flatc", - "@score_lifecycle_health//examples/control_application:control_daemon", - "@score_lifecycle_health//examples/control_application:lmcontrol", - "@score_lifecycle_health//examples/cpp_supervised_app", - "@score_lifecycle_health//examples/rust_supervised_app", - "@score_lifecycle_health//score/launch_manager", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hm_flatcfg.fbs", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hmcore_flatcfg.fbs", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:lm_flatcfg.fbs", - "@score_lifecycle_health//scripts/config_mapping:lifecycle_config", + "@score_lifecycle//examples/control_application:control_daemon", + "@score_lifecycle//examples/control_application:lmcontrol", + "@score_lifecycle//examples/cpp_supervised_app", + "@score_lifecycle//examples/rust_supervised_app", + "@score_lifecycle//score/launch_manager", + "@score_lifecycle//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs", + "@score_lifecycle//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json", + "@score_lifecycle//scripts/config_mapping:lifecycle_config", ], env = { - "FIT_CPP_SUPERVISED_APP_PATH": "$(rootpath @score_lifecycle_health//examples/cpp_supervised_app)", - "FIT_LAUNCH_MANAGER_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager)", + "FIT_CPP_SUPERVISED_APP_PATH": "$(rootpath @score_lifecycle//examples/cpp_supervised_app)", + "FIT_LAUNCH_MANAGER_PATH": "$(rootpath @score_lifecycle//score/launch_manager)", "FIT_FLATC_PATH": "$(rootpath @flatbuffers//:flatc)", - "FIT_LIFECYCLE_CONFIG_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json)", - "FIT_LIFECYCLE_CONFIG_TOOL_PATH": "$(rootpath @score_lifecycle_health//scripts/config_mapping:lifecycle_config)", + "FIT_LIFECYCLE_CONFIG_SCHEMA_PATH": "$(rootpath @score_lifecycle//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json)", + "FIT_LIFECYCLE_CONFIG_TOOL_PATH": "$(rootpath @score_lifecycle//scripts/config_mapping:lifecycle_config)", "FIT_LIFECYCLE_DAEMON_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_config.json)", - "FIT_LIFECYCLE_HMCORE_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hmcore_flatcfg.fbs)", - "FIT_LIFECYCLE_HM_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hm_flatcfg.fbs)", - "FIT_LIFECYCLE_LM_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:lm_flatcfg.fbs)", + "FIT_LIFECYCLE_LM_SCHEMA_PATH": "$(rootpath @score_lifecycle//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs)", "FIT_LIFECYCLE_PARALLEL_LAUNCH_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_parallel_launch_config.json)", - "FIT_RUST_SUPERVISED_APP_PATH": "$(rootpath @score_lifecycle_health//examples/rust_supervised_app)", + "FIT_RUST_SUPERVISED_APP_PATH": "$(rootpath @score_lifecycle//examples/rust_supervised_app)", "RUST_BACKTRACE": "1", }, env_inherit = ["FIT_ENABLE_SETCAP"], @@ -216,22 +212,18 @@ score_py_pytest( "//feature_integration_tests/configs:lifecycle_daemon_retry_recovers_config.json", "//feature_integration_tests/test_cases/support_apps/flaky_startup_app", "@flatbuffers//:flatc", - "@score_lifecycle_health//score/launch_manager", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hm_flatcfg.fbs", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hmcore_flatcfg.fbs", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:lm_flatcfg.fbs", - "@score_lifecycle_health//scripts/config_mapping:lifecycle_config", + "@score_lifecycle//score/launch_manager", + "@score_lifecycle//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs", + "@score_lifecycle//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json", + "@score_lifecycle//scripts/config_mapping:lifecycle_config", ], env = { "FIT_FLAKY_STARTUP_APP_PATH": "$(rootpath //feature_integration_tests/test_cases/support_apps/flaky_startup_app)", "FIT_FLATC_PATH": "$(rootpath @flatbuffers//:flatc)", - "FIT_LAUNCH_MANAGER_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager)", - "FIT_LIFECYCLE_CONFIG_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json)", - "FIT_LIFECYCLE_CONFIG_TOOL_PATH": "$(rootpath @score_lifecycle_health//scripts/config_mapping:lifecycle_config)", - "FIT_LIFECYCLE_HMCORE_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hmcore_flatcfg.fbs)", - "FIT_LIFECYCLE_HM_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hm_flatcfg.fbs)", - "FIT_LIFECYCLE_LM_SCHEMA_PATH": "$(rootpath @score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:lm_flatcfg.fbs)", + "FIT_LAUNCH_MANAGER_PATH": "$(rootpath @score_lifecycle//score/launch_manager)", + "FIT_LIFECYCLE_CONFIG_SCHEMA_PATH": "$(rootpath @score_lifecycle//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json)", + "FIT_LIFECYCLE_CONFIG_TOOL_PATH": "$(rootpath @score_lifecycle//scripts/config_mapping:lifecycle_config)", + "FIT_LIFECYCLE_LM_SCHEMA_PATH": "$(rootpath @score_lifecycle//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs)", "FIT_LIFECYCLE_RETRY_EXHAUSTS_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_retry_exhausts_config.json)", "FIT_LIFECYCLE_RETRY_RECOVERS_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_retry_recovers_config.json)", }, diff --git a/feature_integration_tests/test_cases/daemon_helpers.py b/feature_integration_tests/test_cases/daemon_helpers.py index 2f409a51114..905ac06e814 100644 --- a/feature_integration_tests/test_cases/daemon_helpers.py +++ b/feature_integration_tests/test_cases/daemon_helpers.py @@ -49,9 +49,7 @@ ), "@score_lifecycle_health//scripts/config_mapping:lifecycle_config": "FIT_LIFECYCLE_CONFIG_TOOL_PATH", "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json": "FIT_LIFECYCLE_CONFIG_SCHEMA_PATH", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:lm_flatcfg.fbs": "FIT_LIFECYCLE_LM_SCHEMA_PATH", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hm_flatcfg.fbs": "FIT_LIFECYCLE_HM_SCHEMA_PATH", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hmcore_flatcfg.fbs": "FIT_LIFECYCLE_HMCORE_SCHEMA_PATH", + "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs": "FIT_LIFECYCLE_LM_SCHEMA_PATH", "@flatbuffers//:flatc": "FIT_FLATC_PATH", } @@ -361,32 +359,28 @@ def _generate_runtime_config(config_template: str, runtime_root: Path, etc_dir: ) flatc = _resolve_target_path("@flatbuffers//:flatc") - buffers = ( - ( - "lm_demo", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:lm_flatcfg.fbs", - ), - ("hm_demo", "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hm_flatcfg.fbs"), - ( - "hmcore", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/alive_monitor/config:hmcore_flatcfg.fbs", - ), + lm_schema = _resolve_target_path( + "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs" + ) + generated_config = generated_dir / f"{rendered_config.stem}_gen.json" + # launch_manager defaults to loading "etc/launch_manager_config.bin", and flatc names its + # output after the input file's stem, so the input must be named to match. + flatc_input = generated_dir / "launch_manager_config.json" + shutil.copy2(generated_config, flatc_input) + subprocess.run( + [ + str(flatc), + "--binary", + "--strict-json", + "-o", + str(etc_dir), + str(lm_schema), + str(flatc_input), + ], + capture_output=True, + text=True, + check=True, ) - for name, schema_target in buffers: - subprocess.run( - [ - str(flatc), - "--binary", - "--strict-json", - "-o", - str(etc_dir), - str(_resolve_target_path(schema_target)), - str(generated_dir / f"{name}.json"), - ], - capture_output=True, - text=True, - check=True, - ) def start_launch_manager_daemon( diff --git a/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD b/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD index e887c051c65..ffc6f541c13 100644 --- a/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD +++ b/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD @@ -20,6 +20,6 @@ cc_binary( srcs = ["main.cpp"], visibility = ["//feature_integration_tests:__subpackages__"], deps = [ - "@score_lifecycle_health//score/launch_manager:lifecycle_cc", + "@score_lifecycle//score/launch_manager:lifecycle_cc", ], ) diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py b/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py index e064597374e..a2d359c8e09 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py @@ -548,8 +548,8 @@ def test_watchdog_detection(self, launch_manager_daemon: dict[str, Any], version watchdog_patterns = [ rf"Got kRunning timeout for process.*\(\s*{re.escape(app_name)}\s*\)", rf"unexpected termination of process.*\(\s*{re.escape(app_name)}\s*\)", - rf"Alive Supervision \(\s*{re.escape(app_name)}_alive_supervision\s*\) switched to FAILED", - rf"Alive Supervision \(\s*{re.escape(app_name)}_alive_supervision\s*\) switched to EXPIRED", + rf"Alive Supervision \(\s*{re.escape(app_name)}\s*\) switched to FAILED", + rf"Alive Supervision \(\s*{re.escape(app_name)}\s*\) switched to EXPIRED", ] assert any(re.search(pattern, logs) for pattern in watchdog_patterns), ( f"No target-specific watchdog diagnostics found for {app_name}.\nDaemon logs:\n{logs}" From ff5090398c051791f0425bb7e89a2bae9393a2b5 Mon Sep 17 00:00:00 2001 From: Saumya-R Date: Mon, 21 Sep 2026 18:39:27 +0530 Subject: [PATCH 4/9] adding review comment fixes --- .../test_cases/daemon_helpers.py | 302 ++++++++++-------- .../test_cases/tests/basic/conftest.py | 33 -- 2 files changed, 163 insertions(+), 172 deletions(-) delete mode 100644 feature_integration_tests/test_cases/tests/basic/conftest.py diff --git a/feature_integration_tests/test_cases/daemon_helpers.py b/feature_integration_tests/test_cases/daemon_helpers.py index 905ac06e814..622aa691cd5 100644 --- a/feature_integration_tests/test_cases/daemon_helpers.py +++ b/feature_integration_tests/test_cases/daemon_helpers.py @@ -31,9 +31,9 @@ _TARGET_ENV_MAP = { - "@score_lifecycle_health//score/launch_manager:launch_manager": "FIT_LAUNCH_MANAGER_PATH", - "@score_lifecycle_health//examples/rust_supervised_app:rust_supervised_app": "FIT_RUST_SUPERVISED_APP_PATH", - "@score_lifecycle_health//examples/cpp_supervised_app:cpp_supervised_app": "FIT_CPP_SUPERVISED_APP_PATH", + "@score_lifecycle//score/launch_manager:launch_manager": "FIT_LAUNCH_MANAGER_PATH", + "@score_lifecycle//examples/rust_supervised_app:rust_supervised_app": "FIT_RUST_SUPERVISED_APP_PATH", + "@score_lifecycle//examples/cpp_supervised_app:cpp_supervised_app": "FIT_CPP_SUPERVISED_APP_PATH", "//feature_integration_tests/configs:lifecycle_daemon_config.json": "FIT_LIFECYCLE_DAEMON_CONFIG_PATH", "//feature_integration_tests/configs:lifecycle_daemon_parallel_launch_config.json": ( "FIT_LIFECYCLE_PARALLEL_LAUNCH_CONFIG_PATH" @@ -47,9 +47,9 @@ "//feature_integration_tests/configs:lifecycle_daemon_retry_exhausts_config.json": ( "FIT_LIFECYCLE_RETRY_EXHAUSTS_CONFIG_PATH" ), - "@score_lifecycle_health//scripts/config_mapping:lifecycle_config": "FIT_LIFECYCLE_CONFIG_TOOL_PATH", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json": "FIT_LIFECYCLE_CONFIG_SCHEMA_PATH", - "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs": "FIT_LIFECYCLE_LM_SCHEMA_PATH", + "@score_lifecycle//scripts/config_mapping:lifecycle_config": "FIT_LIFECYCLE_CONFIG_TOOL_PATH", + "@score_lifecycle//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json": "FIT_LIFECYCLE_CONFIG_SCHEMA_PATH", + "@score_lifecycle//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs": "FIT_LIFECYCLE_LM_SCHEMA_PATH", "@flatbuffers//:flatc": "FIT_FLATC_PATH", } @@ -64,8 +64,13 @@ def _run(cmd: list[str]) -> str: cwd=_repo_root(), capture_output=True, text=True, - check=True, + check=False, ) + if completed.returncode != 0: + raise RuntimeError( + f"Command failed (rc={completed.returncode}): {' '.join(cmd)}\n" + f"stdout:\n{completed.stdout}\nstderr:\n{completed.stderr}" + ) return completed.stdout.strip() @@ -98,26 +103,17 @@ def _resolve_from_env(target: str) -> Path | None: def _resolve_target_path(target: str) -> Path: - """Resolve an executable/file path from a bazel target label.""" + """Resolve an executable/file path from a bazel target label via its runfile env var.""" env_resolved = _resolve_from_env(target) if env_resolved is not None: return env_resolved - _run(["bazel", "build", target]) - output = _run(["bazel", "cquery", "--output=files", target]) - candidates = [line.strip() for line in output.splitlines() if line.strip()] - if not candidates: - raise RuntimeError(f"No files produced by target: {target}") - - execution_root = Path(_run(["bazel", "info", "execution_root"])) - for item in candidates: - candidate = Path(item) - if not candidate.is_absolute(): - candidate = execution_root / candidate - if candidate.exists(): - return candidate - - raise RuntimeError(f"No existing artifact found for target: {target}. Candidates: {candidates!r}") + env_var = _TARGET_ENV_MAP.get(target) + raise RuntimeError( + f"Could not resolve target {target!r}: environment variable " + f"{env_var!r} is not set or does not point to an existing file. " + "Ensure the corresponding data dependency is declared on the test target." + ) def get_binary_path(target: str) -> Path: @@ -318,6 +314,8 @@ def stop(self) -> None: if self.is_running(): os.killpg(os.getpgid(self.process.pid), signal.SIGKILL) self.process.wait(timeout=5) + if self.process.stdout is not None: + self.process.stdout.close() self._thread.join(timeout=1) def get_logs(self) -> str: @@ -347,41 +345,119 @@ def _generate_runtime_config(config_template: str, runtime_root: Path, etc_dir: rendered_config.write_text(json.dumps(config), encoding="utf-8") generated_dir = etc_dir / "generated" generated_dir.mkdir() - config_tool = _resolve_target_path("@score_lifecycle_health//scripts/config_mapping:lifecycle_config") + config_tool = _resolve_target_path("@score_lifecycle//scripts/config_mapping:lifecycle_config") config_schema = _resolve_target_path( - "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json" + "@score_lifecycle//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json" ) - subprocess.run( + config_mapping_result = subprocess.run( [str(config_tool), str(rendered_config), "--schema", str(config_schema), "-o", str(generated_dir)], capture_output=True, text=True, - check=True, + check=False, ) + if config_mapping_result.returncode != 0: + raise RuntimeError( + f"Command failed (rc={config_mapping_result.returncode}): {config_tool} {rendered_config} " + f"--schema {config_schema} -o {generated_dir}\n" + f"stdout:\n{config_mapping_result.stdout}\nstderr:\n{config_mapping_result.stderr}" + ) flatc = _resolve_target_path("@flatbuffers//:flatc") lm_schema = _resolve_target_path( - "@score_lifecycle_health//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs" + "@score_lifecycle//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs" ) generated_config = generated_dir / f"{rendered_config.stem}_gen.json" # launch_manager defaults to loading "etc/launch_manager_config.bin", and flatc names its # output after the input file's stem, so the input must be named to match. flatc_input = generated_dir / "launch_manager_config.json" shutil.copy2(generated_config, flatc_input) - subprocess.run( - [ - str(flatc), - "--binary", - "--strict-json", - "-o", - str(etc_dir), - str(lm_schema), - str(flatc_input), - ], + flatc_cmd = [ + str(flatc), + "--binary", + "--strict-json", + "-o", + str(etc_dir), + str(lm_schema), + str(flatc_input), + ] + flatc_result = subprocess.run( + flatc_cmd, capture_output=True, text=True, - check=True, + check=False, + ) + if flatc_result.returncode != 0: + raise RuntimeError( + f"Command failed (rc={flatc_result.returncode}): {' '.join(flatc_cmd)}\n" + f"stdout:\n{flatc_result.stdout}\nstderr:\n{flatc_result.stderr}" + ) + + +def _spawn_daemon( + work_dir: Path, + etc_dir: Path, + runtime_root: Path, + config_template: str, + staged_binaries: list[tuple[Path, Path, int]], + grant_sandbox_capabilities: bool = False, +) -> tuple[ManagedDaemon, bool, str]: + """Stage launch_manager plus `staged_binaries` (src, dst, mode), render its runtime + config, and start it as a supervised subprocess. + + Returns `(daemon, sandbox_privileged, sandbox_privileged_reason)`; the latter two are + `(False, "not requested")` unless `grant_sandbox_capabilities` is set. Fails the test via + `pytest.fail` if the daemon exits within the startup grace period. + """ + launch_manager = _resolve_target_path("@score_lifecycle//score/launch_manager:launch_manager") + lm_dst = work_dir / "launch_manager" + shutil.copy2(launch_manager, lm_dst) + lm_dst.chmod(0o755) + + if grant_sandbox_capabilities: + sandbox_privileged, sandbox_privileged_reason = _grant_sandbox_capabilities(lm_dst) + else: + sandbox_privileged, sandbox_privileged_reason = False, "not requested" + + for src, dst, mode in staged_binaries: + shutil.copy2(src, dst) + dst.chmod(mode) + + _generate_runtime_config(config_template, runtime_root, etc_dir) + + env = os.environ.copy() + env.setdefault("ECUCFG_ENV_VAR_ROOTFOLDER", str(etc_dir)) + + lines: list[str] = [] + process = subprocess.Popen( + [str(lm_dst)], + cwd=work_dir, + env=env, + stdout=subprocess.PIPE, + stderr=subprocess.STDOUT, + text=True, + start_new_session=True, ) + def _collect_output() -> None: + assert process.stdout is not None + for line in process.stdout: + line = line.rstrip("\n") + if line: + lines.append(line) + + thread = threading.Thread(target=_collect_output, daemon=True) + thread.start() + + daemon = ManagedDaemon(process=process, _lines=lines, _thread=thread) + + # Give startup a chance to complete and fail early if config is broken. + time.sleep(1.0) + if not daemon.is_running(): + logs = daemon.get_logs() + pytest.fail(f"launch_manager failed to start. Logs:\n{logs}") + + return daemon, sandbox_privileged, sandbox_privileged_reason + def start_launch_manager_daemon( tmp_path_factory: pytest.TempPathFactory, @@ -404,6 +480,7 @@ def start_launch_manager_daemon( """ runtime_root = Path(tempfile.mkdtemp(prefix="lifecycle_fit-", dir=_tmpdir_root())) + daemon = None try: work_dir = tmp_path_factory.mktemp("lm-daemon") etc_dir = work_dir / "etc" @@ -412,71 +489,43 @@ def start_launch_manager_daemon( bin_dir = runtime_root / "bin" bin_dir.mkdir(parents=True, exist_ok=True) - launch_manager = _resolve_target_path("@score_lifecycle_health//score/launch_manager:launch_manager") - rust_supervised = _resolve_target_path( - "@score_lifecycle_health//examples/rust_supervised_app:rust_supervised_app" - ) - cpp_supervised = _resolve_target_path("@score_lifecycle_health//examples/cpp_supervised_app:cpp_supervised_app") - - lm_dst = work_dir / "launch_manager" - shutil.copy2(launch_manager, lm_dst) - lm_dst.chmod(0o755) - sandbox_privileged, sandbox_privileged_reason = _grant_sandbox_capabilities(lm_dst) - - for key, src in (("rust", rust_supervised), ("cpp", cpp_supervised)): - dst = bin_dir / src.name - shutil.copy2(src, dst) - dst.chmod(0o000 if key in blocked_apps else 0o755) - - _generate_runtime_config(config_template, runtime_root, etc_dir) - - env = os.environ.copy() - env.setdefault("ECUCFG_ENV_VAR_ROOTFOLDER", str(etc_dir)) + rust_supervised = _resolve_target_path("@score_lifecycle//examples/rust_supervised_app:rust_supervised_app") + cpp_supervised = _resolve_target_path("@score_lifecycle//examples/cpp_supervised_app:cpp_supervised_app") + staged_binaries = [ + (src, bin_dir / src.name, 0o000 if key in blocked_apps else 0o755) + for key, src in (("rust", rust_supervised), ("cpp", cpp_supervised)) + ] - lines: list[str] = [] - process = subprocess.Popen( - [str(lm_dst)], - cwd=work_dir, - env=env, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - start_new_session=True, + daemon, sandbox_privileged, sandbox_privileged_reason = _spawn_daemon( + work_dir, + etc_dir, + runtime_root, + config_template, + staged_binaries, + grant_sandbox_capabilities=True, ) - def _collect_output() -> None: - assert process.stdout is not None - for line in process.stdout: - line = line.rstrip("\n") - if line: - lines.append(line) - - thread = threading.Thread(target=_collect_output, daemon=True) - thread.start() - - daemon = ManagedDaemon(process=process, _lines=lines, _thread=thread) - - # Give startup a chance to complete and fail early if config is broken. - time.sleep(1.0) - if not daemon.is_running(): - logs = daemon.get_logs() - pytest.skip(f"launch_manager failed to start in this environment. Logs:\n{logs}") - apps = { "rust": bin_dir / "rust_supervised_app", "cpp": bin_dir / "cpp_supervised_app", } if wait_for_apps and not _wait_for_apps({k: v for k, v in apps.items() if k not in blocked_apps}): - process_snapshot = _run(["ps", "-eo", "pid,args"]) - daemon.stop() - _cleanup_runtime_root(runtime_root) + process_snapshot = subprocess.run( + ["ps", "-eo", "pid,args"], + capture_output=True, + text=True, + check=False, + ) pytest.fail( "Launch Manager did not bring supervised apps to running state within timeout.\n" f"Expected apps: {apps}\n" f"Daemon logs:\n{daemon.get_logs()}\n" - f"Process snapshot:\n{process_snapshot}" + f"Process snapshot (rc={process_snapshot.returncode}):\n" + f"{process_snapshot.stdout}{process_snapshot.stderr}" ) except BaseException: + if daemon is not None: + daemon.stop() _cleanup_runtime_root(runtime_root) raise @@ -506,6 +555,7 @@ def start_flaky_retry_daemon( what the calling test is checking. """ runtime_root = Path(tempfile.mkdtemp(prefix="lifecycle_fit_retries-", dir=_tmpdir_root())) + daemon = None try: work_dir = tmp_path_factory.mktemp("lm-retry-daemon") etc_dir = work_dir / "etc" @@ -514,55 +564,20 @@ def start_flaky_retry_daemon( bin_dir = runtime_root / "bin" bin_dir.mkdir(parents=True, exist_ok=True) - launch_manager = _resolve_target_path("@score_lifecycle_health//score/launch_manager:launch_manager") flaky_app = _resolve_target_path( "//feature_integration_tests/test_cases/support_apps/flaky_startup_app:flaky_startup_app" ) - lm_dst = work_dir / "launch_manager" - shutil.copy2(launch_manager, lm_dst) - lm_dst.chmod(0o755) - app_dst = bin_dir / "flaky_startup_app" - shutil.copy2(flaky_app, app_dst) - app_dst.chmod(0o755) + staged_binaries = [(flaky_app, app_dst, 0o755)] counter_path = runtime_root / "flaky_startup_app.counter" if counter_path.exists(): counter_path.unlink() - _generate_runtime_config(config_template, runtime_root, etc_dir) - - env = os.environ.copy() - env.setdefault("ECUCFG_ENV_VAR_ROOTFOLDER", str(etc_dir)) - - lines: list[str] = [] - process = subprocess.Popen( - [str(lm_dst)], - cwd=work_dir, - env=env, - stdout=subprocess.PIPE, - stderr=subprocess.STDOUT, - text=True, - start_new_session=True, - ) - - def _collect_output() -> None: - assert process.stdout is not None - for line in process.stdout: - line = line.rstrip("\n") - if line: - lines.append(line) - - thread = threading.Thread(target=_collect_output, daemon=True) - thread.start() - - daemon = ManagedDaemon(process=process, _lines=lines, _thread=thread) - - time.sleep(1.0) - if not daemon.is_running(): - logs = daemon.get_logs() - pytest.skip(f"launch_manager failed to start in this environment. Logs:\n{logs}") + daemon, _, _ = _spawn_daemon(work_dir, etc_dir, runtime_root, config_template, staged_binaries) except BaseException: + if daemon is not None: + daemon.stop() _cleanup_runtime_root(runtime_root) raise @@ -577,16 +592,26 @@ def _collect_output() -> None: } +def _stop_daemon(daemon_info: dict[str, Any], app_paths: list[Path]) -> None: + """Stop `daemon_info["daemon"]`, pkill each of `app_paths` by cmdline, then clean up + its runtime root. Runs unconditionally even if stopping the daemon itself raises. + """ + try: + daemon_info["daemon"].stop() + finally: + for app_path in app_paths: + subprocess.run( + ["pkill", "-f", pgrep_cmdline_pattern(str(app_path))], + capture_output=True, + text=True, + check=False, + ) + _cleanup_runtime_root(daemon_info["runtime_root"]) + + def stop_flaky_retry_daemon(daemon_info: dict[str, Any]) -> None: """Tear down a daemon started by `start_flaky_retry_daemon`.""" - daemon_info["daemon"].stop() - subprocess.run( - ["pkill", "-f", pgrep_cmdline_pattern(str(daemon_info["app_path"]))], - capture_output=True, - text=True, - check=False, - ) - _cleanup_runtime_root(daemon_info["runtime_root"]) + _stop_daemon(daemon_info, [daemon_info["app_path"]]) def read_retry_attempt_count(counter_path: Path) -> int: @@ -599,8 +624,7 @@ def read_retry_attempt_count(counter_path: Path) -> int: def stop_launch_manager_daemon(daemon_info: dict[str, Any]) -> None: """Tear down a daemon started by `start_launch_manager_daemon`.""" - daemon_info["daemon"].stop() - _cleanup_runtime_root(daemon_info["runtime_root"]) + _stop_daemon(daemon_info, list(daemon_info["apps"].values())) @pytest.fixture(scope="class") diff --git a/feature_integration_tests/test_cases/tests/basic/conftest.py b/feature_integration_tests/test_cases/tests/basic/conftest.py deleted file mode 100644 index da0b01b28c9..00000000000 --- a/feature_integration_tests/test_cases/tests/basic/conftest.py +++ /dev/null @@ -1,33 +0,0 @@ -# ******************************************************************************* -# Copyright (c) 2026 Contributors to the Eclipse Foundation -# -# See the NOTICE file(s) distributed with this work for additional -# information regarding copyright ownership. -# -# This program and the accompanying materials are made available under the -# terms of the Apache License Version 2.0 which is available at -# https://www.apache.org/licenses/LICENSE-2.0 -# -# SPDX-License-Identifier: Apache-2.0 -# ******************************************************************************* -import pytest - - -def pytest_sessionfinish(session: pytest.Session, exitstatus: int) -> None: - """Tolerate an empty ``-m cpp`` selection under tests/basic/ instead of failing the build. - - fit_cpp_orch runs ``-m cpp`` here, but every test under tests/basic/ is currently - @pytest.mark.rust-only, so today's selection is legitimately empty. Without this, - pytest's NO_TESTS_COLLECTED exit code would fail fit_cpp_orch permanently until a cpp - test exists. Once a cpp-marked test is added under tests/basic/, testscollected > 0 - and this hook no longer applies - the target then runs (and can fail) normally. - """ - if exitstatus == pytest.ExitCode.NO_TESTS_COLLECTED and session.testscollected == 0: - markexpr = session.config.getoption("markexpr", "") - if "cpp" in markexpr: - print( - "fit_cpp_orch: no @pytest.mark.cpp tests exist under tests/basic/ yet - " - "treating the empty selection as a pass. Add one and this target will " - "start actually running it." - ) - session.exitstatus = pytest.ExitCode.OK From f2d6db4292db1ef01a2a14458043f7f6bbe9b444 Mon Sep 17 00:00:00 2001 From: Saumya-R Date: Tue, 29 Sep 2026 22:41:14 +0530 Subject: [PATCH 5/9] addressing review comments --- feature_integration_tests/configs/BUILD | 4 +- .../configs/lifecycle_daemon_config.json | 12 ++ ...fecycle_daemon_parallel_launch_config.json | 100 ----------- ...son => lifecycle_daemon_retry_config.json} | 2 +- ...ifecycle_daemon_retry_recovers_config.json | 76 --------- feature_integration_tests/test_cases/BUILD | 16 +- .../test_cases/daemon_helpers.py | 156 +++++++++++++----- .../test_cases/lifecycle_scenario.py | 29 +++- .../support_apps/flaky_startup_app/BUILD | 2 +- .../test_cases/tests/lifecycle/conftest.py | 30 ++++ .../lifecycle/test_conditional_launching.py | 5 +- .../test_process_launching_with_daemon.py | 153 ++++++++++++----- .../tests/lifecycle/test_retry_exhaustion.py | 55 ++---- .../internals/persistency/kvs_build_helpers.h | 70 +------- ...orward-visibility-to-config-combiner.patch | 10 -- patches/lifecycle/BUILD | 0 16 files changed, 336 insertions(+), 384 deletions(-) delete mode 100644 feature_integration_tests/configs/lifecycle_daemon_parallel_launch_config.json rename feature_integration_tests/configs/{lifecycle_daemon_retry_exhausts_config.json => lifecycle_daemon_retry_config.json} (97%) delete mode 100644 feature_integration_tests/configs/lifecycle_daemon_retry_recovers_config.json create mode 100644 feature_integration_tests/test_cases/tests/lifecycle/conftest.py delete mode 100644 patches/lifecycle/001-forward-visibility-to-config-combiner.patch delete mode 100644 patches/lifecycle/BUILD diff --git a/feature_integration_tests/configs/BUILD b/feature_integration_tests/configs/BUILD index b81d0f5bb4f..1bdd56eca30 100644 --- a/feature_integration_tests/configs/BUILD +++ b/feature_integration_tests/configs/BUILD @@ -16,9 +16,7 @@ exports_files( "dlt_config_x86_64.json", "qemu_bridge_config.json", "lifecycle_daemon_config.json", - "lifecycle_daemon_parallel_launch_config.json", - "lifecycle_daemon_retry_recovers_config.json", - "lifecycle_daemon_retry_exhausts_config.json", + "lifecycle_daemon_retry_config.json", ], ) diff --git a/feature_integration_tests/configs/lifecycle_daemon_config.json b/feature_integration_tests/configs/lifecycle_daemon_config.json index b8760188bc5..5bb3e10f811 100644 --- a/feature_integration_tests/configs/lifecycle_daemon_config.json +++ b/feature_integration_tests/configs/lifecycle_daemon_config.json @@ -53,6 +53,12 @@ "environmental_variables": { "PROCESSIDENTIFIER": "cpp_supervised_app", "IDENTIFIER": "cpp_supervised_app" + }, + "sandbox": { + "uid": 1001, + "gid": 1001, + "scheduling_policy": "SCHED_FIFO", + "scheduling_priority": 20 } } }, @@ -73,6 +79,12 @@ "environmental_variables": { "PROCESSIDENTIFIER": "rust_supervised_app", "IDENTIFIER": "rust_supervised_app" + }, + "sandbox": { + "uid": 1001, + "gid": 1001, + "scheduling_policy": "SCHED_RR", + "scheduling_priority": 10 } } } diff --git a/feature_integration_tests/configs/lifecycle_daemon_parallel_launch_config.json b/feature_integration_tests/configs/lifecycle_daemon_parallel_launch_config.json deleted file mode 100644 index c2aca25e498..00000000000 --- a/feature_integration_tests/configs/lifecycle_daemon_parallel_launch_config.json +++ /dev/null @@ -1,100 +0,0 @@ -{ - "schema_version": 1, - "defaults": { - "deployment_config": { - "bin_dir": "__FIT_RUNTIME_ROOT__/bin", - "ready_timeout": 2.0, - "shutdown_timeout": 2.0, - "ready_recovery_action": { - "restart": { - "number_of_attempts": 2 - } - }, - "recovery_action": { - "switch_run_target": { - "run_target": "fallback_run_target" - } - }, - "sandbox": { - "uid": 1001, - "gid": 1001, - "scheduling_policy": "SCHED_OTHER", - "scheduling_priority": 0 - } - }, - "component_properties": { - "application_profile": { - "application_type": "Reporting", - "is_self_terminating": false, - "alive_supervision": { - "reporting_cycle": 0.1, - "min_indications": 1, - "max_indications": 3, - "failed_cycles_tolerance": 1 - } - }, - "ready_condition": { - "process_state": "Running" - } - } - }, - "components": { - "cpp_supervised_app": { - "component_properties": { - "binary_name": "cpp_supervised_app", - "application_profile": { - "application_type": "Reporting_And_Supervised" - }, - "process_arguments": [ - "-d50" - ] - }, - "deployment_config": { - "environmental_variables": { - "PROCESSIDENTIFIER": "cpp_supervised_app", - "IDENTIFIER": "cpp_supervised_app" - } - } - }, - "rust_supervised_app": { - "component_properties": { - "binary_name": "rust_supervised_app", - "application_profile": { - "application_type": "Reporting_And_Supervised" - }, - "process_arguments": [ - "-d50" - ] - }, - "deployment_config": { - "environmental_variables": { - "PROCESSIDENTIFIER": "rust_supervised_app", - "IDENTIFIER": "rust_supervised_app" - } - } - } - }, - "run_targets": { - "Startup": { - "depends_on": [ - "cpp_supervised_app", - "rust_supervised_app" - ], - "recovery_action": { - "switch_run_target": { - "run_target": "fallback_run_target" - } - } - } - }, - "initial_run_target": "Startup", - "alive_supervision": { - "evaluation_cycle": 0.05 - }, - "fallback_run_target": { - "depends_on": [ - "cpp_supervised_app", - "rust_supervised_app" - ] - } -} diff --git a/feature_integration_tests/configs/lifecycle_daemon_retry_exhausts_config.json b/feature_integration_tests/configs/lifecycle_daemon_retry_config.json similarity index 97% rename from feature_integration_tests/configs/lifecycle_daemon_retry_exhausts_config.json rename to feature_integration_tests/configs/lifecycle_daemon_retry_config.json index e161ed7f234..c0e41057f66 100644 --- a/feature_integration_tests/configs/lifecycle_daemon_retry_exhausts_config.json +++ b/feature_integration_tests/configs/lifecycle_daemon_retry_config.json @@ -44,7 +44,7 @@ "binary_name": "flaky_startup_app", "process_arguments": [ "__FIT_RUNTIME_ROOT__/flaky_startup_app.counter", - "999" + "__FIT_CRASHES_BEFORE_SUCCESS__" ] }, "deployment_config": { diff --git a/feature_integration_tests/configs/lifecycle_daemon_retry_recovers_config.json b/feature_integration_tests/configs/lifecycle_daemon_retry_recovers_config.json deleted file mode 100644 index 207e84c667d..00000000000 --- a/feature_integration_tests/configs/lifecycle_daemon_retry_recovers_config.json +++ /dev/null @@ -1,76 +0,0 @@ -{ - "schema_version": 1, - "defaults": { - "deployment_config": { - "bin_dir": "__FIT_RUNTIME_ROOT__/bin", - "ready_timeout": 2.0, - "shutdown_timeout": 2.0, - "ready_recovery_action": { - "restart": { - "number_of_attempts": 0 - } - }, - "recovery_action": { - "switch_run_target": { - "run_target": "fallback_run_target" - } - }, - "sandbox": { - "uid": 1001, - "gid": 1001, - "scheduling_policy": "SCHED_OTHER", - "scheduling_priority": 0 - } - }, - "component_properties": { - "application_profile": { - "application_type": "Reporting", - "is_self_terminating": false, - "alive_supervision": { - "reporting_cycle": 0.1, - "min_indications": 1, - "max_indications": 3, - "failed_cycles_tolerance": 1 - } - }, - "ready_condition": { - "process_state": "Running" - } - } - }, - "components": { - "flaky_startup_app": { - "component_properties": { - "binary_name": "flaky_startup_app", - "process_arguments": [ - "__FIT_RUNTIME_ROOT__/flaky_startup_app.counter", - "2" - ] - }, - "deployment_config": { - "ready_recovery_action": { - "restart": { - "number_of_attempts": 2 - } - }, - "environmental_variables": { - "PROCESSIDENTIFIER": "flaky_startup_app" - } - } - } - }, - "run_targets": { - "Startup": { - "depends_on": [ - "flaky_startup_app" - ] - } - }, - "initial_run_target": "Startup", - "alive_supervision": { - "evaluation_cycle": 0.05 - }, - "fallback_run_target": { - "depends_on": [] - } -} diff --git a/feature_integration_tests/test_cases/BUILD b/feature_integration_tests/test_cases/BUILD index 38950a062d5..079f90ec6ed 100644 --- a/feature_integration_tests/test_cases/BUILD +++ b/feature_integration_tests/test_cases/BUILD @@ -158,8 +158,8 @@ score_py_pytest( "conftest.py", "daemon_helpers.py", "test_properties.py", + "tests/lifecycle/conftest.py", "//feature_integration_tests/configs:lifecycle_daemon_config.json", - "//feature_integration_tests/configs:lifecycle_daemon_parallel_launch_config.json", "@flatbuffers//:flatc", "@score_lifecycle//examples/control_application:control_daemon", "@score_lifecycle//examples/control_application:lmcontrol", @@ -178,7 +178,6 @@ score_py_pytest( "FIT_LIFECYCLE_CONFIG_TOOL_PATH": "$(rootpath @score_lifecycle//scripts/config_mapping:lifecycle_config)", "FIT_LIFECYCLE_DAEMON_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_config.json)", "FIT_LIFECYCLE_LM_SCHEMA_PATH": "$(rootpath @score_lifecycle//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs)", - "FIT_LIFECYCLE_PARALLEL_LAUNCH_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_parallel_launch_config.json)", "FIT_RUST_SUPERVISED_APP_PATH": "$(rootpath @score_lifecycle//examples/rust_supervised_app)", "RUST_BACKTRACE": "1", }, @@ -189,6 +188,9 @@ score_py_pytest( # skip themselves accordingly (see daemon_helpers._grant_sandbox_capabilities). Everything # else only signals same-uid processes and needs no privilege escalation. FIT_ENABLE_SETCAP=1 # is for local/manual runs outside the sandbox where the sestcap grant can actually take effect. + # "exclusive": launch_manager uses fixed POSIX shm names on the host-wide /dev/shm, so it must + # not run concurrently with another launch_manager-driven test target. + tags = ["exclusive"], deps = all_requirements, ) @@ -207,9 +209,10 @@ score_py_pytest( data = [ "conftest.py", "daemon_helpers.py", + "fit_scenario.py", + "lifecycle_scenario.py", "test_properties.py", - "//feature_integration_tests/configs:lifecycle_daemon_retry_exhausts_config.json", - "//feature_integration_tests/configs:lifecycle_daemon_retry_recovers_config.json", + "//feature_integration_tests/configs:lifecycle_daemon_retry_config.json", "//feature_integration_tests/test_cases/support_apps/flaky_startup_app", "@flatbuffers//:flatc", "@score_lifecycle//score/launch_manager", @@ -224,10 +227,11 @@ score_py_pytest( "FIT_LIFECYCLE_CONFIG_SCHEMA_PATH": "$(rootpath @score_lifecycle//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json)", "FIT_LIFECYCLE_CONFIG_TOOL_PATH": "$(rootpath @score_lifecycle//scripts/config_mapping:lifecycle_config)", "FIT_LIFECYCLE_LM_SCHEMA_PATH": "$(rootpath @score_lifecycle//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs)", - "FIT_LIFECYCLE_RETRY_EXHAUSTS_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_retry_exhausts_config.json)", - "FIT_LIFECYCLE_RETRY_RECOVERS_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_retry_recovers_config.json)", + "FIT_LIFECYCLE_RETRY_CONFIG_PATH": "$(rootpath //feature_integration_tests/configs:lifecycle_daemon_retry_config.json)", }, pytest_config = "//:pyproject.toml", + # See fit_lifecycle_daemon: launch_manager's fixed /dev/shm names forbid concurrent targets. + tags = ["exclusive"], deps = all_requirements, ) diff --git a/feature_integration_tests/test_cases/daemon_helpers.py b/feature_integration_tests/test_cases/daemon_helpers.py index 622aa691cd5..e03863d9aa8 100644 --- a/feature_integration_tests/test_cases/daemon_helpers.py +++ b/feature_integration_tests/test_cases/daemon_helpers.py @@ -35,18 +35,10 @@ "@score_lifecycle//examples/rust_supervised_app:rust_supervised_app": "FIT_RUST_SUPERVISED_APP_PATH", "@score_lifecycle//examples/cpp_supervised_app:cpp_supervised_app": "FIT_CPP_SUPERVISED_APP_PATH", "//feature_integration_tests/configs:lifecycle_daemon_config.json": "FIT_LIFECYCLE_DAEMON_CONFIG_PATH", - "//feature_integration_tests/configs:lifecycle_daemon_parallel_launch_config.json": ( - "FIT_LIFECYCLE_PARALLEL_LAUNCH_CONFIG_PATH" - ), "//feature_integration_tests/test_cases/support_apps/flaky_startup_app:flaky_startup_app": ( "FIT_FLAKY_STARTUP_APP_PATH" ), - "//feature_integration_tests/configs:lifecycle_daemon_retry_recovers_config.json": ( - "FIT_LIFECYCLE_RETRY_RECOVERS_CONFIG_PATH" - ), - "//feature_integration_tests/configs:lifecycle_daemon_retry_exhausts_config.json": ( - "FIT_LIFECYCLE_RETRY_EXHAUSTS_CONFIG_PATH" - ), + "//feature_integration_tests/configs:lifecycle_daemon_retry_config.json": "FIT_LIFECYCLE_RETRY_CONFIG_PATH", "@score_lifecycle//scripts/config_mapping:lifecycle_config": "FIT_LIFECYCLE_CONFIG_TOOL_PATH", "@score_lifecycle//score/launch_manager/src/daemon/src/configuration/config_schema:launch_manager.schema.json": "FIT_LIFECYCLE_CONFIG_SCHEMA_PATH", "@score_lifecycle//score/launch_manager/src/daemon/src/configuration:lm_flatcfg_fbs": "FIT_LIFECYCLE_LM_SCHEMA_PATH", @@ -314,9 +306,12 @@ def stop(self) -> None: if self.is_running(): os.killpg(os.getpgid(self.process.pid), signal.SIGKILL) self.process.wait(timeout=5) - if self.process.stdout is not None: - self.process.stdout.close() + # Launched apps inherit the daemon's stdout pipe and can outlive it, so the reader may + # still be blocked in read() holding the buffer lock; close() would then deadlock. + # Only close once the reader has seen EOF (callers pkill the apps afterwards). self._thread.join(timeout=1) + if self.process.stdout is not None and not self._thread.is_alive(): + self.process.stdout.close() def get_logs(self) -> str: return "\n".join(self._lines) @@ -327,19 +322,56 @@ def _cleanup_runtime_root(runtime_root: Path) -> None: shutil.rmtree(runtime_root, ignore_errors=True) -def _generate_runtime_config(config_template: str, runtime_root: Path, etc_dir: Path) -> None: - """Render and serialize an isolated launch-manager config for one daemon.""" +def _generate_runtime_config( + config_template: str, + runtime_root: Path, + etc_dir: Path, + sandbox_privileged: bool = True, + independent_apps: bool = False, + ready_timeout_s: float | None = None, + crashes_before_success: int | None = None, +) -> None: + """Render and serialize an isolated launch-manager config for one daemon. + + Test variants are rendered from one template here rather than kept as copied config + files, so they cannot drift from the content other tests assert on: + `independent_apps` drops every component's `depends_on`, `ready_timeout_s` overrides + `defaults.deployment_config.ready_timeout`, and `crashes_before_success` fills the + `__FIT_CRASHES_BEFORE_SUCCESS__` process argument. + + Non-`SCHED_OTHER` scheduling policies need `CAP_SYS_NICE`; unlike the uid/gid sandbox + fields, launch_manager treats a failed `sched_setscheduler()` as fatal for the component + (triggering its `recovery_action` instead of just leaving the policy unapplied), which + would otherwise crash-loop every component in the config - not just the one a given test + cares about - whenever the capability grant wasn't obtained. When `sandbox_privileged` is + `False`, downgrade any configured non-default scheduling policy back to the harmless + `SCHED_OTHER`/`0` so the daemon still starts cleanly; scheduling-specific tests already key + off `sandbox_privileged` themselves and skip rather than assert against it in that case. + """ config = json.loads(_resolve_target_path(config_template).read_text(encoding="utf-8")) config["defaults"]["deployment_config"]["bin_dir"] = str(runtime_root / "bin") - + if ready_timeout_s is not None: + config["defaults"]["deployment_config"]["ready_timeout"] = ready_timeout_s + if independent_apps: + for component in config["components"].values(): + component["component_properties"].pop("depends_on", None) + + if not sandbox_privileged: + sandboxes = [config["defaults"]["deployment_config"].get("sandbox")] + sandboxes += [ + component.get("deployment_config", {}).get("sandbox") for component in config["components"].values() + ] + for sandbox in sandboxes: + if sandbox and sandbox.get("scheduling_policy") not in (None, "SCHED_OTHER"): + sandbox["scheduling_policy"] = "SCHED_OTHER" + sandbox["scheduling_priority"] = 0 + + placeholders = {"__FIT_RUNTIME_ROOT__/flaky_startup_app.counter": str(runtime_root / "flaky_startup_app.counter")} + if crashes_before_success is not None: + placeholders["__FIT_CRASHES_BEFORE_SUCCESS__"] = str(crashes_before_success) for component in config["components"].values(): arguments = component["component_properties"].get("process_arguments", []) - component["component_properties"]["process_arguments"] = [ - str(runtime_root / "flaky_startup_app.counter") - if argument == "__FIT_RUNTIME_ROOT__/flaky_startup_app.counter" - else argument - for argument in arguments - ] + component["component_properties"]["process_arguments"] = [placeholders.get(a, a) for a in arguments] rendered_config = etc_dir / "lifecycle_config.json" rendered_config.write_text(json.dumps(config), encoding="utf-8") @@ -393,6 +425,25 @@ def _generate_runtime_config(config_template: str, runtime_root: Path, etc_dir: ) +# launch_manager creates POSIX shm objects with deterministic names ("/ipc_shared_mem", +# "/_nudge~._.~me_") using O_CREAT|O_EXCL. POSIX shm lives on the host-wide /dev/shm tmpfs +# (not isolated by `unshare -i`), so a second daemon started before the first has shm_unlink'ed +# gets EEXIST and silently never launches its components. Daemon lifetimes must not overlap: +# within a process this registry enforces it; across Bazel test processes the lifecycle +# targets are tagged "exclusive". +_live_daemons: list[ManagedDaemon] = [] + + +def _assert_no_live_daemon() -> None: + _live_daemons[:] = [d for d in _live_daemons if d.is_running()] + if _live_daemons: + pytest.fail( + f"Another launch_manager (pid={_live_daemons[0].pid()}) is still running; overlapping " + "daemon lifetimes collide on launch_manager's fixed POSIX shm names. Don't start a " + "daemon from a test that also holds the class-scoped `launch_manager_daemon` fixture." + ) + + def _spawn_daemon( work_dir: Path, etc_dir: Path, @@ -400,14 +451,18 @@ def _spawn_daemon( config_template: str, staged_binaries: list[tuple[Path, Path, int]], grant_sandbox_capabilities: bool = False, + **config_options: Any, ) -> tuple[ManagedDaemon, bool, str]: """Stage launch_manager plus `staged_binaries` (src, dst, mode), render its runtime - config, and start it as a supervised subprocess. + config (`config_options` are passed to `_generate_runtime_config`), and start it as a + supervised subprocess. Returns `(daemon, sandbox_privileged, sandbox_privileged_reason)`; the latter two are `(False, "not requested")` unless `grant_sandbox_capabilities` is set. Fails the test via - `pytest.fail` if the daemon exits within the startup grace period. + `pytest.fail` if the daemon exits within the startup grace period, or if another daemon + started by this process is still alive (see `_live_daemons`). """ + _assert_no_live_daemon() launch_manager = _resolve_target_path("@score_lifecycle//score/launch_manager:launch_manager") lm_dst = work_dir / "launch_manager" shutil.copy2(launch_manager, lm_dst) @@ -422,7 +477,9 @@ def _spawn_daemon( shutil.copy2(src, dst) dst.chmod(mode) - _generate_runtime_config(config_template, runtime_root, etc_dir) + _generate_runtime_config( + config_template, runtime_root, etc_dir, sandbox_privileged=sandbox_privileged, **config_options + ) env = os.environ.copy() env.setdefault("ECUCFG_ENV_VAR_ROOTFOLDER", str(etc_dir)) @@ -449,6 +506,7 @@ def _collect_output() -> None: thread.start() daemon = ManagedDaemon(process=process, _lines=lines, _thread=thread) + _live_daemons.append(daemon) # Give startup a chance to complete and fail early if config is broken. time.sleep(1.0) @@ -462,8 +520,10 @@ def _collect_output() -> None: def start_launch_manager_daemon( tmp_path_factory: pytest.TempPathFactory, blocked_apps: frozenset[str] = frozenset(), + stalled_apps: frozenset[str] = frozenset(), wait_for_apps: bool = True, - config_template: str = "//feature_integration_tests/configs:lifecycle_daemon_config.json", + independent_apps: bool = False, + ready_timeout_s: float | None = None, ) -> dict[str, Any]: """Start a real launch_manager process with generated flatbuffer config. @@ -472,11 +532,17 @@ def start_launch_manager_daemon( chmod's them back to 0o755. Used to exercise the dependency-gating negative path: assert the dependent app stays down while its dependency is withheld, then unblock and assert it starts - and, with - an independent config (no depends_on between the two apps), the inverse: + `independent_apps=True` (no depends_on between the two apps), the inverse: assert the other app starts anyway, proving it isn't gated at all. + `independent_apps`/`ready_timeout_s` are rendered onto lifecycle_daemon_config.json + (see `_generate_runtime_config`). + + `stalled_apps` names are replaced by a shell stub that runs but never reports + Running, so launch_manager spends the full `ready_timeout` (and retries) on + them - unlike a blocked app, whose exec fails immediately. - Each invocation receives its own directory beneath `TEST_TMPDIR`, so it can - run concurrently with the class-scoped fixture or another Bazel test process. + Each invocation receives its own directory beneath `TEST_TMPDIR`, but must not overlap + another live daemon (including the class-scoped fixture): see `_live_daemons`. """ runtime_root = Path(tempfile.mkdtemp(prefix="lifecycle_fit-", dir=_tmpdir_root())) @@ -491,8 +557,16 @@ def start_launch_manager_daemon( rust_supervised = _resolve_target_path("@score_lifecycle//examples/rust_supervised_app:rust_supervised_app") cpp_supervised = _resolve_target_path("@score_lifecycle//examples/cpp_supervised_app:cpp_supervised_app") + stall_stub = work_dir / "stall_stub.sh" + # `exec -a "$0"` keeps the staged app path as argv[0], so is_running()/pkill's anchored + # cmdline pattern still matches the stub; a single exec'd process also dies on SIGTERM. + stall_stub.write_text('#!/bin/bash\nexec -a "$0" sleep infinity\n', encoding="utf-8") staged_binaries = [ - (src, bin_dir / src.name, 0o000 if key in blocked_apps else 0o755) + ( + stall_stub if key in stalled_apps else src, + bin_dir / src.name, + 0o000 if key in blocked_apps else 0o755, + ) for key, src in (("rust", rust_supervised), ("cpp", cpp_supervised)) ] @@ -500,9 +574,11 @@ def start_launch_manager_daemon( work_dir, etc_dir, runtime_root, - config_template, + "//feature_integration_tests/configs:lifecycle_daemon_config.json", staged_binaries, grant_sandbox_capabilities=True, + independent_apps=independent_apps, + ready_timeout_s=ready_timeout_s, ) apps = { @@ -542,10 +618,9 @@ def start_launch_manager_daemon( def start_flaky_retry_daemon( tmp_path_factory: pytest.TempPathFactory, - config_template: str, crashes_before_success: int, ) -> dict[str, Any]: - """Start launch_manager against a single-component retry config. + """Start launch_manager against the single-component lifecycle_daemon_retry_config.json. Drives `flaky_startup_app` (see support_apps/flaky_startup_app/main.cpp), which aborts on its first `crashes_before_success` startup attempts and stays running @@ -574,7 +649,14 @@ def start_flaky_retry_daemon( if counter_path.exists(): counter_path.unlink() - daemon, _, _ = _spawn_daemon(work_dir, etc_dir, runtime_root, config_template, staged_binaries) + daemon, _, _ = _spawn_daemon( + work_dir, + etc_dir, + runtime_root, + "//feature_integration_tests/configs:lifecycle_daemon_retry_config.json", + staged_binaries, + crashes_before_success=crashes_before_success, + ) except BaseException: if daemon is not None: daemon.stop() @@ -625,13 +707,3 @@ def read_retry_attempt_count(counter_path: Path) -> int: def stop_launch_manager_daemon(daemon_info: dict[str, Any]) -> None: """Tear down a daemon started by `start_launch_manager_daemon`.""" _stop_daemon(daemon_info, list(daemon_info["apps"].values())) - - -@pytest.fixture(scope="class") -def launch_manager_daemon(tmp_path_factory: pytest.TempPathFactory) -> dict[str, Any]: - """Start a real launch_manager process with generated flatbuffer config.""" - daemon_info = start_launch_manager_daemon(tmp_path_factory) - try: - yield daemon_info - finally: - stop_launch_manager_daemon(daemon_info) diff --git a/feature_integration_tests/test_cases/lifecycle_scenario.py b/feature_integration_tests/test_cases/lifecycle_scenario.py index 29753b0efc5..c4daf75b013 100644 --- a/feature_integration_tests/test_cases/lifecycle_scenario.py +++ b/feature_integration_tests/test_cases/lifecycle_scenario.py @@ -11,14 +11,19 @@ # SPDX-License-Identifier: Apache-2.0 # ******************************************************************************* """ -Helpers and base scenario class for lifecycle feature integration tests. +Helpers and base scenario classes for lifecycle feature integration tests. ``LifecycleScenario`` is a ``FitScenario`` subclass that supplies the shared ``temp_dir`` fixture so individual test classes do not have to duplicate it. + +``RetryDaemonScenario`` provides the equivalent class-scoped-fixture convention +for the flaky-retry daemon tests, which don't fit ``FitScenario`` (no scenario +binary, `command`, or `version` parametrization involved). """ from collections.abc import Generator from pathlib import Path +from typing import Any import pytest from fit_scenario import FitScenario, temp_dir_common @@ -48,3 +53,25 @@ def temp_dir( Parametrized scenario version (``"rust"`` or ``"cpp"``). """ yield from temp_dir_common(tmp_path_factory, self.__class__.__name__, version) + + +class RetryDaemonScenario: + """ + Base class for flaky-retry launch_manager daemon lifecycle tests. + + Subclasses set ``crashes_before_success``; the + ``retry_daemon`` fixture starts one launch_manager instance per test class + against `flaky_startup_app` and tears it down afterwards. + """ + + crashes_before_success: int + + @pytest.fixture(scope="class") + def retry_daemon(self, tmp_path_factory: pytest.TempPathFactory) -> Generator[dict[str, Any], None, None]: + from daemon_helpers import start_flaky_retry_daemon, stop_flaky_retry_daemon + + daemon_info = start_flaky_retry_daemon(tmp_path_factory, self.crashes_before_success) + try: + yield daemon_info + finally: + stop_flaky_retry_daemon(daemon_info) diff --git a/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD b/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD index ffc6f541c13..0eda8fe1688 100644 --- a/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD +++ b/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD @@ -14,7 +14,7 @@ load("@rules_cc//cc:defs.bzl", "cc_binary") # Plain "Native" supervised app (no launch_manager lifecycle API integration) used # only to deterministically drive ready_recovery_action.restart in -# lifecycle_daemon_retries_config.json. See main.cpp for behavior. +# lifecycle_daemon_retry_config.json. See main.cpp for behavior. cc_binary( name = "flaky_startup_app", srcs = ["main.cpp"], diff --git a/feature_integration_tests/test_cases/tests/lifecycle/conftest.py b/feature_integration_tests/test_cases/tests/lifecycle/conftest.py new file mode 100644 index 00000000000..73d67528884 --- /dev/null +++ b/feature_integration_tests/test_cases/tests/lifecycle/conftest.py @@ -0,0 +1,30 @@ +# ******************************************************************************* +# Copyright (c) 2026 Contributors to the Eclipse Foundation +# +# See the NOTICE file(s) distributed with this work for additional +# information regarding copyright ownership. +# +# This program and the accompanying materials are made available under the +# terms of the Apache License Version 2.0 which is available at +# https://www.apache.org/licenses/LICENSE-2.0 +# +# SPDX-License-Identifier: Apache-2.0 +# ******************************************************************************* +"""Shared fixtures for lifecycle daemon tests.""" + +from __future__ import annotations + +from typing import Any + +import pytest +from daemon_helpers import start_launch_manager_daemon, stop_launch_manager_daemon + + +@pytest.fixture(scope="class") +def launch_manager_daemon(tmp_path_factory: pytest.TempPathFactory) -> dict[str, Any]: + """Start a real launch_manager process with generated flatbuffer config.""" + daemon_info = start_launch_manager_daemon(tmp_path_factory) + try: + yield daemon_info + finally: + stop_launch_manager_daemon(daemon_info) diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py b/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py index 85cb8898f05..df41f52f62b 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py @@ -24,7 +24,6 @@ import pytest from daemon_helpers import ( is_running, - launch_manager_daemon, start_launch_manager_daemon, stop_launch_manager_daemon, wait_until, @@ -81,8 +80,8 @@ class TestConditionalLaunchingBlocksOnMissingDependency: Runs its own launch_manager instance (rather than the shared class-scoped `launch_manager_daemon` fixture) with cpp_supervised_app withheld, so it can observe the negative case: rust must not start while its dependency cannot. - It uses a unique runtime root and generated configuration beneath - `TEST_TMPDIR`, so it is independent of other lifecycle daemon instances. + It lives in its own class so the class-scoped fixture is torn down first: overlapping + daemons collide on launch_manager's fixed POSIX shm names (daemon_helpers._live_daemons). Not parametrized on `version`: dependency gating is independent of which scenario variant is under test elsewhere, so this runs exactly once. diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py b/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py index a2d359c8e09..9dd2abc1c09 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py @@ -40,7 +40,6 @@ from daemon_helpers import ( first_pid, is_running, - launch_manager_daemon, pgrep_cmdline_pattern, signal_process, start_launch_manager_daemon, @@ -254,11 +253,9 @@ def test_config_defines_uid_gid_scheduling_and_priority(self, version: str) -> N assert isinstance(sandbox.get("scheduling_priority"), int), "Expected integer scheduling priority" assert isinstance(sandbox.get("scheduling_policy"), str), "Expected scheduling policy string" - @add_test_properties( - partially_verifies=["feat_req__lifecycle__uid_gid_support"], - test_type="requirements-based", - derivation_technique="requirements-analysis", - ) + # Not decorated with @add_test_properties: this test is unconditionally skipped in CI/CD + # (see below), so it never actually exercises feat_req__lifecycle__uid_gid_support and + # shouldn't claim to verify it until it can run there. # Skipped in CI/CD (both rust/cpp): requires launch_manager to gain cap_setuid/cap_setgid via # setcap, which needs both FIT_ENABLE_SETCAP=1 (unset in the GitHub Actions workflow) and # unsandboxed execution (linux-sandbox's PR_SET_NO_NEW_PRIVS makes the grant inert at exec @@ -300,15 +297,19 @@ def test_launched_process_uid_gid_matches_config_when_applied( assert effective_gid == expected_gid, ( f"Effective gid mismatch for {app_name}: expected {expected_gid}, got {effective_gid}" ) + # Only meaningful when the configured sandbox uid actually differs from the runner's own + # uid; some local/dev configs (e.g. lifecycle_daemon_config.json's uid 1001) coincide with + # a common dev-user uid, in which case effective_uid == os.getuid() even when the sandbox + # identity was genuinely applied, and this check can't tell the two cases apart. + if expected_uid != os.getuid(): + assert effective_uid != os.getuid(), ( + f"{app_name} is running as the test runner's own uid ({effective_uid}); sandbox " + "identity was not actually applied" + ) - @add_test_properties( - partially_verifies=[ - "feat_req__lifecycle__launch_priority_support", - "feat_req__lifecycle__scheduling_policy", - ], - test_type="requirements-based", - derivation_technique="requirements-analysis", - ) + # Not decorated with @add_test_properties: this test is unconditionally skipped in CI/CD + # (see below), so it never actually exercises feat_req__lifecycle__launch_priority_support / + # feat_req__lifecycle__scheduling_policy and shouldn't claim to verify them until it can run there. # Skipped in CI/CD (both rust/cpp): requires launch_manager to gain cap_sys_nice via setcap, # which needs both FIT_ENABLE_SETCAP=1 (unset in the GitHub Actions workflow) and unsandboxed # execution (linux-sandbox's PR_SET_NO_NEW_PRIVS makes the grant inert at exec time even if @@ -331,7 +332,8 @@ def test_launched_process_scheduling_matches_config_when_applied( config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" config = json.loads(config_path.read_text(encoding="utf-8")) - sandbox = config["defaults"]["deployment_config"]["sandbox"] + component_sandbox = config["components"][app_name].get("deployment_config", {}).get("sandbox") + sandbox = component_sandbox or config["defaults"]["deployment_config"]["sandbox"] configured_policy = sandbox["scheduling_policy"] configured_priority = int(sandbox["scheduling_priority"]) @@ -363,6 +365,59 @@ def test_launched_process_scheduling_matches_config_when_applied( f"Scheduling priority mismatch for {app_name}: expected {configured_priority}, got {rt_priority}" ) + @add_test_properties( + partially_verifies=["feat_req__lifecycle__scheduling_policy", "feat_req__lifecycle__launch_priority_support"], + test_type="requirements-based", + derivation_technique="requirements-analysis", + ) + def test_scheduling_policy_is_non_default_and_applied( + self, + launch_manager_daemon: dict[str, Any], + version: str, + ) -> None: + """Verify the launched process's scheduling policy differs from launch_manager's own. + + Both apps carry a non-default sandbox policy (`rust_supervised_app`: `SCHED_RR`/`10`, + `cpp_supervised_app`: `SCHED_FIFO`/`20`) vs. the OS default `SCHED_OTHER`/`0` that + launch_manager itself runs under. Comparing the launched app against the daemon's own + scheduling, rather than against a fixed constant, proves the policy was actually applied + rather than coincidentally matching the process's inherited default. + """ + daemon_info = launch_manager_daemon + if not daemon_info["sandbox_privileged"]: + pytest.skip( + "launch_manager was not granted cap_sys_nice in this environment; " + f"scheduling policy cannot be applied. Reason: {daemon_info['sandbox_privileged_reason']}" + ) + + app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" + app_path = str(daemon_info["apps"][version]) + + started = wait_until(lambda: is_running(app_path), timeout_s=8.0) + assert started, f"{app_name} was not launched before scheduling verification" + + daemon_sched = self._proc_sched_policy_and_priority(str(daemon_info["daemon"].pid())) + assert daemon_sched is not None, "Could not read scheduling metadata via chrt for launch_manager" + + pid = None + app_sched = None + for _ in range(20): + pid = first_pid(app_path) + if pid is None: + time.sleep(0.1) + continue + app_sched = self._proc_sched_policy_and_priority(pid) + if app_sched is not None: + break + time.sleep(0.1) + assert pid is not None, f"Could not resolve PID for {app_name}" + assert app_sched is not None, f"Could not read scheduling metadata via chrt for {app_name} pid={pid}" + + assert app_sched != daemon_sched, ( + f"{app_name}'s scheduling {app_sched} matches launch_manager's own {daemon_sched}; " + "sandbox scheduling policy was not actually applied" + ) + @add_test_properties( partially_verifies=["feat_req__lifecycle__secpol_non_root"], test_type="requirements-based", @@ -392,6 +447,15 @@ def test_launch_manager_and_apps_are_not_running_as_root( effective_uid, _ = proc_ids assert effective_uid != 0, f"{app_name} is unexpectedly running as root" + +class TestSupervisedAppRecovery: + """Kill-and-restart recovery against a dedicated launch_manager instance. + + Kept out of TestProcessLaunchingWithDaemon so its own daemon never overlaps that class's + `launch_manager_daemon` fixture: concurrent daemons collide on launch_manager's fixed + POSIX shm names (see daemon_helpers._live_daemons). + """ + @add_test_properties( partially_verifies=[ "feat_req__lifecycle__monitor_abnormal_term", @@ -457,10 +521,10 @@ class TestParallelLaunch: Runs its own launch_manager instance (rather than the shared class-scoped `launch_manager_daemon` fixture used by TestProcessLaunchingWithDaemon), for two reasons: - 1. `lifecycle_daemon_parallel_launch_config.json` has no depends_on between the - two apps, unlike the shared fixture's config - that's the whole point. - 2. Every daemon receives an independent runtime directory and generated config - beneath `TEST_TMPDIR`, so it cannot interfere with the shared fixture. + 1. It renders the config with `independent_apps=True` (no depends_on between the + two apps), unlike the shared fixture's config - that's the whole point. + 2. It is a separate class, so the shared fixture is torn down before this daemon starts; + overlapping daemons collide on launch_manager's fixed POSIX shm names. Parametrized on `version` only because the module-level `pytestmark` applies it to every class in this file; parallel launch itself is independent of which @@ -483,29 +547,39 @@ def test_independent_processes_launch_without_waiting_on_each_other( cpp_supervised_app, so it cannot demonstrate parallel launch - both apps eventually running there is equally consistent with strict serialization. - Runs against `lifecycle_daemon_parallel_launch_config.json`, where neither - app depends on the other, and withholds one app's binary (non-executable) at - a time. If launch order were still serialized (e.g. alphabetically or by - declaration order), withholding the first-launched app would also block the - second. The other app reaching Running regardless of which one is withheld - shows launch does not wait on the withheld one, i.e. genuine parallel launch. + Renders that config with `independent_apps=True`, so neither app depends on + the other, and stalls one app at a time: it is replaced by a + stub that runs but never reports Running, so a serialized launcher would sit + on it for the full `ready_timeout` (10 s, plus retries) before starting the + next. The other app must be up within 4 s of daemon startup - + well under one `ready_timeout` - regardless of which one is stalled, so the + pass window cannot be met by strictly sequential launch in either order. `version` is unused but required by the module-scope parametrize. """ - for blocked, other in (("cpp", "rust"), ("rust", "cpp")): + ready_timeout_s = 10.0 # rendered over the base config's 2.0 s + parallel_window_s = 4.0 # + ~1 s daemon startup grace, still well under ready_timeout + assert parallel_window_s < ready_timeout_s / 2 + for stalled, other in (("cpp", "rust"), ("rust", "cpp")): daemon_info = start_launch_manager_daemon( tmp_path_factory, - blocked_apps=frozenset({blocked}), + stalled_apps=frozenset({stalled}), wait_for_apps=False, - config_template="//feature_integration_tests/configs:lifecycle_daemon_parallel_launch_config.json", + independent_apps=True, + ready_timeout_s=ready_timeout_s, ) try: + stalled_path = str(daemon_info["apps"][stalled]) other_path = str(daemon_info["apps"][other]) - other_started = wait_until(lambda: is_running(other_path), timeout_s=8.0) + other_started = wait_until(lambda: is_running(other_path), timeout_s=parallel_window_s) assert other_started, ( - f"{other}_supervised_app did not start while {blocked}_supervised_app was " - "withheld, even though neither depends on the other - launch is not parallel" + f"{other}_supervised_app did not start within {parallel_window_s}s while " + f"{stalled}_supervised_app was stalled (ready_timeout={ready_timeout_s}s), even " + "though neither depends on the other - launch is serialized, not parallel" ) + # Rules out a vacuous pass: the stalled stub must be up too, i.e. both were + # in flight concurrently rather than the stub simply never being launched. + assert is_running(stalled_path), f"stalled {stalled}_supervised_app stub was never launched" finally: stop_launch_manager_daemon(daemon_info) @@ -534,25 +608,28 @@ def test_watchdog_detection(self, launch_manager_daemon: dict[str, Any], version text=True, check=False, ) - if result.returncode != 0: - pytest.skip(f"{app_name} not active; activate Running run target before watchdog check") + # The fixture already waited for the app to reach Running, so a missing process here + # means it died under supervision - a genuine failure, not a skip. + assert result.returncode == 0, f"{app_name} died before the watchdog check" pid = result.stdout.strip().split("\n")[0] sandbox_privileged = launch_manager_daemon["sandbox_privileged"] sent, reason = signal_process(pid, "-STOP", sandbox_privileged=sandbox_privileged) assert sent, f"Could not signal {app_name} (pid={pid}): {reason}" try: - # Allow supervision/watchdog loop to detect stalled process. - time.sleep(4.0) - logs = daemon.get_logs() watchdog_patterns = [ rf"Got kRunning timeout for process.*\(\s*{re.escape(app_name)}\s*\)", rf"unexpected termination of process.*\(\s*{re.escape(app_name)}\s*\)", rf"Alive Supervision \(\s*{re.escape(app_name)}\s*\) switched to FAILED", rf"Alive Supervision \(\s*{re.escape(app_name)}\s*\) switched to EXPIRED", ] - assert any(re.search(pattern, logs) for pattern in watchdog_patterns), ( - f"No target-specific watchdog diagnostics found for {app_name}.\nDaemon logs:\n{logs}" + # Poll rather than sleep: detection latency varies under CI load. + detected = wait_until( + lambda: any(re.search(pattern, daemon.get_logs()) for pattern in watchdog_patterns), + timeout_s=8.0, + ) + assert detected, ( + f"No target-specific watchdog diagnostics found for {app_name}.\nDaemon logs:\n{daemon.get_logs()}" ) finally: signal_process(pid, "-CONT", sandbox_privileged=sandbox_privileged) diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py b/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py index f0fc82bcc58..b43f52aa480 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py @@ -33,57 +33,32 @@ from daemon_helpers import ( is_running, read_retry_attempt_count, - start_flaky_retry_daemon, - stop_flaky_retry_daemon, wait_until, ) +from lifecycle_scenario import RetryDaemonScenario from test_properties import add_test_properties -# Must match "number_of_attempts" in both lifecycle_daemon_retry_*_config.json. +# Must match flaky_startup_app's "number_of_attempts" in lifecycle_daemon_retry_config.json. _NUMBER_OF_ATTEMPTS = 2 -@pytest.fixture(scope="class") -def recovers_daemon(tmp_path_factory: pytest.TempPathFactory): - daemon_info = start_flaky_retry_daemon( - tmp_path_factory, - "//feature_integration_tests/configs:lifecycle_daemon_retry_recovers_config.json", - crashes_before_success=2, - ) - try: - yield daemon_info - finally: - stop_flaky_retry_daemon(daemon_info) - - -@pytest.fixture(scope="class") -def exhausts_daemon(tmp_path_factory: pytest.TempPathFactory): - daemon_info = start_flaky_retry_daemon( - tmp_path_factory, - "//feature_integration_tests/configs:lifecycle_daemon_retry_exhausts_config.json", - crashes_before_success=999, - ) - try: - yield daemon_info - finally: - stop_flaky_retry_daemon(daemon_info) - - -class TestRetrySucceedsWithinConfiguredAttempts: +class TestRetrySucceedsWithinConfiguredAttempts(RetryDaemonScenario): """The component crashes fewer times than `number_of_attempts` allows.""" + crashes_before_success = 2 + @add_test_properties( partially_verifies=["feat_req__lifecycle__retries_configurable"], test_type="requirements-based", derivation_technique="requirements-analysis", ) - def test_component_recovers_within_configured_attempts(self, recovers_daemon: dict[str, Any]) -> None: + def test_component_recovers_within_configured_attempts(self, retry_daemon: dict[str, Any]) -> None: """Daemon retries a failing component up to `number_of_attempts` and lets it reach Running once it stops crashing. """ - app_path = recovers_daemon["app_path"] - counter_path = recovers_daemon["counter_path"] - expected_attempts = recovers_daemon["crashes_before_success"] + 1 + app_path = retry_daemon["app_path"] + counter_path = retry_daemon["counter_path"] + expected_attempts = retry_daemon["crashes_before_success"] + 1 # Check the attempt counter before is_running(): a crashing attempt is still # technically "running" for the microseconds before it aborts, so polling @@ -106,21 +81,23 @@ def test_component_recovers_within_configured_attempts(self, recovers_daemon: di assert is_running(app_path), "flaky_startup_app stopped running after recovering" -class TestRetryExhaustionTriggersRecovery: +class TestRetryExhaustionTriggersRecovery(RetryDaemonScenario): """The component always crashes, exceeding `number_of_attempts`.""" + crashes_before_success = 999 + @add_test_properties( partially_verifies=["feat_req__lifecycle__retries_configurable"], test_type="requirements-based", derivation_technique="requirements-analysis", ) - def test_daemon_gives_up_after_configured_attempts(self, exhausts_daemon: dict[str, Any]) -> None: + def test_daemon_gives_up_after_configured_attempts(self, retry_daemon: dict[str, Any]) -> None: """Daemon stops restarting a component once `number_of_attempts` is exhausted, instead of retrying forever, and executes the run target's `recovery_action` (switch to `fallback_run_target`). """ - app_path = exhausts_daemon["app_path"] - counter_path = exhausts_daemon["counter_path"] + app_path = retry_daemon["app_path"] + counter_path = retry_daemon["counter_path"] settled = wait_until( lambda: read_retry_attempt_count(counter_path) >= _NUMBER_OF_ATTEMPTS + 1, @@ -145,4 +122,4 @@ def test_daemon_gives_up_after_configured_attempts(self, exhausts_daemon: dict[s "flaky_startup_app is still running after exhausting retries; recovery_action " "(switch_run_target -> fallback_run_target) should have stopped further attempts" ) - assert exhausts_daemon["daemon"].is_running(), "Launch Manager daemon crashed instead of switching run target" + assert retry_daemon["daemon"].is_running(), "Launch Manager daemon crashed instead of switching run target" diff --git a/feature_integration_tests/test_scenarios/cpp/src/internals/persistency/kvs_build_helpers.h b/feature_integration_tests/test_scenarios/cpp/src/internals/persistency/kvs_build_helpers.h index e0f65444aa8..e82e591597d 100644 --- a/feature_integration_tests/test_scenarios/cpp/src/internals/persistency/kvs_build_helpers.h +++ b/feature_integration_tests/test_scenarios/cpp/src/internals/persistency/kvs_build_helpers.h @@ -15,80 +15,22 @@ #define INTERNALS_PERSISTENCY_KVS_BUILD_HELPERS_H_ #include "kvs_parameters.h" +#include "internals/log_helpers.h" #include #include -#include -#include -#include #include -#include #include #include namespace kvs_build_helpers { -/** - * @brief Return the current UNIX timestamp as a decimal string (seconds). - * - * Used to populate the "timestamp" field in structured JSON log lines so that - * the C++ output matches the Rust tracing JSON shape expected by the FIT log - * filters. - * - * @return String containing the number of seconds since the UNIX epoch. - */ -inline std::string unix_seconds_string() { - const auto now = std::chrono::system_clock::now(); - const auto secs = - std::chrono::duration_cast(now.time_since_epoch()).count(); - return std::to_string(secs); -} - -/** - * @brief Emit a structured JSON INFO log line to stdout. - * - * Matches the Rust tracing JSON format expected by the FIT LogContainer so - * that Python test assertions can use find_log() uniformly for both Rust and - * C++ scenarios. - * - * Example output: - * @code - * {"timestamp":"1234567890","level":"INFO","fields":{"key":"my_key","value":42.0}, - * "target":"cpp_test_scenarios::scenarios::persistency::my_module","threadId":"ThreadId(1)"} - * @endcode - * - * @param fields JSON fragment for the "fields" object, e.g. @c "\"key\":\"x\",\"value\":1.0" - * @param target Module target string embedded in the log line. - */ -inline void log_info(const std::string& fields, const std::string& target) { - std::cout << "{\"timestamp\":\"" << unix_seconds_string() - << "\",\"level\":\"INFO\",\"fields\":{" << fields - << "},\"target\":\"" << target - << "\",\"threadId\":\"ThreadId(1)\"}\n"; -} - -/** - * @brief Format a double value to match Python's str(float) representation. - * - * For whole-number values (e.g. 42.0, 200.0) this appends ".0" so that the - * resulting string matches what Python's f-string interpolation produces. - * Non-integer values (e.g. 3.14) are printed as-is by the default stream. - * - * @param v Double value to format. - * @return String representation matching Python float str(). - */ -inline std::string format_double_python(double v) { - std::ostringstream oss; - oss.imbue(std::locale::classic()); // Ensure '.' decimal separator regardless of process locale. - oss << v; - std::string s = oss.str(); - if (s.find('.') == std::string::npos && s.find('e') == std::string::npos && - s.find('E') == std::string::npos) { - s += ".0"; - } - return s; -} +// Generic structured-log helpers live in the feature-neutral internals/log_helpers.h so the +// log line format (including json_escape of `target`) is defined once for all scenarios. +using log_helpers::format_double_python; +using log_helpers::log_info; +using log_helpers::unix_seconds_string; /** * @brief Convert an optional KvsDefaults mode to the boolean flag expected by KvsBuilder. diff --git a/patches/lifecycle/001-forward-visibility-to-config-combiner.patch b/patches/lifecycle/001-forward-visibility-to-config-combiner.patch deleted file mode 100644 index 512d319bbe0..00000000000 --- a/patches/lifecycle/001-forward-visibility-to-config-combiner.patch +++ /dev/null @@ -1,10 +0,0 @@ -diff --git a/scripts/config_mapping/config.bzl b/scripts/config_mapping/config.bzl -index 0a07e26d..ccdde084 100644 ---- a/scripts/config_mapping/config.bzl -+++ b/scripts/config_mapping/config.bzl -@@ -233,4 +233,5 @@ def launch_manager_config( - ], - }), - dir_name = flatbuffer_out_dir, -+ visibility = kwargs.get("visibility"), - ) diff --git a/patches/lifecycle/BUILD b/patches/lifecycle/BUILD deleted file mode 100644 index e69de29bb2d..00000000000 From 7efe933a8c65ebcf0edc15b7ea8826e9d3aa04e8 Mon Sep 17 00:00:00 2001 From: Saumya-R Date: Tue, 29 Sep 2026 23:41:38 +0530 Subject: [PATCH 6/9] adding review comment fixes --- .../test_cases/daemon_helpers.py | 199 ++++++++++++++---- .../test_process_launching_with_daemon.py | 75 ++++--- 2 files changed, 195 insertions(+), 79 deletions(-) diff --git a/feature_integration_tests/test_cases/daemon_helpers.py b/feature_integration_tests/test_cases/daemon_helpers.py index e03863d9aa8..f3ef30c999b 100644 --- a/feature_integration_tests/test_cases/daemon_helpers.py +++ b/feature_integration_tests/test_cases/daemon_helpers.py @@ -150,7 +150,20 @@ def wait_until(predicate, timeout_s: float, interval_s: float = 0.2) -> bool: return False -_SETCAP_CAPS = "cap_setuid,cap_setgid,cap_sys_nice+ep" +# cap_kill: once the sandbox uid differs from the runner's (see `_generate_runtime_config`), +# launch_manager (still running as the runner uid) needs it to terminate/restart its own children. +_SETCAP_CAPS = "cap_setuid,cap_setgid,cap_sys_nice,cap_kill+ep" + +# Sandbox uid used when capabilities are granted. It must differ from the runner's own uid, +# otherwise the uid/gid test cannot tell an applied sandbox identity from an inherited one +# (the config's 1001 is also the default first-user / GitHub-hosted-runner uid). +_SANDBOX_UID = 65533 + +# Copies of `kill` (cap_kill) and `cat` (cap_sys_ptrace + cap_dac_read_search), staged when capabilities are granted, +# so the runner can signal apps running under `_SANDBOX_UID` and read their /proc//environ +# using only the setcap sudoers rule (no sudo kill). +_privileged_kill: Path | None = None +_privileged_cat: Path | None = None def _mount_nosuid(path: Path) -> bool: @@ -175,7 +188,11 @@ def _mount_nosuid(path: Path) -> bool: return False -def _grant_sandbox_capabilities(binary_path: Path) -> tuple[bool, str]: +def _grant_sandbox_capabilities( + binary_path: Path, + caps: str = _SETCAP_CAPS, + required: tuple[str, ...] = ("cap_setuid", "cap_setgid"), +) -> tuple[bool, str]: """Best-effort grant of the capabilities launch_manager needs to apply sandbox uid/gid and scheduling policy without running as root. Returns `(granted, reason)`: `granted` is a *verified* result (re-read via `getcap`, not just the setcap exit code) so tests @@ -201,7 +218,7 @@ def _grant_sandbox_capabilities(binary_path: Path) -> tuple[bool, str]: setcap_enabled = os.environ.get("FIT_ENABLE_SETCAP") == "1" attempts: list[tuple[list[str], str]] = [ - (["setcap", _SETCAP_CAPS, str(binary_path)], "plain setcap (requires running as root)") + (["setcap", caps, str(binary_path)], "plain setcap (requires running as root)") ] if setcap_enabled: if shutil.which("sudo") is None: @@ -209,7 +226,7 @@ def _grant_sandbox_capabilities(binary_path: Path) -> tuple[bool, str]: else: attempts.insert( 0, - (["sudo", "-n", "setcap", _SETCAP_CAPS, str(binary_path)], "sudo -n setcap"), + (["sudo", "-n", "setcap", caps, str(binary_path)], "sudo -n setcap"), ) else: attempts.append(([], "FIT_ENABLE_SETCAP not set to '1'; skipping sudo setcap attempt")) @@ -233,7 +250,7 @@ def _grant_sandbox_capabilities(binary_path: Path) -> tuple[bool, str]: getcap = shutil.which("getcap") if getcap is not None: verify = subprocess.run([getcap, str(binary_path)], capture_output=True, text=True, check=False) - if "cap_setuid" not in verify.stdout or "cap_setgid" not in verify.stdout: + if any(cap not in verify.stdout for cap in required): nosuid_hint = " (path is on a 'nosuid' mount)" if _mount_nosuid(binary_path) else "" failures.append( f"{label} reported success but getcap did not confirm the capabilities" @@ -249,11 +266,13 @@ def _grant_sandbox_capabilities(binary_path: Path) -> tuple[bool, str]: def signal_process(pid: str, sig: str, *, sandbox_privileged: bool) -> tuple[bool, str]: """Send `sig` (e.g. "-9", "-STOP", "-CONT") to `pid`, escalating via sudo if needed. - Under sandbox capabilities, supervised apps run as the configured sandbox uid/gid, - not the runner's own uid, so a plain `kill` fails. Falls back to `sudo -n kill` when - `FIT_ENABLE_SETCAP=1` (same sudoers scope as `_grant_sandbox_capabilities`). + Under sandbox capabilities, supervised apps run as `_SANDBOX_UID`, not the runner's own + uid, so a plain `kill` fails. Falls back to the cap_kill `kill` copy staged by + `_spawn_daemon`, then to `sudo -n kill` when `FIT_ENABLE_SETCAP=1`. """ attempts: list[list[str]] = [["kill", sig, pid]] + if sandbox_privileged and _privileged_kill is not None: + attempts.append([str(_privileged_kill), sig, pid]) if sandbox_privileged and os.environ.get("FIT_ENABLE_SETCAP") == "1" and shutil.which("sudo") is not None: attempts.append(["sudo", "-n", "kill", sig, pid]) @@ -306,9 +325,14 @@ def stop(self) -> None: if self.is_running(): os.killpg(os.getpgid(self.process.pid), signal.SIGKILL) self.process.wait(timeout=5) - # Launched apps inherit the daemon's stdout pipe and can outlive it, so the reader may - # still be blocked in read() holding the buffer lock; close() would then deadlock. - # Only close once the reader has seen EOF (callers pkill the apps afterwards). + + def close_output(self) -> None: + """Join the stdout reader and close the pipe. + + Launched apps inherit the daemon's stdout pipe and can outlive it, so call this only + after the apps are killed too; otherwise the reader is still blocked in read() holding + the buffer lock and close() would deadlock. + """ self._thread.join(timeout=1) if self.process.stdout is not None and not self._thread.is_alive(): self.process.stdout.close() @@ -327,6 +351,8 @@ def _generate_runtime_config( runtime_root: Path, etc_dir: Path, sandbox_privileged: bool = True, + *, + remap_sandbox_uid: bool = False, independent_apps: bool = False, ready_timeout_s: float | None = None, crashes_before_success: int | None = None, @@ -347,6 +373,11 @@ def _generate_runtime_config( `False`, downgrade any configured non-default scheduling policy back to the harmless `SCHED_OTHER`/`0` so the daemon still starts cleanly; scheduling-specific tests already key off `sandbox_privileged` themselves and skip rather than assert against it in that case. + + `remap_sandbox_uid` rewrites every sandbox uid to `_SANDBOX_UID` and gid to the runner's own + gid (so the group bits of the runner-owned runtime dirs still grant access), and opens + `runtime_root` to that group. Read the rendered `etc_dir / "lifecycle_config.json"` for the + ids actually applied. """ config = json.loads(_resolve_target_path(config_template).read_text(encoding="utf-8")) config["defaults"]["deployment_config"]["bin_dir"] = str(runtime_root / "bin") @@ -356,11 +387,14 @@ def _generate_runtime_config( for component in config["components"].values(): component["component_properties"].pop("depends_on", None) + sandboxes = [config["defaults"]["deployment_config"].get("sandbox")] + sandboxes += [component.get("deployment_config", {}).get("sandbox") for component in config["components"].values()] + if remap_sandbox_uid: + for sandbox in filter(None, sandboxes): + sandbox["uid"] = _SANDBOX_UID + sandbox["gid"] = os.getgid() + runtime_root.chmod(0o750) if not sandbox_privileged: - sandboxes = [config["defaults"]["deployment_config"].get("sandbox")] - sandboxes += [ - component.get("deployment_config", {}).get("sandbox") for component in config["components"].values() - ] for sandbox in sandboxes: if sandbox and sandbox.get("scheduling_policy") not in (None, "SCHED_OTHER"): sandbox["scheduling_policy"] = "SCHED_OTHER" @@ -468,17 +502,33 @@ def _spawn_daemon( shutil.copy2(launch_manager, lm_dst) lm_dst.chmod(0o755) + global _privileged_kill, _privileged_cat + _privileged_kill = _privileged_cat = None if grant_sandbox_capabilities: sandbox_privileged, sandbox_privileged_reason = _grant_sandbox_capabilities(lm_dst) else: sandbox_privileged, sandbox_privileged_reason = False, "not requested" + if sandbox_privileged: + # Only move apps off the runner's uid if the runner can still signal them and read + # their /proc//environ afterwards. + kill_tool, kill_reason = _stage_privileged_tool(work_dir, "kill", ("cap_kill",)) + cat_tool, cat_reason = _stage_privileged_tool(work_dir, "cat", ("cap_sys_ptrace", "cap_dac_read_search")) + if kill_tool is not None and cat_tool is not None: + _privileged_kill, _privileged_cat = kill_tool, cat_tool + else: + sandbox_privileged_reason += f"; sandbox uid not remapped: {kill_reason}; {cat_reason}" for src, dst, mode in staged_binaries: shutil.copy2(src, dst) dst.chmod(mode) _generate_runtime_config( - config_template, runtime_root, etc_dir, sandbox_privileged=sandbox_privileged, **config_options + config_template, + runtime_root, + etc_dir, + sandbox_privileged=sandbox_privileged, + remap_sandbox_uid=_privileged_kill is not None, + **config_options, ) env = os.environ.copy() @@ -508,11 +558,17 @@ def _collect_output() -> None: daemon = ManagedDaemon(process=process, _lines=lines, _thread=thread) _live_daemons.append(daemon) - # Give startup a chance to complete and fail early if config is broken. - time.sleep(1.0) - if not daemon.is_running(): - logs = daemon.get_logs() - pytest.fail(f"launch_manager failed to start. Logs:\n{logs}") + # The caller never receives `daemon` if this raises, so tear it down here (daemon, any + # app it already launched, and its output pipe) rather than leaving a detached + # launch_manager running and blocking every later start via _live_daemons. + try: + # Give startup a chance to complete and fail early if config is broken. + time.sleep(1.0) + if not daemon.is_running(): + pytest.fail(f"launch_manager failed to start. Logs:\n{daemon.get_logs()}") + except BaseException: + _teardown(daemon, [dst for _, dst, _ in staged_binaries], runtime_root) + raise return daemon, sandbox_privileged, sandbox_privileged_reason @@ -546,13 +602,17 @@ def start_launch_manager_daemon( """ runtime_root = Path(tempfile.mkdtemp(prefix="lifecycle_fit-", dir=_tmpdir_root())) + bin_dir = runtime_root / "bin" + apps = { + "rust": bin_dir / "rust_supervised_app", + "cpp": bin_dir / "cpp_supervised_app", + } daemon = None try: work_dir = tmp_path_factory.mktemp("lm-daemon") etc_dir = work_dir / "etc" etc_dir.mkdir(parents=True, exist_ok=True) - bin_dir = runtime_root / "bin" bin_dir.mkdir(parents=True, exist_ok=True) rust_supervised = _resolve_target_path("@score_lifecycle//examples/rust_supervised_app:rust_supervised_app") @@ -581,10 +641,6 @@ def start_launch_manager_daemon( ready_timeout_s=ready_timeout_s, ) - apps = { - "rust": bin_dir / "rust_supervised_app", - "cpp": bin_dir / "cpp_supervised_app", - } if wait_for_apps and not _wait_for_apps({k: v for k, v in apps.items() if k not in blocked_apps}): process_snapshot = subprocess.run( ["ps", "-eo", "pid,args"], @@ -600,9 +656,8 @@ def start_launch_manager_daemon( f"{process_snapshot.stdout}{process_snapshot.stderr}" ) except BaseException: - if daemon is not None: - daemon.stop() - _cleanup_runtime_root(runtime_root) + # An app that did reach Running (e.g. on the wait timeout) outlives the daemon. + _teardown(daemon, list(apps.values()), runtime_root) raise return { @@ -613,6 +668,7 @@ def start_launch_manager_daemon( "sandbox_privileged": sandbox_privileged, "sandbox_privileged_reason": sandbox_privileged_reason, "runtime_root": runtime_root, + "runtime_config": etc_dir / "lifecycle_config.json", } @@ -630,19 +686,19 @@ def start_flaky_retry_daemon( what the calling test is checking. """ runtime_root = Path(tempfile.mkdtemp(prefix="lifecycle_fit_retries-", dir=_tmpdir_root())) + bin_dir = runtime_root / "bin" + app_dst = bin_dir / "flaky_startup_app" daemon = None try: work_dir = tmp_path_factory.mktemp("lm-retry-daemon") etc_dir = work_dir / "etc" etc_dir.mkdir(parents=True, exist_ok=True) - bin_dir = runtime_root / "bin" bin_dir.mkdir(parents=True, exist_ok=True) flaky_app = _resolve_target_path( "//feature_integration_tests/test_cases/support_apps/flaky_startup_app:flaky_startup_app" ) - app_dst = bin_dir / "flaky_startup_app" staged_binaries = [(flaky_app, app_dst, 0o755)] counter_path = runtime_root / "flaky_startup_app.counter" @@ -658,9 +714,7 @@ def start_flaky_retry_daemon( crashes_before_success=crashes_before_success, ) except BaseException: - if daemon is not None: - daemon.stop() - _cleanup_runtime_root(runtime_root) + _teardown(daemon, [app_dst], runtime_root) raise return { @@ -674,21 +728,74 @@ def start_flaky_retry_daemon( } -def _stop_daemon(daemon_info: dict[str, Any], app_paths: list[Path]) -> None: - """Stop `daemon_info["daemon"]`, pkill each of `app_paths` by cmdline, then clean up - its runtime root. Runs unconditionally even if stopping the daemon itself raises. +def _teardown(daemon: ManagedDaemon | None, app_paths: list[Path], runtime_root: Path) -> None: + """Stop `daemon` (if started), pkill each of `app_paths` by cmdline, close the daemon's + output pipe, then clean up `runtime_root`. Every step runs even if an earlier one raises. + + Supervised apps setpgid() into their own process group, so stopping the daemon's group + never reaches them; they must be killed explicitly. """ try: - daemon_info["daemon"].stop() + if daemon is not None: + daemon.stop() finally: - for app_path in app_paths: - subprocess.run( - ["pkill", "-f", pgrep_cmdline_pattern(str(app_path))], - capture_output=True, - text=True, - check=False, - ) - _cleanup_runtime_root(daemon_info["runtime_root"]) + try: + for app_path in app_paths: + _kill_app(app_path) + if daemon is not None: + daemon.close_output() + finally: + _cleanup_runtime_root(runtime_root) + + +def _stage_privileged_tool(work_dir: Path, name: str, caps: tuple[str, ...]) -> tuple[Path | None, str]: + """Copy the external `name` binary into `work_dir` and grant it `caps` via the same setcap + path as launch_manager. Returns `(path, reason)`; `path` is None if the grant failed. + + The copy keeps its basename (procps `kill` is multi-call and dispatches on argv[0]) and is + mode 0700: only the runner can execute it, and the runner can already `sudo -n setcap`. + """ + src = shutil.which(name) + if src is None: + return None, f"{name} binary not found on PATH" + dst = work_dir / "privileged" / name + dst.parent.mkdir(exist_ok=True) + shutil.copy2(Path(src).resolve(), dst) + dst.chmod(0o700) + granted, reason = _grant_sandbox_capabilities(dst, ",".join(caps) + "+ep", caps) + return (dst if granted else None), f"{'/'.join(caps)} on {name}: {reason}" + + +def read_proc_file(pid: str, name: str) -> bytes: + """Read `/proc//`, via the privileged `cat` copy when one is staged. + + Files like `environ` need ptrace access, and after its setuid() the app is non-dumpable so + its /proc files are root-owned 0400; the runner lacks both once the app runs as `_SANDBOX_UID`. + """ + path = f"/proc/{pid}/{name}" + if _privileged_cat is None: + return Path(path).read_bytes() + return subprocess.run([str(_privileged_cat), path], capture_output=True, check=True).stdout + + +def _kill_app(app_path: Path) -> None: + """SIGKILL every process whose cmdline matches `app_path`, via the cap_kill `kill` copy + when one is staged (apps running as `_SANDBOX_UID` ignore a plain pkill: EPERM).""" + if _privileged_kill is None: + subprocess.run( + ["pkill", "-9", "-f", pgrep_cmdline_pattern(str(app_path))], capture_output=True, text=True, check=False + ) + return + pids = subprocess.run( + ["pgrep", "-f", pgrep_cmdline_pattern(str(app_path))], capture_output=True, text=True, check=False + ).stdout.split() + if pids: + subprocess.run([str(_privileged_kill), "-9", *pids], capture_output=True, text=True, check=False) + + +def _stop_daemon(daemon_info: dict[str, Any], app_paths: list[Path]) -> None: + """Tear down a started daemon, its `app_paths`, and its runtime root.""" + _teardown(daemon_info["daemon"], app_paths, daemon_info["runtime_root"]) def stop_flaky_retry_daemon(daemon_info: dict[str, Any]) -> None: diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py b/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py index 9dd2abc1c09..a9530989014 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py @@ -40,7 +40,7 @@ from daemon_helpers import ( first_pid, is_running, - pgrep_cmdline_pattern, + read_proc_file, signal_process, start_launch_manager_daemon, stop_launch_manager_daemon, @@ -73,7 +73,7 @@ def _proc_cmdline(pid: str) -> list[str]: @staticmethod def _proc_environ(pid: str) -> dict[str, str]: """Read process environment from /proc as a key/value mapping.""" - raw = Path(f"/proc/{pid}/environ").read_bytes() + raw = read_proc_file(pid, "environ") env: dict[str, str] = {} for item in raw.split(b"\0"): if not item: @@ -276,11 +276,20 @@ def test_launched_process_uid_gid_matches_config_when_applied( app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" app_path = str(daemon_info["apps"][version]) - config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" - config = json.loads(config_path.read_text(encoding="utf-8")) - sandbox = config["defaults"]["deployment_config"]["sandbox"] + # The rendered config, not the source JSON: under capabilities the helper remaps the + # sandbox uid away from the runner's own (daemon_helpers._SANDBOX_UID). + config = json.loads(daemon_info["runtime_config"].read_text(encoding="utf-8")) + component_sandbox = config["components"][app_name].get("deployment_config", {}).get("sandbox") + sandbox = component_sandbox or config["defaults"]["deployment_config"]["sandbox"] expected_uid = int(sandbox["uid"]) expected_gid = int(sandbox["gid"]) + if expected_uid == os.getuid(): + # Only when the cap_kill grant failed (uid not remapped); matching uids would pass + # whether or not launch_manager applied the sandbox identity. + pytest.skip( + f"Sandbox uid {expected_uid} equals the test runner's own uid; cannot distinguish an " + f"applied sandbox identity from an inherited one. Reason: {daemon_info['sandbox_privileged_reason']}" + ) started = wait_until(lambda: is_running(app_path), timeout_s=8.0) assert started, f"{app_name} was not launched before uid/gid verification" @@ -297,15 +306,10 @@ def test_launched_process_uid_gid_matches_config_when_applied( assert effective_gid == expected_gid, ( f"Effective gid mismatch for {app_name}: expected {expected_gid}, got {effective_gid}" ) - # Only meaningful when the configured sandbox uid actually differs from the runner's own - # uid; some local/dev configs (e.g. lifecycle_daemon_config.json's uid 1001) coincide with - # a common dev-user uid, in which case effective_uid == os.getuid() even when the sandbox - # identity was genuinely applied, and this check can't tell the two cases apart. - if expected_uid != os.getuid(): - assert effective_uid != os.getuid(), ( - f"{app_name} is running as the test runner's own uid ({effective_uid}); sandbox " - "identity was not actually applied" - ) + assert effective_uid != os.getuid(), ( + f"{app_name} is running as the test runner's own uid ({effective_uid}); sandbox " + "identity was not actually applied" + ) # Not decorated with @add_test_properties: this test is unconditionally skipped in CI/CD # (see below), so it never actually exercises feat_req__lifecycle__launch_priority_support / @@ -365,11 +369,9 @@ def test_launched_process_scheduling_matches_config_when_applied( f"Scheduling priority mismatch for {app_name}: expected {configured_priority}, got {rt_priority}" ) - @add_test_properties( - partially_verifies=["feat_req__lifecycle__scheduling_policy", "feat_req__lifecycle__launch_priority_support"], - test_type="requirements-based", - derivation_technique="requirements-analysis", - ) + # Not decorated with @add_test_properties: like the two tests above, this is unconditionally + # skipped in CI/CD (no cap_sys_nice grant), so it never actually exercises + # feat_req__lifecycle__scheduling_policy / feat_req__lifecycle__launch_priority_support there. def test_scheduling_policy_is_non_default_and_applied( self, launch_manager_daemon: dict[str, Any], @@ -595,25 +597,32 @@ class TestHealthMonitoringWithDaemon: test_type="requirements-based", derivation_technique="requirements-analysis", ) - def test_watchdog_detection(self, launch_manager_daemon: dict[str, Any], version: str) -> None: - """Verify watchdog detects an unresponsive app (stopped, not reporting health) and reacts.""" - daemon = launch_manager_daemon["daemon"] + def test_watchdog_detection(self, tmp_path_factory: pytest.TempPathFactory, version: str) -> None: + """Verify watchdog detects an unresponsive app (stopped, not reporting health) and reacts. + + Uses its own daemon per version: the rust run's watchdog failure triggers recovery + (restart, then a run-target switch), so a shared daemon would hand the cpp run a + restarting or relaunched app. + """ + daemon_info = start_launch_manager_daemon(tmp_path_factory) + try: + self._check_watchdog_detection(daemon_info, version) + finally: + stop_launch_manager_daemon(daemon_info) + + @staticmethod + def _check_watchdog_detection(daemon_info: dict[str, Any], version: str) -> None: + daemon = daemon_info["daemon"] app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" # Stop the supervised process to emulate a non-reporting workload. - app_path = str(launch_manager_daemon["apps"][version]) - result = subprocess.run( - ["pgrep", "-f", pgrep_cmdline_pattern(app_path)], - capture_output=True, - text=True, - check=False, - ) - # The fixture already waited for the app to reach Running, so a missing process here + app_path = str(daemon_info["apps"][version]) + # start_launch_manager_daemon already waited for Running, so a missing process here # means it died under supervision - a genuine failure, not a skip. - assert result.returncode == 0, f"{app_name} died before the watchdog check" + pid = first_pid(app_path) + assert pid is not None, f"{app_name} died before the watchdog check" - pid = result.stdout.strip().split("\n")[0] - sandbox_privileged = launch_manager_daemon["sandbox_privileged"] + sandbox_privileged = daemon_info["sandbox_privileged"] sent, reason = signal_process(pid, "-STOP", sandbox_privileged=sandbox_privileged) assert sent, f"Could not signal {app_name} (pid={pid}): {reason}" try: From 75970283e4e74c1eb5c04e2f78e21aee7f60d046 Mon Sep 17 00:00:00 2001 From: Saumya-R Date: Wed, 30 Sep 2026 00:25:15 +0530 Subject: [PATCH 7/9] review comment fix --- feature_integration_tests/README.md | 37 ++- feature_integration_tests/test_cases/BUILD | 20 +- .../test_cases/daemon_helpers.py | 289 ++++++++---------- .../test_cases/lifecycle_scenario.py | 28 +- .../support_apps/flaky_startup_app/BUILD | 4 +- .../test_cases/tests/lifecycle/conftest.py | 4 +- .../lifecycle/test_conditional_launching.py | 74 ++--- .../test_conditional_launching_scenario.py | 121 +++----- .../test_process_launching_with_daemon.py | 259 ++++++---------- .../tests/lifecycle/test_retry_exhaustion.py | 77 ++--- .../lifecycle/conditional_launching.cpp | 48 +-- .../lifecycle/conditional_launching.rs | 11 +- 12 files changed, 401 insertions(+), 571 deletions(-) diff --git a/feature_integration_tests/README.md b/feature_integration_tests/README.md index 88255555f24..33f20bdf4fb 100644 --- a/feature_integration_tests/README.md +++ b/feature_integration_tests/README.md @@ -48,32 +48,30 @@ bazel run //feature_integration_tests/test_scenarios/rust:rust_test_scenarios -- bazel test --config=linux-x86_64 //feature_integration_tests/test_cases:fit --test_output=streamed ``` -To run the lifecycle tests directly with `pytest` and build the scenario binaries on demand: +To run the lifecycle scenario-stub tests directly with `pytest` and build the scenario binaries on demand: ```sh -python3 -m pytest feature_integration_tests/test_cases/tests/lifecycle/ \ - --build-scenarios \ - -m rust \ - --rust-target-name=//feature_integration_tests/test_scenarios/rust:rust_test_scenarios \ - -q -v - -python3 -m pytest feature_integration_tests/test_cases/tests/lifecycle/ \ - --build-scenarios \ - -m cpp \ - -q -v +python3 -m pytest feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching_scenario.py \ + --build-scenarios -m rust -q -v + +python3 -m pytest feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching_scenario.py \ + --build-scenarios -m cpp -q -v ``` -The Rust override is required because plain `--build-scenarios` defaults to -`//feature_integration_tests/test_scenarios/rust:rust_test_scenarios`, while the -lifecycle tests need the reduced lifecycle-only Rust target. +The daemon-driven lifecycle tests (`test_conditional_launching.py`, `test_process_launching_with_daemon.py`, +`test_retry_exhaustion.py`) resolve `launch_manager`, the supervised apps and the config tools from the +`FIT_*_PATH` variables set by their Bazel targets, and carry no `rust`/`cpp` marker, so run them via +`bazel test //feature_integration_tests/test_cases:fit_lifecycle_daemon` / `:fit_lifecycle_retries` +rather than plain `pytest -m rust|cpp` (which would deselect them). #### Sandbox uid/gid and scheduling-policy tests Some lifecycle daemon tests (`test_launched_process_uid_gid_matches_config_when_applied`, -`test_launched_process_scheduling_matches_config_when_applied`) verify that `launch_manager` +`test_launched_process_scheduling_matches_config_when_applied`, +`test_scheduling_policy_is_non_default_and_applied`) verify that `launch_manager` applies the sandbox `uid`/`gid` and scheduling policy from `feature_integration_tests/configs/lifecycle_daemon_config.json`. This requires granting -`launch_manager` the `cap_setuid,cap_setgid,cap_sys_nice` file capabilities via `setcap`, which +`launch_manager` the `cap_setuid,cap_setgid,cap_sys_nice,cap_kill` file capabilities via `setcap`, which in turn requires `CAP_SETFCAP` — not available to a non-root test runner by default, so these tests opt in via the `FIT_ENABLE_SETCAP` env var (backed by a passwordless sudoers rule scoped to the `setcap` binary, e.g. ` ALL=(root) NOPASSWD: /usr/sbin/setcap`, with no trailing @@ -84,7 +82,7 @@ passed via `--test_env` (not `--action_env`, which only affects build actions). the full suite are relevant: ```sh -# Default: matches CI/CD exactly (sandboxed, no FIT_ENABLE_SETCAP) — the two capability tests skip. +# Default: matches CI/CD exactly (sandboxed, no FIT_ENABLE_SETCAP) — the three capability tests skip. bazel test --config=linux-x86_64 --nocache_test_results //feature_integration_tests/test_cases:fit \ --test_output=all --test_arg=-rs --test_verbose_timeout_warnings @@ -123,8 +121,9 @@ provision a passwordless `sudo setcap` rule. As a result, the following subtests - `test_process_launching_with_daemon.py::TestProcessLaunchingWithDaemon::test_launched_process_uid_gid_matches_config_when_applied[rust|cpp]` - `test_process_launching_with_daemon.py::TestProcessLaunchingWithDaemon::test_launched_process_scheduling_matches_config_when_applied[rust|cpp]` +- `test_process_launching_with_daemon.py::TestProcessLaunchingWithDaemon::test_scheduling_policy_is_non_default_and_applied[rust|cpp]` -Reason: both depend on `launch_manager` successfully gaining `cap_setuid,cap_setgid,cap_sys_nice` +Reason: all three depend on `launch_manager` successfully gaining `cap_setuid,cap_setgid,cap_sys_nice` via `setcap` (see `daemon_helpers._grant_sandbox_capabilities`), which fails in CI for two independent reasons, either sufficient on its own: @@ -135,7 +134,7 @@ independent reasons, either sufficient on its own: never attempt the `sudo -n setcap` path; and the CI runner has no passwordless sudoers entry for `setcap` regardless. -This is by design: `_grant_sandbox_capabilities` degrades gracefully (never raises) and the two +This is by design: `_grant_sandbox_capabilities` degrades gracefully (never raises) and the three capability-dependent subtests self-skip with a diagnostic reason instead of failing the build. All other subtests in `fit_lifecycle_daemon` only check same-uid process behavior and require no privilege escalation, so they run and pass normally in CI. diff --git a/feature_integration_tests/test_cases/BUILD b/feature_integration_tests/test_cases/BUILD index 079f90ec6ed..36bbcab4b09 100644 --- a/feature_integration_tests/test_cases/BUILD +++ b/feature_integration_tests/test_cases/BUILD @@ -140,10 +140,8 @@ test_suite( ], ) -# Daemon-driven lifecycle tests (test_conditional_launching.py, test_process_launching_with_daemon.py) -# exercise launch_manager against both supervised apps together and don't depend on which -# scenario-binary language variant is under test elsewhere, so they run once here rather than -# being routed - and silently deselected - by the rust/cpp scenario-binary language marker. +# Daemon-driven lifecycle tests against a real launch_manager. No -m rust/cpp filter: these tests +# don't use the scenario binaries, and some carry no language marker, so a filter would deselect them. score_py_pytest( name = "fit_lifecycle_daemon", timeout = "long", @@ -161,8 +159,6 @@ score_py_pytest( "tests/lifecycle/conftest.py", "//feature_integration_tests/configs:lifecycle_daemon_config.json", "@flatbuffers//:flatc", - "@score_lifecycle//examples/control_application:control_daemon", - "@score_lifecycle//examples/control_application:lmcontrol", "@score_lifecycle//examples/cpp_supervised_app", "@score_lifecycle//examples/rust_supervised_app", "@score_lifecycle//score/launch_manager", @@ -183,13 +179,11 @@ score_py_pytest( }, env_inherit = ["FIT_ENABLE_SETCAP"], pytest_config = "//:pyproject.toml", - # Runs sandboxed: PR_SET_NO_NEW_PRIVS blocks setuid (sudo) / file-capability (setcap) escalation - # at exec time, so sandbox_privileged is always False here and the uid/gid/scheduling tests - # skip themselves accordingly (see daemon_helpers._grant_sandbox_capabilities). Everything - # else only signals same-uid processes and needs no privilege escalation. FIT_ENABLE_SETCAP=1 - # is for local/manual runs outside the sandbox where the sestcap grant can actually take effect. - # "exclusive": launch_manager uses fixed POSIX shm names on the host-wide /dev/shm, so it must - # not run concurrently with another launch_manager-driven test target. + # Under the default linux-sandbox, PR_SET_NO_NEW_PRIVS makes the setcap grant inert, so the + # uid/gid/scheduling tests skip (as in CI). They run only with --spawn_strategy=local and + # --test_env=FIT_ENABLE_SETCAP=1 (see README.md). + # "exclusive": launch_manager uses fixed POSIX shm names on the host-wide /dev/shm, so no other + # launch_manager-driven target may run concurrently. tags = ["exclusive"], deps = all_requirements, ) diff --git a/feature_integration_tests/test_cases/daemon_helpers.py b/feature_integration_tests/test_cases/daemon_helpers.py index f3ef30c999b..0d0ae45257a 100644 --- a/feature_integration_tests/test_cases/daemon_helpers.py +++ b/feature_integration_tests/test_cases/daemon_helpers.py @@ -29,7 +29,6 @@ import pytest - _TARGET_ENV_MAP = { "@score_lifecycle//score/launch_manager:launch_manager": "FIT_LAUNCH_MANAGER_PATH", "@score_lifecycle//examples/rust_supervised_app:rust_supervised_app": "FIT_RUST_SUPERVISED_APP_PATH", @@ -46,26 +45,6 @@ } -def _repo_root() -> Path: - return Path(__file__).resolve().parents[2] - - -def _run(cmd: list[str]) -> str: - completed = subprocess.run( - cmd, - cwd=_repo_root(), - capture_output=True, - text=True, - check=False, - ) - if completed.returncode != 0: - raise RuntimeError( - f"Command failed (rc={completed.returncode}): {' '.join(cmd)}\n" - f"stdout:\n{completed.stdout}\nstderr:\n{completed.stderr}" - ) - return completed.stdout.strip() - - def _resolve_from_env(target: str) -> Path | None: """Resolve a target path from Bazel-provided runfile environment variables.""" env_var = _TARGET_ENV_MAP.get(target) @@ -108,17 +87,14 @@ def _resolve_target_path(target: str) -> Path: ) -def get_binary_path(target: str) -> Path: - """Compatibility helper used by daemon tests for bazel labels.""" - return _resolve_target_path(target) - - def pgrep_cmdline_pattern(binary_path: str) -> str: """Build POSIX ERE pattern matching binary with optional arguments.""" return rf"^{re.escape(binary_path)}([[:space:]]|$)" def is_running(binary_path: str | Path) -> bool: + """True if some process's cmdline starts with `binary_path` (pgrep). This is process + existence only, not launch_manager's reported Running state.""" result = subprocess.run( ["pgrep", "-f", pgrep_cmdline_pattern(str(binary_path))], capture_output=True, @@ -129,6 +105,7 @@ def is_running(binary_path: str | Path) -> bool: def first_pid(binary_path: str | Path) -> str | None: + """First pgrep match for `binary_path` (see `is_running`), or None.""" result = subprocess.run( ["pgrep", "-f", pgrep_cmdline_pattern(str(binary_path))], capture_output=True, @@ -142,6 +119,7 @@ def first_pid(binary_path: str | Path) -> str | None: def wait_until(predicate, timeout_s: float, interval_s: float = 0.2) -> bool: + """Poll `predicate` until it is truthy (True) or `timeout_s` elapses (False).""" deadline = time.time() + timeout_s while time.time() < deadline: if predicate(): @@ -150,29 +128,25 @@ def wait_until(predicate, timeout_s: float, interval_s: float = 0.2) -> bool: return False -# cap_kill: once the sandbox uid differs from the runner's (see `_generate_runtime_config`), -# launch_manager (still running as the runner uid) needs it to terminate/restart its own children. +# cap_setuid/cap_setgid: apply sandbox uid/gid; cap_sys_nice: apply non-SCHED_OTHER policies; +# cap_kill: launch_manager (runner uid) must signal children running as `_SANDBOX_UID`. _SETCAP_CAPS = "cap_setuid,cap_setgid,cap_sys_nice,cap_kill+ep" -# Sandbox uid used when capabilities are granted. It must differ from the runner's own uid, -# otherwise the uid/gid test cannot tell an applied sandbox identity from an inherited one -# (the config's 1001 is also the default first-user / GitHub-hosted-runner uid). +# Sandbox uid substituted when capabilities are granted. Must differ from the runner's uid, else +# the uid test cannot tell an applied identity from an inherited one (config's 1001 is commonly +# the runner's own uid). The gid is NOT remapped to a distinct value: see `_generate_runtime_config`. _SANDBOX_UID = 65533 -# Copies of `kill` (cap_kill) and `cat` (cap_sys_ptrace + cap_dac_read_search), staged when capabilities are granted, -# so the runner can signal apps running under `_SANDBOX_UID` and read their /proc//environ -# using only the setcap sudoers rule (no sudo kill). +# Capability-granted copies of `kill` (cap_kill) and `cat` (cap_sys_ptrace, cap_dac_read_search), +# staged per daemon by `_spawn_daemon` and deleted by `_teardown`. They let the runner signal apps +# running as `_SANDBOX_UID` and read their /proc//environ with only the setcap sudoers rule. _privileged_kill: Path | None = None _privileged_cat: Path | None = None def _mount_nosuid(path: Path) -> bool: - """Best-effort check whether `path` lives on a filesystem mounted `nosuid`. - - A `nosuid` mount silently strips file capabilities at exec time even when `setcap` - itself reports success, which otherwise looks identical to "grant never happened" - from the caller's point of view. - """ + """Best-effort check (via `findmnt`) whether `path` is on a `nosuid` mount, which drops + file capabilities at exec time. Used only to enrich a failed-grant diagnostic.""" try: findmnt = shutil.which("findmnt") if findmnt is None: @@ -193,25 +167,17 @@ def _grant_sandbox_capabilities( caps: str = _SETCAP_CAPS, required: tuple[str, ...] = ("cap_setuid", "cap_setgid"), ) -> tuple[bool, str]: - """Best-effort grant of the capabilities launch_manager needs to apply sandbox uid/gid - and scheduling policy without running as root. Returns `(granted, reason)`: `granted` - is a *verified* result (re-read via `getcap`, not just the setcap exit code) so tests - can key off a real, established precondition instead of assuming root; `reason` is a - human-readable diagnostic that is safe to surface directly in a pytest.skip() message. - - Requires CAP_SETFCAP to write the capability xattr, which a non-root test runner does not - have by default. Set FIT_ENABLE_SETCAP=1 to opt into a `sudo -n setcap` attempt, backed by - a passwordless sudoers rule scoped to the setcap binary (e.g. ` ALL=(root) NOPASSWD: - /usr/sbin/setcap`, with NO trailing arguments pinned — the target path is a fresh tmp_path - on every test run, so a rule that also pins the argument list will never match). Without - the flag, only a plain (non-sudo) setcap is tried, which only succeeds if the runner is - already root. - - Under `bazel test`, undeclared env vars (like FIT_ENABLE_SETCAP) do not reach the test - process unless passed via `--test_env=FIT_ENABLE_SETCAP=1` (NOT `--action_env`, which only - affects build actions). `bazel run` inherits the invoking shell's environment directly, so - `--action_env` is a no-op for this variable there; it is only needed to force a rebuild - when it affects action inputs, which it does not here. + """Best-effort `setcap caps binary_path`. Never raises. + + Returns `(granted, reason)`. `granted` is True only if `getcap` reads back every cap in + `required` (when `getcap` is available); `reason` is a diagnostic suitable for a skip message. + + Tries `sudo -n setcap` first when FIT_ENABLE_SETCAP=1 (needs a passwordless sudoers rule for + the setcap binary with no pinned arguments), then plain `setcap` (succeeds only as root). + Under `bazel test` the variable must be passed with `--test_env`, and the grant is inert + inside linux-sandbox (PR_SET_NO_NEW_PRIVS) even when setcap succeeds. + + Limitation: only `required` is verified; the other caps in `caps` are assumed granted with it. """ if shutil.which("setcap") is None: return False, "setcap binary not found on PATH" @@ -264,11 +230,11 @@ def _grant_sandbox_capabilities( def signal_process(pid: str, sig: str, *, sandbox_privileged: bool) -> tuple[bool, str]: - """Send `sig` (e.g. "-9", "-STOP", "-CONT") to `pid`, escalating via sudo if needed. + """Send `sig` (e.g. "-9", "-STOP", "-CONT") to `pid`. Returns `(sent, reason)`; never raises. - Under sandbox capabilities, supervised apps run as `_SANDBOX_UID`, not the runner's own - uid, so a plain `kill` fails. Falls back to the cap_kill `kill` copy staged by - `_spawn_daemon`, then to `sudo -n kill` when `FIT_ENABLE_SETCAP=1`. + Tries plain `kill`, then (if `sandbox_privileged`) the staged cap_kill copy, then + `sudo -n kill` when FIT_ENABLE_SETCAP=1. The sudo fallback only works with a sudoers rule + for `kill`, which the documented setup does not provide. """ attempts: list[list[str]] = [["kill", sig, pid]] if sandbox_privileged and _privileged_kill is not None: @@ -304,7 +270,7 @@ def _tmpdir_root() -> Path: @dataclass class ManagedDaemon: - """A subprocess wrapper with line-buffered output collection.""" + """A launch_manager subprocess plus the stdout/stderr lines its reader thread collected.""" process: subprocess.Popen[str] _lines: list[str] @@ -317,6 +283,8 @@ def pid(self) -> int: return self.process.pid def stop(self) -> None: + """SIGTERM the daemon's process group, SIGKILL after 5 s. Does not reach supervised + apps, which launch_manager moves into their own process groups (see `_teardown`).""" if self.is_running(): os.killpg(os.getpgid(self.process.pid), signal.SIGTERM) deadline = time.time() + 5.0 @@ -327,11 +295,11 @@ def stop(self) -> None: self.process.wait(timeout=5) def close_output(self) -> None: - """Join the stdout reader and close the pipe. + """Join the stdout reader (1 s) and close the pipe. - Launched apps inherit the daemon's stdout pipe and can outlive it, so call this only - after the apps are killed too; otherwise the reader is still blocked in read() holding - the buffer lock and close() would deadlock. + Call only after the supervised apps are dead: they inherit the pipe, so the reader sees + EOF only then. If the reader is still alive the pipe is left open (leaked) rather than + closed under it, which would deadlock. """ self._thread.join(timeout=1) if self.process.stdout is not None and not self._thread.is_alive(): @@ -357,27 +325,23 @@ def _generate_runtime_config( ready_timeout_s: float | None = None, crashes_before_success: int | None = None, ) -> None: - """Render and serialize an isolated launch-manager config for one daemon. - - Test variants are rendered from one template here rather than kept as copied config - files, so they cannot drift from the content other tests assert on: - `independent_apps` drops every component's `depends_on`, `ready_timeout_s` overrides - `defaults.deployment_config.ready_timeout`, and `crashes_before_success` fills the - `__FIT_CRASHES_BEFORE_SUCCESS__` process argument. - - Non-`SCHED_OTHER` scheduling policies need `CAP_SYS_NICE`; unlike the uid/gid sandbox - fields, launch_manager treats a failed `sched_setscheduler()` as fatal for the component - (triggering its `recovery_action` instead of just leaving the policy unapplied), which - would otherwise crash-loop every component in the config - not just the one a given test - cares about - whenever the capability grant wasn't obtained. When `sandbox_privileged` is - `False`, downgrade any configured non-default scheduling policy back to the harmless - `SCHED_OTHER`/`0` so the daemon still starts cleanly; scheduling-specific tests already key - off `sandbox_privileged` themselves and skip rather than assert against it in that case. - - `remap_sandbox_uid` rewrites every sandbox uid to `_SANDBOX_UID` and gid to the runner's own - gid (so the group bits of the runner-owned runtime dirs still grant access), and opens - `runtime_root` to that group. Read the rendered `etc_dir / "lifecycle_config.json"` for the - ids actually applied. + """Render `config_template` for one daemon and serialize it to `etc_dir`. + + Writes the rendered JSON to `etc_dir/lifecycle_config.json` (tests read expected values from + it) and the flatbuffer to `etc_dir/launch_manager_config.bin` via the upstream config mapper + and `flatc`. Raises RuntimeError with the tool's output if either step fails. + + Rendering always sets `bin_dir` to `runtime_root/bin` and the flaky-app counter path; optional + variants (rendered, not kept as copied config files, so they cannot drift): + - `independent_apps`: drop every component's `depends_on`. + - `ready_timeout_s`: override `defaults.deployment_config.ready_timeout`. + - `crashes_before_success`: fill the `__FIT_CRASHES_BEFORE_SUCCESS__` argument. + - `remap_sandbox_uid`: set every sandbox uid to `_SANDBOX_UID` and gid to the runner's gid, + and chmod `runtime_root` 0750. Limitation: the gid then equals the runner's, so a gid + check cannot prove setgid() was applied. + - `sandbox_privileged=False`: downgrade non-`SCHED_OTHER` policies to `SCHED_OTHER`/0. + launch_manager treats a failed sched_setscheduler() as fatal for the component, so without + CAP_SYS_NICE every such component would crash-loop. """ config = json.loads(_resolve_target_path(config_template).read_text(encoding="utf-8")) config["defaults"]["deployment_config"]["bin_dir"] = str(runtime_root / "bin") @@ -459,16 +423,16 @@ def _generate_runtime_config( ) -# launch_manager creates POSIX shm objects with deterministic names ("/ipc_shared_mem", -# "/_nudge~._.~me_") using O_CREAT|O_EXCL. POSIX shm lives on the host-wide /dev/shm tmpfs -# (not isolated by `unshare -i`), so a second daemon started before the first has shm_unlink'ed -# gets EEXIST and silently never launches its components. Daemon lifetimes must not overlap: -# within a process this registry enforces it; across Bazel test processes the lifecycle -# targets are tagged "exclusive". +# launch_manager creates POSIX shm objects with fixed names ("/ipc_shared_mem", "/_nudge~._.~me_") +# using O_CREAT|O_EXCL on the host-wide /dev/shm (not isolated by `unshare -i`). A second daemon +# started before the first has unlinked them gets EEXIST and silently launches nothing. So daemon +# lifetimes must not overlap: this registry enforces it within one pytest process; across Bazel +# test processes the lifecycle targets are tagged "exclusive". _live_daemons: list[ManagedDaemon] = [] def _assert_no_live_daemon() -> None: + """Fail the test if a daemon started by this process is still running.""" _live_daemons[:] = [d for d in _live_daemons if d.is_running()] if _live_daemons: pytest.fail( @@ -487,20 +451,24 @@ def _spawn_daemon( grant_sandbox_capabilities: bool = False, **config_options: Any, ) -> tuple[ManagedDaemon, bool, str]: - """Stage launch_manager plus `staged_binaries` (src, dst, mode), render its runtime - config (`config_options` are passed to `_generate_runtime_config`), and start it as a - supervised subprocess. - - Returns `(daemon, sandbox_privileged, sandbox_privileged_reason)`; the latter two are - `(False, "not requested")` unless `grant_sandbox_capabilities` is set. Fails the test via - `pytest.fail` if the daemon exits within the startup grace period, or if another daemon - started by this process is still alive (see `_live_daemons`). + """Copy launch_manager (0700) and `staged_binaries` (src, dst, mode) into place, render the + config (`config_options` go to `_generate_runtime_config`), and start launch_manager in its + own session with stdout+stderr collected by a reader thread. + + With `grant_sandbox_capabilities`, tries to setcap launch_manager; if that succeeds it also + stages the privileged `kill`/`cat` copies, and only when both are staged remaps the sandbox + uid. Returns `(daemon, sandbox_privileged, sandbox_privileged_reason)`. + + `pytest.fail`s if another daemon from this process is alive or if launch_manager exits within + the 1 s startup window; in the latter case (or any exception after Popen) it tears down the + daemon itself, since the caller never receives it. """ _assert_no_live_daemon() launch_manager = _resolve_target_path("@score_lifecycle//score/launch_manager:launch_manager") lm_dst = work_dir / "launch_manager" shutil.copy2(launch_manager, lm_dst) - lm_dst.chmod(0o755) + # 0700: once granted cap_setuid it must not be executable by other local users. + lm_dst.chmod(0o700) global _privileged_kill, _privileged_cat _privileged_kill = _privileged_cat = None @@ -509,13 +477,15 @@ def _spawn_daemon( else: sandbox_privileged, sandbox_privileged_reason = False, "not requested" if sandbox_privileged: - # Only move apps off the runner's uid if the runner can still signal them and read - # their /proc//environ afterwards. + # Remap the uid only if the runner can still signal the apps and read their environ. kill_tool, kill_reason = _stage_privileged_tool(work_dir, "kill", ("cap_kill",)) cat_tool, cat_reason = _stage_privileged_tool(work_dir, "cat", ("cap_sys_ptrace", "cap_dac_read_search")) if kill_tool is not None and cat_tool is not None: _privileged_kill, _privileged_cat = kill_tool, cat_tool else: + for tool in (kill_tool, cat_tool): + if tool is not None: + tool.unlink() sandbox_privileged_reason += f"; sandbox uid not remapped: {kill_reason}; {cat_reason}" for src, dst, mode in staged_binaries: @@ -558,11 +528,10 @@ def _collect_output() -> None: daemon = ManagedDaemon(process=process, _lines=lines, _thread=thread) _live_daemons.append(daemon) - # The caller never receives `daemon` if this raises, so tear it down here (daemon, any - # app it already launched, and its output pipe) rather than leaving a detached - # launch_manager running and blocking every later start via _live_daemons. + # The caller never receives `daemon` if this raises, so tear it down here; otherwise a + # detached launch_manager keeps running and blocks every later start via _live_daemons. try: - # Give startup a chance to complete and fail early if config is broken. + # Startup window: a broken config makes launch_manager exit within it. time.sleep(1.0) if not daemon.is_running(): pytest.fail(f"launch_manager failed to start. Logs:\n{daemon.get_logs()}") @@ -581,24 +550,20 @@ def start_launch_manager_daemon( independent_apps: bool = False, ready_timeout_s: float | None = None, ) -> dict[str, Any]: - """Start a real launch_manager process with generated flatbuffer config. - - `blocked_apps` names ("rust"/"cpp") are copied into place but left - non-executable, so launch_manager cannot start them until the caller - chmod's them back to 0o755. Used to exercise the dependency-gating - negative path: assert the dependent app stays down while its - dependency is withheld, then unblock and assert it starts - and, with - `independent_apps=True` (no depends_on between the two apps), the inverse: - assert the other app starts anyway, proving it isn't gated at all. - `independent_apps`/`ready_timeout_s` are rendered onto lifecycle_daemon_config.json - (see `_generate_runtime_config`). - - `stalled_apps` names are replaced by a shell stub that runs but never reports - Running, so launch_manager spends the full `ready_timeout` (and retries) on - them - unlike a blocked app, whose exec fails immediately. - - Each invocation receives its own directory beneath `TEST_TMPDIR`, but must not overlap - another live daemon (including the class-scoped fixture): see `_live_daemons`. + """Start launch_manager on lifecycle_daemon_config.json with rust_ and cpp_supervised_app. + + Always attempts the sandbox capability grant (see `_spawn_daemon`); check the returned + `sandbox_privileged` before asserting on uid/gid/scheduling. + - `blocked_apps` ("rust"/"cpp"): staged with mode 0000, so exec fails until the caller + chmods them 0755. Used to withhold a dependency. + - `stalled_apps`: replaced by a stub that runs but never reports Running, so launch_manager + waits out `ready_timeout` on it. + - `independent_apps`, `ready_timeout_s`: rendered into the config (`_generate_runtime_config`). + - `wait_for_apps`: `pytest.fail` unless every non-blocked app is running within 8 s. + Limitation: "running" means a matching process exists (pgrep), not launch_manager's Running state. + + Uses a fresh runtime root under `TEST_TMPDIR`; must not overlap another live daemon. + Returns the daemon info dict consumed by `stop_launch_manager_daemon`. """ runtime_root = Path(tempfile.mkdtemp(prefix="lifecycle_fit-", dir=_tmpdir_root())) @@ -618,8 +583,8 @@ def start_launch_manager_daemon( rust_supervised = _resolve_target_path("@score_lifecycle//examples/rust_supervised_app:rust_supervised_app") cpp_supervised = _resolve_target_path("@score_lifecycle//examples/cpp_supervised_app:cpp_supervised_app") stall_stub = work_dir / "stall_stub.sh" - # `exec -a "$0"` keeps the staged app path as argv[0], so is_running()/pkill's anchored - # cmdline pattern still matches the stub; a single exec'd process also dies on SIGTERM. + # `exec -a "$0"` keeps the staged app path as argv[0], so the anchored pgrep/pkill pattern + # matches the stub, and it stays a single process that dies on SIGTERM. stall_stub.write_text('#!/bin/bash\nexec -a "$0" sleep infinity\n', encoding="utf-8") staged_binaries = [ ( @@ -656,7 +621,7 @@ def start_launch_manager_daemon( f"{process_snapshot.stdout}{process_snapshot.stderr}" ) except BaseException: - # An app that did reach Running (e.g. on the wait timeout) outlives the daemon. + # Kill apps too: any that already started outlive the daemon (own process group). _teardown(daemon, list(apps.values()), runtime_root) raise @@ -676,14 +641,12 @@ def start_flaky_retry_daemon( tmp_path_factory: pytest.TempPathFactory, crashes_before_success: int, ) -> dict[str, Any]: - """Start launch_manager against the single-component lifecycle_daemon_retry_config.json. - - Drives `flaky_startup_app` (see support_apps/flaky_startup_app/main.cpp), which - aborts on its first `crashes_before_success` startup attempts and stays running - from then on, so `ready_recovery_action.restart.number_of_attempts` can be - exercised deterministically instead of relying on a real, racy startup failure. - Does not wait for the app to reach Running: whether it ever does is exactly - what the calling test is checking. + """Start launch_manager on lifecycle_daemon_retry_config.json with only `flaky_startup_app`. + + The app aborts on its first `crashes_before_success` launches and then stays up, counting + every launch in `counter_path`, so tests can observe `number_of_attempts` deterministically. + No capability grant is requested. Does not wait for the app: whether it ever runs is what + the caller checks. Returns the daemon info dict consumed by `stop_flaky_retry_daemon`. """ runtime_root = Path(tempfile.mkdtemp(prefix="lifecycle_fit_retries-", dir=_tmpdir_root())) bin_dir = runtime_root / "bin" @@ -701,9 +664,8 @@ def start_flaky_retry_daemon( ) staged_binaries = [(flaky_app, app_dst, 0o755)] + # runtime_root is a fresh mkdtemp, so the counter starts at 0. counter_path = runtime_root / "flaky_startup_app.counter" - if counter_path.exists(): - counter_path.unlink() daemon, _, _ = _spawn_daemon( work_dir, @@ -725,15 +687,17 @@ def start_flaky_retry_daemon( "counter_path": counter_path, "crashes_before_success": crashes_before_success, "runtime_root": runtime_root, + "runtime_config": etc_dir / "lifecycle_config.json", } def _teardown(daemon: ManagedDaemon | None, app_paths: list[Path], runtime_root: Path) -> None: - """Stop `daemon` (if started), pkill each of `app_paths` by cmdline, close the daemon's - output pipe, then clean up `runtime_root`. Every step runs even if an earlier one raises. + """Stop `daemon` (if any), SIGKILL each of `app_paths` by cmdline, close the daemon's output, + delete the privileged tool copies and remove `runtime_root`. Each step runs even if an + earlier one raises. - Supervised apps setpgid() into their own process group, so stopping the daemon's group - never reaches them; they must be killed explicitly. + Apps are killed explicitly because they run in their own process groups, which `stop()` + does not reach. """ try: if daemon is not None: @@ -745,15 +709,26 @@ def _teardown(daemon: ManagedDaemon | None, app_paths: list[Path], runtime_root: if daemon is not None: daemon.close_output() finally: + _drop_privileged_tools() _cleanup_runtime_root(runtime_root) +def _drop_privileged_tools() -> None: + """Delete the capability-granted `kill`/`cat` copies. `work_dir` is not removed (pytest keeps + recent basetemps), so they would otherwise stay on disk.""" + global _privileged_kill, _privileged_cat + for tool in (_privileged_kill, _privileged_cat): + if tool is not None: + tool.unlink(missing_ok=True) + _privileged_kill = _privileged_cat = None + + def _stage_privileged_tool(work_dir: Path, name: str, caps: tuple[str, ...]) -> tuple[Path | None, str]: - """Copy the external `name` binary into `work_dir` and grant it `caps` via the same setcap - path as launch_manager. Returns `(path, reason)`; `path` is None if the grant failed. + """Copy the `name` binary from PATH to `work_dir/privileged/name` (mode 0700, runner only) and + grant it `caps` via `_grant_sandbox_capabilities`. Returns `(path, reason)`; `path` is None + if the grant was not verified (the copy is then left for the caller to delete). - The copy keeps its basename (procps `kill` is multi-call and dispatches on argv[0]) and is - mode 0700: only the runner can execute it, and the runner can already `sudo -n setcap`. + The basename is kept because procps `kill` dispatches on argv[0]. """ src = shutil.which(name) if src is None: @@ -769,18 +744,24 @@ def _stage_privileged_tool(work_dir: Path, name: str, caps: tuple[str, ...]) -> def read_proc_file(pid: str, name: str) -> bytes: """Read `/proc//`, via the privileged `cat` copy when one is staged. - Files like `environ` need ptrace access, and after its setuid() the app is non-dumpable so - its /proc files are root-owned 0400; the runner lacks both once the app runs as `_SANDBOX_UID`. + Needed for files like `environ`: once the app runs as `_SANDBOX_UID` it is non-dumpable and + the runner lacks ptrace access to it. Raises RuntimeError (with stderr) if `cat` fails. """ path = f"/proc/{pid}/{name}" if _privileged_cat is None: return Path(path).read_bytes() - return subprocess.run([str(_privileged_cat), path], capture_output=True, check=True).stdout + result = subprocess.run([str(_privileged_cat), path], capture_output=True, check=False) + if result.returncode != 0: + raise RuntimeError( + f"Command failed (rc={result.returncode}): {_privileged_cat} {path}\n" + f"stderr:\n{result.stderr.decode(errors='replace')}" + ) + return result.stdout def _kill_app(app_path: Path) -> None: - """SIGKILL every process whose cmdline matches `app_path`, via the cap_kill `kill` copy - when one is staged (apps running as `_SANDBOX_UID` ignore a plain pkill: EPERM).""" + """SIGKILL every process whose cmdline starts with `app_path`. Uses the cap_kill `kill` copy + when staged, since a plain pkill gets EPERM for apps running as `_SANDBOX_UID`. Best effort.""" if _privileged_kill is None: subprocess.run( ["pkill", "-9", "-f", pgrep_cmdline_pattern(str(app_path))], capture_output=True, text=True, check=False diff --git a/feature_integration_tests/test_cases/lifecycle_scenario.py b/feature_integration_tests/test_cases/lifecycle_scenario.py index c4daf75b013..69ec6cc4af0 100644 --- a/feature_integration_tests/test_cases/lifecycle_scenario.py +++ b/feature_integration_tests/test_cases/lifecycle_scenario.py @@ -11,14 +11,13 @@ # SPDX-License-Identifier: Apache-2.0 # ******************************************************************************* """ -Helpers and base scenario classes for lifecycle feature integration tests. +Base classes for lifecycle FITs. -``LifecycleScenario`` is a ``FitScenario`` subclass that supplies the shared -``temp_dir`` fixture so individual test classes do not have to duplicate it. +``LifecycleScenario``: ``FitScenario`` base for the scenario-binary tests +(test_conditional_launching_scenario.py); adds the class-scoped ``temp_dir`` fixture. -``RetryDaemonScenario`` provides the equivalent class-scoped-fixture convention -for the flaky-retry daemon tests, which don't fit ``FitScenario`` (no scenario -binary, `command`, or `version` parametrization involved). +``RetryDaemonScenario``: base for the flaky-retry daemon tests (test_retry_exhaustion.py), +which run no scenario binary; adds the class-scoped ``retry_daemon`` fixture. """ from collections.abc import Generator @@ -30,11 +29,7 @@ class LifecycleScenario(FitScenario): - """ - Base class for lifecycle feature integration tests. - - Provides the ``temp_dir`` fixture shared by all lifecycle test classes. - """ + """Base for lifecycle scenario-binary test classes; provides ``temp_dir``.""" @pytest.fixture(scope="class") def temp_dir( @@ -43,7 +38,7 @@ def temp_dir( version: str, ) -> Generator[Path, None, None]: """ - Provide a temporary working directory for the lifecycle tests. + Per-class, per-version temporary directory for the scenario run. Parameters ---------- @@ -56,18 +51,17 @@ def temp_dir( class RetryDaemonScenario: - """ - Base class for flaky-retry launch_manager daemon lifecycle tests. + """Base for flaky-retry daemon test classes. - Subclasses set ``crashes_before_success``; the - ``retry_daemon`` fixture starts one launch_manager instance per test class - against `flaky_startup_app` and tears it down afterwards. + Subclasses set ``crashes_before_success``; ``retry_daemon`` starts one launch_manager per + class via ``daemon_helpers.start_flaky_retry_daemon`` and tears it down afterwards. """ crashes_before_success: int @pytest.fixture(scope="class") def retry_daemon(self, tmp_path_factory: pytest.TempPathFactory) -> Generator[dict[str, Any], None, None]: + # Lazy import: the scenario-lifecycle targets import this module without shipping daemon_helpers.py. from daemon_helpers import start_flaky_retry_daemon, stop_flaky_retry_daemon daemon_info = start_flaky_retry_daemon(tmp_path_factory, self.crashes_before_success) diff --git a/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD b/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD index 0eda8fe1688..4eb147a7b69 100644 --- a/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD +++ b/feature_integration_tests/test_cases/support_apps/flaky_startup_app/BUILD @@ -12,8 +12,8 @@ # ******************************************************************************* load("@rules_cc//cc:defs.bzl", "cc_binary") -# Plain "Native" supervised app (no launch_manager lifecycle API integration) used -# only to deterministically drive ready_recovery_action.restart in +# "Reporting" supervised app (calls report_running() via lifecycle_cc) used only to +# deterministically drive ready_recovery_action.restart in # lifecycle_daemon_retry_config.json. See main.cpp for behavior. cc_binary( name = "flaky_startup_app", diff --git a/feature_integration_tests/test_cases/tests/lifecycle/conftest.py b/feature_integration_tests/test_cases/tests/lifecycle/conftest.py index 73d67528884..1d49444b948 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/conftest.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/conftest.py @@ -22,7 +22,9 @@ @pytest.fixture(scope="class") def launch_manager_daemon(tmp_path_factory: pytest.TempPathFactory) -> dict[str, Any]: - """Start a real launch_manager process with generated flatbuffer config.""" + """One launch_manager (with both supervised apps running) per test class and `version` + param; see `daemon_helpers.start_launch_manager_daemon`. Tests using it must not start + another daemon.""" daemon_info = start_launch_manager_daemon(tmp_path_factory) try: yield daemon_info diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py b/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py index df41f52f62b..dcca057842c 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py @@ -11,10 +11,9 @@ # SPDX-License-Identifier: Apache-2.0 # ******************************************************************************* """ -Feature integration tests for conditional launching against a real Launch Manager. - -Unlike scenario-stub checks, these tests validate behavior from an actual -launch_manager process started with lifecycle daemon configuration. +Dependency-based launching FITs against a real launch_manager on lifecycle_daemon_config.json, +where rust_supervised_app `depends_on` cpp_supervised_app. (test_conditional_launching_scenario.py +only tests the FIT's own scenario stub.) """ import json @@ -33,7 +32,7 @@ @pytest.mark.parametrize("version", ["rust", "cpp"], scope="class") class TestConditionalLaunchingWithDaemon: - """Verify dependency-based conditional launching with real daemon behavior.""" + """Startup launch of each supervised app, per `version`, via the class-scoped fixture.""" @add_test_properties( partially_verifies=["feat_req__lifecycle__launch_support"], @@ -41,7 +40,7 @@ class TestConditionalLaunchingWithDaemon: derivation_technique="requirements-analysis", ) def test_startup_launches_conditioned_processes(self, launch_manager_daemon: dict[str, Any], version: str) -> None: - """Verify supervised processes are launched as part of conditional startup.""" + """The `version` app is running (pgrep) within 8 s of daemon start.""" daemon_info = launch_manager_daemon app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" app_path = str(daemon_info["apps"][version]) @@ -50,48 +49,17 @@ def test_startup_launches_conditioned_processes(self, launch_manager_daemon: dic assert started, f"{app_name} was not launched in conditional startup" -class TestConditionalLaunchingDependencyOrdering: - """Verify cpp-before-rust ordering is declared in config. - - Not parametrized on `version`: inspects the static config only, independent of - the scenario variant under test elsewhere. - - Real ordering/gating evidence lives in - TestConditionalLaunchingBlocksOnMissingDependency below - a start-tick - comparison used to live here but was near-vacuous (both processes launch - within the same ~10ms tick regardless of ordering) and was removed. - """ - - def test_dependency_is_declared_in_lifecycle_config(self) -> None: - """Verify runtime configuration defines rust conditional dependency on cpp.""" - config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" - config = json.loads(config_path.read_text(encoding="utf-8")) - - rust_component = config["components"]["rust_supervised_app"]["component_properties"] - depends_on = rust_component.get("depends_on", []) - assert "cpp_supervised_app" in depends_on, ( - "Expected rust_supervised_app to depend on cpp_supervised_app in lifecycle daemon config" - ) - - class TestConditionalLaunchingBlocksOnMissingDependency: - """Verify rust startup is actually gated on cpp, not merely correlated with it. + """rust startup is gated on cpp, observed by withholding cpp. - Runs its own launch_manager instance (rather than the shared class-scoped - `launch_manager_daemon` fixture) with cpp_supervised_app withheld, so it can - observe the negative case: rust must not start while its dependency cannot. - It lives in its own class so the class-scoped fixture is torn down first: overlapping - daemons collide on launch_manager's fixed POSIX shm names (daemon_helpers._live_daemons). - - Not parametrized on `version`: dependency gating is independent of which - scenario variant is under test elsewhere, so this runs exactly once. + Own daemon, in its own class so the class-scoped fixture is torn down first (fixed shm + names; see daemon_helpers._live_daemons). Not parametrized: runs once. """ @add_test_properties( partially_verifies=[ "feat_req__lifecycle__waitfor_support", "feat_req__lifecycle__dependency_check", - "feat_req__lifecycle__cond_process_start", "feat_req__lifecycle__process_ordering", "feat_req__lifecycle__define_swc_dependencies", ], @@ -101,7 +69,23 @@ class TestConditionalLaunchingBlocksOnMissingDependency: def test_rust_stays_down_until_cpp_dependency_becomes_available( self, tmp_path_factory: pytest.TempPathFactory ) -> None: - """Verify rust does not start while cpp is withheld, and does once cpp is unblocked.""" + """While cpp is non-executable, rust must stay down for 4 s and the daemon must log the cpp + launch failure at least twice (it keeps retrying rather than aborting). After cpp is made + executable, cpp and then rust must be running within 8 s each. + + Limitations: "running" is pgrep process existence; rust has a single dependency, so + `dependency_check` ("all dependencies") cannot be told apart from "any dependency". + No `cond_process_start` claim: that requirement is about starting on the return value + of earlier processes, which this config does not use. + """ + # Precondition: without this edge the negative check below would pass vacuously. + config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" + config = json.loads(config_path.read_text(encoding="utf-8")) + rust_depends_on = config["components"]["rust_supervised_app"]["component_properties"].get("depends_on", []) + assert "cpp_supervised_app" in rust_depends_on, ( + "Expected rust_supervised_app to depend on cpp_supervised_app in lifecycle daemon config" + ) + daemon_info = start_launch_manager_daemon( tmp_path_factory, blocked_apps=frozenset({"cpp"}), @@ -111,15 +95,14 @@ def test_rust_stays_down_until_cpp_dependency_becomes_available( cpp_path = daemon_info["apps"]["cpp"] rust_path = str(daemon_info["apps"]["rust"]) - # cpp cannot execute (mode 0o000): rust must not appear while it is withheld. + # cpp is mode 0000, so its exec fails: rust must not appear meanwhile. rust_started_early = wait_until(lambda: is_running(rust_path), timeout_s=4.0) assert not rust_started_early, ( "rust_supervised_app started even though its cpp_supervised_app dependency " "was withheld (non-executable); dependency gating was not enforced" ) - # Repeated, path-specific launch failures demonstrate the daemon keeps - # processing the unavailable dependency rather than aborting once. + # >= 2 path-specific launch failures: the daemon keeps retrying cpp, not giving up once. cpp_launch_failure = f"File does not exist or is not executable: {cpp_path}" failures_observed = wait_until( lambda: daemon_info["daemon"].get_logs().count(cpp_launch_failure) >= 2, @@ -132,8 +115,7 @@ def test_rust_stays_down_until_cpp_dependency_becomes_available( ) ) - # Once cpp becomes executable and reaches Running, its dependent rust app - # should be released as well. + # Unblock cpp: it, then its dependent rust, must start. cpp_path.chmod(0o755) cpp_started = wait_until(lambda: is_running(cpp_path), timeout_s=8.0) assert cpp_started, "cpp_supervised_app did not start after becoming executable" diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching_scenario.py b/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching_scenario.py index 6f9343213f3..b54d66eeeea 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching_scenario.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching_scenario.py @@ -10,19 +10,12 @@ # # SPDX-License-Identifier: Apache-2.0 # ******************************************************************************* -"""Scenario-level smoke tests for the conditional-launching test-scenario binary. - -These exercise the bespoke wait-condition poller in test_scenarios/{rust,cpp}/.../lifecycle/ -conditional_launching.{rs,cpp} directly: preconditions (a path, an env var, a running process) -are really established or really withheld, so the assertions verify that *this stub* observes -and enforces them, not merely that it echoes back what was configured. - -This is a fact about the FIT's own test code, not about launch_manager - no test here starts -or drives an actual launch_manager instance, so none of them verify a `feat_req__lifecycle__*` -requirement of the lifecycle module. That verification belongs to the daemon-driven tests in -test_conditional_launching.py / test_process_launching_with_daemon.py, or a future test that -exercises this same wait-condition logic through launch_manager's real config. Hence no add -`@add_test_properties(partially_verifies=[...])` claims have been added to classes in this file. +"""Tests of the FIT's own conditional-launching scenario binary (rust and cpp), not of launch_manager. + +The scenario polls `path:`, `env:` and `process:` wait conditions; these tests really create or +withhold each condition and check what the stub reports. No test here drives launch_manager, so +none carries a `partially_verifies` claim: lifecycle requirement coverage lives in +test_conditional_launching.py and test_process_launching_with_daemon.py. """ import os @@ -45,7 +38,7 @@ class TestConditionalLaunchingScenario(LifecycleScenario): - """Verify the scenario actually waits for and detects satisfied conditions.""" + """All three conditions are really satisfied before the scenario starts.""" @pytest.fixture(scope="class") def scenario_name(self) -> str: @@ -57,13 +50,8 @@ def flag_path(self, temp_dir: Path) -> Path: @pytest.fixture(scope="class", autouse=True) def satisfied_preconditions(self, flag_path: Path) -> Generator[None, None, None]: - """Really establish the preconditions the scenario is told to wait for. - - The flag file is created up front (path condition already met), the env var is - set in this process (inherited by the scenario subprocess), and a real `sleep` - process is kept alive for the duration of the scenario run (process condition). - Torn down afterwards so this class does not leak state into later tests. - """ + """Create the flag file, set the env var (inherited by the scenario) and keep a `sleep` + process alive for the class; all three are undone afterwards.""" flag_path.write_text("ready", encoding="utf-8") os.environ[_CONDITION_ENV_VAR] = "1" process = subprocess.Popen([_CONDITION_PROCESS_NAME, "30"]) @@ -77,8 +65,7 @@ def satisfied_preconditions(self, flag_path: Path) -> Generator[None, None, None @pytest.fixture(scope="class") def test_config(self, flag_path: Path, satisfied_preconditions: None) -> dict[str, Any]: - # Depends on `satisfied_preconditions` explicitly (rather than relying on autouse - # ordering) so preconditions are guaranteed established before `results` executes. + # Explicit dependency so the preconditions exist before `results` runs the scenario. return { "test": { "wait_conditions": [ @@ -96,7 +83,7 @@ def test_conditions_already_satisfied_allow_immediate_success( results: ScenarioResult, version: str, ) -> None: - """Verify the scenario succeeds once path/env/process conditions are all really met.""" + """The scenario exits successfully.""" assert results.return_code == ResultCode.SUCCESS, ( f"Expected success with satisfied preconditions, got: {results}" ) @@ -107,7 +94,7 @@ def test_each_condition_is_individually_confirmed_satisfied( flag_path: Path, version: str, ) -> None: - """Verify the scenario reports each condition as satisfied, not just configured.""" + """The scenario logs "Condition satisfied" for each condition and "All dependencies satisfied".""" expected_messages = [ f"Condition satisfied: path:{flag_path}", f"Condition satisfied: env:{_CONDITION_ENV_VAR}", @@ -118,23 +105,20 @@ def test_each_condition_is_individually_confirmed_satisfied( log = logs_info_level.find_log("message", value=expected) assert log is not None, f"Expected scenario to log: {expected}" - def test_timeout_and_polling_interval_are_honored( + def test_timeout_and_polling_interval_are_logged( self, logs_info_level: Any, version: str, ) -> None: - """Verify the scenario logs the configured wait timing values.""" + """The scenario logs the configured polling interval and timeout. Only the logged values + are checked; that polling actually uses them is covered by the late-condition test.""" assert logs_info_level.find_log("message", value="Polling interval: 50ms") is not None assert logs_info_level.find_log("message", value="Condition timeout: 2000ms") is not None class TestConditionalLaunchingScenarioTimesOutOnUnmetConditions(LifecycleScenario): - """Verify the scenario fails when its wait conditions are never satisfied. - - Without this, an implementation that always reports success regardless of whether - a path exists, an env var is set, or a process is running would still pass the - happy-path test above. - """ + """No condition is ever satisfied: the scenario must fail with a timeout. Catches a stub that + reports success without checking the conditions.""" @pytest.fixture(scope="class") def scenario_name(self) -> str: @@ -162,8 +146,7 @@ def capture_stderr(self) -> bool: return True def test_scenario_fails_when_conditions_stay_unmet(self, results: ScenarioResult, version: str) -> None: - """Verify the scenario reports failure - and specifically a wait-condition timeout, - not merely any nonzero exit - when conditions are never satisfied.""" + """Non-success exit with a wait-condition timeout ("Timed out" ... "condition") on stderr.""" assert results.return_code != ResultCode.SUCCESS, ( f"Expected failure when wait conditions are never satisfied, got: {results}" ) @@ -174,14 +157,8 @@ def test_scenario_fails_when_conditions_stay_unmet(self, results: ScenarioResult class TestConditionalLaunchingScenarioDetectsConditionArrivingLate(LifecycleScenario): - """Verify the scenario is actually re-checking the condition over time (real polling), - rather than only ever observing the condition's state at process start (t=0) or its - absence at the very end (timeout). - - Without this, an implementation that checks the condition exactly once - either at - the very start or only right before giving up - would still pass both the - already-satisfied test and the never-satisfied timeout test above. - """ + """The path condition becomes true 0.5 s into a 3 s wait. Catches a stub that checks only once + (at start or at timeout) instead of polling.""" _DELAY_BEFORE_CONDITION_MET_S = 0.5 _TIMEOUT_MS = 3000 @@ -212,19 +189,10 @@ def results( execution_timeout: float, flag_path: Path, ) -> Generator[ScenarioResult, None, None]: - # Overrides the base class-scoped `results` fixture so the condition is armed before - # the command runs. The base fixture is also pulled in by the autouse `print_to_report` - # fixture ahead of the test body, so starting the trigger inside the test method itself - # is too late: the base fixture would already have run the command to completion and - # timed out before the test method ever executes. - # - # Use a separate subprocess to write the flag after the configured delay instead of a - # `threading.Timer`. Under heavy Bazel load, thread creation and scheduling can itself - # consume a meaningful slice of the delay budget, and the test is specifically checking - # that a condition becomes true mid-wait, not that a Python timer thread can fire before - # the OS reschedules it. A tiny helper subprocess makes the delay deterministic and - # independent of the test runner's thread scheduler while still exercising the real - # polling logic in the scenario under test. + # Overrides the base `results` fixture so the delayed flag writer starts together with the + # scenario; the autouse report fixture runs `results` before the test body, so arming it + # in the test would be too late. A helper subprocess (not a Python timer thread) writes + # the flag, so the delay does not depend on the test runner's thread scheduling. start = time.monotonic() trigger = subprocess.Popen( [ @@ -241,9 +209,7 @@ def results( ) try: result = self._run_command(command, execution_timeout) - # A pytest fixture's `self` and a test method's `self` are different instances of - # the test class, so state can't be handed off via an instance attribute; stash it - # on the class object instead, which both share. + # Fixture and test get different instances, so share the timing via the class. type(self)._elapsed_s = time.monotonic() - start finally: trigger.terminate() @@ -260,9 +226,7 @@ def test_condition_satisfied_partway_through_the_wait_is_detected_promptly( results: ScenarioResult, version: str, ) -> None: - """Verify the scenario succeeds shortly after the condition becomes true mid-wait, - not merely at t=0 or by coincidentally still being true once the full timeout - elapses.""" + """Success no earlier than the 0.5 s delay and well before the 3 s timeout.""" result = results elapsed_s = self._elapsed_s @@ -280,8 +244,7 @@ def test_condition_satisfied_partway_through_the_wait_is_detected_promptly( class TestConditionalLaunchingScenarioRejectsUnsupportedPrefix(LifecycleScenario): - """Verify an unsupported wait-condition prefix is rejected as invalid configuration, - distinct from a legitimate condition that simply times out unmet.""" + """An unknown wait-condition prefix is a configuration error, reported without waiting.""" @pytest.fixture(scope="class") def scenario_name(self) -> str: @@ -307,15 +270,11 @@ def capture_stderr(self) -> bool: @pytest.fixture(scope="class") def results(self, command: list[str], execution_timeout: float) -> ScenarioResult: - # Overrides the base class-scoped `results` fixture to time the single command - # invocation. Without this override, the test body's own `self._run_command(...)` - # call would run the scenario binary a *second* time on top of the one the autouse - # `print_to_report` -> `logs` -> `results` chain already ran ahead of the test. + # Overrides the base `results` fixture to time the one scenario run (a second run in the + # test body would execute the binary twice). Fixture and test get different instances, + # so the timing is shared via the class. start = time.monotonic() result = self._run_command(command, execution_timeout) - # A pytest fixture's `self` and a test method's `self` are different instances of - # the test class, so state can't be handed off via an instance attribute; stash it - # on the class object instead, which both share. type(self)._elapsed_s = time.monotonic() - start return result @@ -324,8 +283,7 @@ def test_unsupported_prefix_is_rejected_immediately( results: ScenarioResult, version: str, ) -> None: - """Verify validation rejects the condition outright, before entering the wait loop, - rather than only surfacing the same error after waiting out the full timeout.""" + """Non-success exit with "Unsupported wait condition prefix" on stderr, in under half the timeout.""" result = results elapsed_s = self._elapsed_s @@ -343,14 +301,8 @@ def test_unsupported_prefix_is_rejected_immediately( class TestConditionalLaunchingScenarioRejectsEmptyConditions(LifecycleScenario): - """Verify an empty wait_conditions list is rejected as invalid configuration up front, - distinct from a legitimate condition that simply times out unmet. - - Without this, a future refactor of parse_wait_conditions/parse_string_array_field could - silently start treating an empty list as "nothing to wait for, immediate success" instead - of a configuration error - a regression this test would otherwise be the only thing to - catch, since the happy-path and unsupported-prefix tests never exercise this branch. - """ + """An empty `wait_conditions` list is a configuration error (not "nothing to wait for"), + reported without waiting.""" @pytest.fixture(scope="class") def scenario_name(self) -> str: @@ -376,9 +328,7 @@ def capture_stderr(self) -> bool: @pytest.fixture(scope="class") def results(self, command: list[str], execution_timeout: float) -> ScenarioResult: - # See TestConditionalLaunchingScenarioRejectsUnsupportedPrefix.results: overrides the - # base class-scoped `results` fixture so the test body doesn't run the scenario binary - # a second time on top of the one the autouse fixture chain already ran. + # Same override as TestConditionalLaunchingScenarioRejectsUnsupportedPrefix.results. start = time.monotonic() result = self._run_command(command, execution_timeout) type(self)._elapsed_s = time.monotonic() - start @@ -389,8 +339,7 @@ def test_empty_conditions_are_rejected_immediately( results: ScenarioResult, version: str, ) -> None: - """Verify validation rejects an empty wait_conditions list outright, before entering - the wait loop, rather than only surfacing the same error after the full timeout.""" + """Non-success exit with "Wait conditions were not provided" on stderr, in under half the timeout.""" result = results elapsed_s = self._elapsed_s diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py b/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py index a9530989014..1862d530f51 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py @@ -11,28 +11,20 @@ # SPDX-License-Identifier: Apache-2.0 # ******************************************************************************* """ -Feature integration tests for lifecycle with running Launch Manager daemon. +Lifecycle FITs against a real launch_manager supervising rust_supervised_app and cpp_supervised_app. -These tests validate actual supervision and lifecycle management behavior -by running test applications under a real Launch Manager daemon instance. +`version` selects which of the two supervised apps a test inspects; both always run under the +same daemon. Run via Bazel (the helpers resolve binaries from the target's FIT_*_PATH env vars): -To run these tests: - - # Run both Rust and C++ variants - pytest feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py -v - - # Run only Rust variant - pytest feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py -v -k rust - - # Run only C++ variant - pytest feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py -v -k cpp + bazel test //feature_integration_tests/test_cases:fit_lifecycle_daemon + bazel test //feature_integration_tests/test_cases:fit_lifecycle_daemon --test_arg=-k --test_arg=rust """ import json +import os import re import subprocess import time -import os from pathlib import Path from typing import Any @@ -54,25 +46,19 @@ class TestProcessLaunchingWithDaemon: - """ - Verify lifecycle management with running Launch Manager daemon. - - These tests demonstrate end-to-end integration including: - - Process launching under supervision - - Execution state reporting to the daemon - - Process monitoring and health checks - - Recovery actions on failure - """ + """Launch-parameter checks (args, env, uid/gid, scheduling, non-root) against one daemon per + `version`, provided by the class-scoped `launch_manager_daemon` fixture. Tests here must not + start their own daemon (fixed shm names; see daemon_helpers._live_daemons).""" @staticmethod def _proc_cmdline(pid: str) -> list[str]: - """Read process cmdline from /proc and split NUL-separated arguments.""" + """Return `/proc//cmdline` as a list of arguments.""" raw = Path(f"/proc/{pid}/cmdline").read_bytes() return [arg.decode("utf-8") for arg in raw.split(b"\0") if arg] @staticmethod def _proc_environ(pid: str) -> dict[str, str]: - """Read process environment from /proc as a key/value mapping.""" + """Return `/proc//environ` as a dict (via the privileged `cat` when staged).""" raw = read_proc_file(pid, "environ") env: dict[str, str] = {} for item in raw.split(b"\0"): @@ -86,7 +72,7 @@ def _proc_environ(pid: str) -> dict[str, str]: @staticmethod def _proc_status_ids(pid: str) -> tuple[int, int] | None: - """Read effective uid/gid from /proc status for a process.""" + """Return the effective `(uid, gid)` from `/proc//status`, or None if unreadable.""" try: lines = Path(f"/proc/{pid}/status").read_text(encoding="utf-8").splitlines() except OSError: @@ -105,7 +91,7 @@ def _proc_status_ids(pid: str) -> tuple[int, int] | None: @staticmethod def _proc_sched_policy_and_priority(pid: str) -> tuple[str, int] | None: - """Read scheduler policy and RT priority from chrt output for a process.""" + """Return `(policy, priority)` parsed from `chrt -p `, or None on failure.""" result = subprocess.run(["chrt", "-p", pid], capture_output=True, text=True, check=False) if result.returncode != 0: return None @@ -127,8 +113,7 @@ def _proc_sched_policy_and_priority(pid: str) -> tuple[str, int] | None: return None return policy, priority - # Dependency-gating coverage (rust-on-cpp startup order) lives in - # test_conditional_launching.py; not duplicated here. + # Dependency gating (rust waits for cpp) is covered in test_conditional_launching.py. @add_test_properties( partially_verifies=["feat_req__lifecycle__launch_support"], @@ -140,7 +125,8 @@ def test_startup_declares_and_launches_multiple_processes( launch_manager_daemon: dict[str, Any], version: str, ) -> None: - """Verify startup run target includes multiple processes and both are launched.""" + """Both processes in the Startup run target's `depends_on` are running (pgrep) and the + daemon is still up. `version` is unused: the check covers both apps.""" config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" config = json.loads(config_path.read_text(encoding="utf-8")) startup_deps = config["run_targets"]["Startup"]["depends_on"] @@ -171,11 +157,8 @@ def test_launch_process_arguments_are_applied( launch_manager_daemon: dict[str, Any], version: str, ) -> None: - """Verify launched process cmdline includes every configured lifecycle argument. - - Expected args are read from the config itself, not hardcoded, so the test - tracks config drift. - """ + """Every configured `process_arguments` entry is present in the live app's cmdline. + Expected values come from the config, so the test tracks config changes.""" daemon_info = launch_manager_daemon app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" app_path = str(daemon_info["apps"][version]) @@ -208,11 +191,8 @@ def test_launch_process_environment_is_applied( launch_manager_daemon: dict[str, Any], version: str, ) -> None: - """Verify launched process environment matches every configured environment variable. - - Expected variables are read from the config itself, not hardcoded, so the - test tracks config drift. - """ + """Every configured `environmental_variables` entry has the configured value in the live + app's /proc environ. Expected values come from the config.""" daemon_info = launch_manager_daemon app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" app_path = str(daemon_info["apps"][version]) @@ -234,38 +214,19 @@ def test_launch_process_environment_is_applied( f"{key} mismatch for {app_name}: expected {expected_value!r}, got {proc_env.get(key)!r}" ) - def test_config_defines_uid_gid_scheduling_and_priority(self, version: str) -> None: - """Sanity-check the lifecycle config's shape for launch user/group and scheduling defaults. - - No requirement tag here: this only confirms the config file is well-formed, not - that launch_manager applies it - that's covered by - test_launched_process_uid_gid_matches_config_when_applied and - test_launched_process_scheduling_matches_config_when_applied below, which inspect - the real launched process. `version` is unused but required by the class-scope - parametrize on this class. - """ - config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" - config = json.loads(config_path.read_text(encoding="utf-8")) - - sandbox = config["defaults"]["deployment_config"]["sandbox"] - assert isinstance(sandbox.get("uid"), int), "Expected integer uid in sandbox defaults" - assert isinstance(sandbox.get("gid"), int), "Expected integer gid in sandbox defaults" - assert isinstance(sandbox.get("scheduling_priority"), int), "Expected integer scheduling priority" - assert isinstance(sandbox.get("scheduling_policy"), str), "Expected scheduling policy string" - - # Not decorated with @add_test_properties: this test is unconditionally skipped in CI/CD - # (see below), so it never actually exercises feat_req__lifecycle__uid_gid_support and - # shouldn't claim to verify it until it can run there. - # Skipped in CI/CD (both rust/cpp): requires launch_manager to gain cap_setuid/cap_setgid via - # setcap, which needs both FIT_ENABLE_SETCAP=1 (unset in the GitHub Actions workflow) and - # unsandboxed execution (linux-sandbox's PR_SET_NO_NEW_PRIVS makes the grant inert at exec - # time even if setcap itself succeeds). See feature_integration_tests/README.md for details. + # No requirement claim (feat_req__lifecycle__uid_gid_support): skips in CI, which neither sets + # FIT_ENABLE_SETCAP=1 nor runs unsandboxed, both needed for the setcap grant. See README.md. def test_launched_process_uid_gid_matches_config_when_applied( self, launch_manager_daemon: dict[str, Any], version: str, ) -> None: - """Verify launched process runs with configured effective uid/gid when runtime applies sandbox identity.""" + """With capabilities granted, the live app's effective uid equals the rendered sandbox uid + and differs from the runner's. Skips without the grant, or if the uid was not remapped. + + Limitation: the rendered gid is the runner's own (see `_generate_runtime_config`), so the + gid assertion cannot prove setgid() was applied. + """ daemon_info = launch_manager_daemon if not daemon_info["sandbox_privileged"]: pytest.skip( @@ -284,7 +245,7 @@ def test_launched_process_uid_gid_matches_config_when_applied( expected_uid = int(sandbox["uid"]) expected_gid = int(sandbox["gid"]) if expected_uid == os.getuid(): - # Only when the cap_kill grant failed (uid not remapped); matching uids would pass + # Happens when the kill/cat copies could not be staged (uid not remapped); matching uids pass # whether or not launch_manager applied the sandbox identity. pytest.skip( f"Sandbox uid {expected_uid} equals the test runner's own uid; cannot distinguish an " @@ -303,6 +264,7 @@ def test_launched_process_uid_gid_matches_config_when_applied( assert effective_uid == expected_uid, ( f"Effective uid mismatch for {app_name}: expected {expected_uid}, got {effective_uid}" ) + # Consistency check only: passes whether or not setgid() was applied (gid == runner's). assert effective_gid == expected_gid, ( f"Effective gid mismatch for {app_name}: expected {expected_gid}, got {effective_gid}" ) @@ -311,19 +273,14 @@ def test_launched_process_uid_gid_matches_config_when_applied( "identity was not actually applied" ) - # Not decorated with @add_test_properties: this test is unconditionally skipped in CI/CD - # (see below), so it never actually exercises feat_req__lifecycle__launch_priority_support / - # feat_req__lifecycle__scheduling_policy and shouldn't claim to verify them until it can run there. - # Skipped in CI/CD (both rust/cpp): requires launch_manager to gain cap_sys_nice via setcap, - # which needs both FIT_ENABLE_SETCAP=1 (unset in the GitHub Actions workflow) and unsandboxed - # execution (linux-sandbox's PR_SET_NO_NEW_PRIVS makes the grant inert at exec time even if - # setcap itself succeeds). See feature_integration_tests/README.md for details. + # No requirement claim (launch_priority_support / scheduling_policy): skips in CI, see above. def test_launched_process_scheduling_matches_config_when_applied( self, launch_manager_daemon: dict[str, Any], version: str, ) -> None: - """Verify launched process uses configured scheduler policy and priority when applied.""" + """With capabilities granted, the live app's `chrt` policy and priority equal the rendered + sandbox values. Skips without the grant (the config is then downgraded to SCHED_OTHER).""" daemon_info = launch_manager_daemon if not daemon_info["sandbox_privileged"]: pytest.skip( @@ -334,8 +291,8 @@ def test_launched_process_scheduling_matches_config_when_applied( app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" app_path = str(daemon_info["apps"][version]) - config_path = Path(__file__).resolve().parents[3] / "configs" / "lifecycle_daemon_config.json" - config = json.loads(config_path.read_text(encoding="utf-8")) + # The rendered config, not the source JSON, like the uid/gid test above. + config = json.loads(daemon_info["runtime_config"].read_text(encoding="utf-8")) component_sandbox = config["components"][app_name].get("deployment_config", {}).get("sandbox") sandbox = component_sandbox or config["defaults"]["deployment_config"]["sandbox"] configured_policy = sandbox["scheduling_policy"] @@ -344,8 +301,7 @@ def test_launched_process_scheduling_matches_config_when_applied( started = wait_until(lambda: is_running(app_path), timeout_s=8.0) assert started, f"{app_name} was not launched before scheduling verification" - # Pid can go stale between resolution and the chrt call if the app restarts; - # retry against a fresh pid rather than failing on that race. + # Retry with a fresh pid in case the app restarted between pgrep and chrt. sched = None pid = None for _ in range(20): @@ -369,22 +325,16 @@ def test_launched_process_scheduling_matches_config_when_applied( f"Scheduling priority mismatch for {app_name}: expected {configured_priority}, got {rt_priority}" ) - # Not decorated with @add_test_properties: like the two tests above, this is unconditionally - # skipped in CI/CD (no cap_sys_nice grant), so it never actually exercises - # feat_req__lifecycle__scheduling_policy / feat_req__lifecycle__launch_priority_support there. + # No requirement claim: skips in CI, see above. def test_scheduling_policy_is_non_default_and_applied( self, launch_manager_daemon: dict[str, Any], version: str, ) -> None: - """Verify the launched process's scheduling policy differs from launch_manager's own. - - Both apps carry a non-default sandbox policy (`rust_supervised_app`: `SCHED_RR`/`10`, - `cpp_supervised_app`: `SCHED_FIFO`/`20`) vs. the OS default `SCHED_OTHER`/`0` that - launch_manager itself runs under. Comparing the launched app against the daemon's own - scheduling, rather than against a fixed constant, proves the policy was actually applied - rather than coincidentally matching the process's inherited default. - """ + """With capabilities granted, the live app's `(policy, priority)` differs from + launch_manager's own, so it was set by launch_manager rather than inherited. The config + gives rust SCHED_RR/10 and cpp SCHED_FIFO/20; the daemon runs SCHED_OTHER/0. Skips + without the grant.""" daemon_info = launch_manager_daemon if not daemon_info["sandbox_privileged"]: pytest.skip( @@ -430,14 +380,21 @@ def test_launch_manager_and_apps_are_not_running_as_root( launch_manager_daemon: dict[str, Any], version: str, ) -> None: - """Verify launch setup executes without root privileges in this integration setup.""" + """launch_manager itself runs with a non-root effective uid and still launches the app + (the requirement: LM can be started as non-root). Also checks the app is non-root. + + Limitation: covers only a plain non-root start (optionally with file capabilities), not + any other "security policy" mechanism. + """ daemon_info = launch_manager_daemon daemon = daemon_info["daemon"] app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" app_path = str(daemon_info["apps"][version]) - assert os.geteuid() != 0, "Test environment unexpectedly runs as root" - assert daemon.pid() > 0, "Launch Manager daemon pid should be available" + assert daemon.is_running(), "Launch Manager daemon is not running" + daemon_ids = self._proc_status_ids(str(daemon.pid())) + assert daemon_ids is not None, f"Could not read /proc status for launch_manager pid={daemon.pid()}" + assert daemon_ids[0] != 0, "launch_manager is running as root" started = wait_until(lambda: is_running(app_path), timeout_s=8.0) assert started, f"{app_name} was not launched before non-root verification" @@ -451,11 +408,10 @@ def test_launch_manager_and_apps_are_not_running_as_root( class TestSupervisedAppRecovery: - """Kill-and-restart recovery against a dedicated launch_manager instance. + """Kill-and-recover against a dedicated daemon per `version`. - Kept out of TestProcessLaunchingWithDaemon so its own daemon never overlaps that class's - `launch_manager_daemon` fixture: concurrent daemons collide on launch_manager's fixed - POSIX shm names (see daemon_helpers._live_daemons). + Separate class so its daemon never overlaps the `launch_manager_daemon` fixture (fixed shm + names; see daemon_helpers._live_daemons). """ @add_test_properties( @@ -471,14 +427,14 @@ def test_supervised_app_recovery( tmp_path_factory: pytest.TempPathFactory, version: str, ) -> None: - """Verify daemon restarts a killed supervised app in place per the retry policy. - - Also confirms the other supervised app is left untouched, proving recovery - went through `ready_recovery_action.restart` rather than a run-target switch. - - Does not claim `feat_req__lifecycle__retries_configurable`: this only exercises a - single restart within the configured attempt budget, it never varies or exhausts - `number_of_attempts`, so the "configurable" half of that requirement is unverified. + """SIGKILL the running app; launch_manager must log its unexpected termination and bring + it back (new pid) without restarting the other, healthy app or dying itself. + + The recovery that runs is the run target's `recovery_action` (switch to + `fallback_run_target`, which contains both apps), not `ready_recovery_action.restart`, + which only covers startup failures. Limitation: the test does not assert which recovery + action ran, only that the app recovered. Does not claim `retries_configurable` + (see test_retry_exhaustion.py). """ daemon_info = start_launch_manager_daemon(tmp_path_factory) try: @@ -504,34 +460,27 @@ def test_supervised_app_recovery( timeout_s=12.0, ) assert restarted, f"{app_name} was not restarted after forced termination" + termination_logged = re.search( + rf"unexpected termination of process\s+{re.escape(app_name)}\b", daemon.get_logs() + ) + assert termination_logged, ( + f"launch_manager did not log the abnormal termination of {app_name}.\nDaemon logs:\n{daemon.get_logs()}" + ) assert daemon.is_running(), "Launch Manager daemon should still be running after recovery" other_new_pid = first_pid(other_app_path) assert other_new_pid == other_old_pid, ( - "The other supervised app was relaunched too, indicating recovery switched the " - "whole run target instead of retrying only the failed app per the configured " - "restart policy" + "The other, healthy supervised app was relaunched too; recovery should only relaunch the failed app" ) finally: stop_launch_manager_daemon(daemon_info) class TestParallelLaunch: - """Verify genuinely parallel launch of independent components. - - Runs its own launch_manager instance (rather than the shared class-scoped - `launch_manager_daemon` fixture used by TestProcessLaunchingWithDaemon), for two reasons: - - 1. It renders the config with `independent_apps=True` (no depends_on between the - two apps), unlike the shared fixture's config - that's the whole point. - 2. It is a separate class, so the shared fixture is torn down before this daemon starts; - overlapping daemons collide on launch_manager's fixed POSIX shm names. - - Parametrized on `version` only because the module-level `pytestmark` applies it - to every class in this file; parallel launch itself is independent of which - scenario variant is under test elsewhere, so `version` is unused here. - """ + """Parallel launch of independent components, with its own daemons rendered with + `independent_apps=True`. `version` is unused (module-level parametrize), so the test runs + twice with identical behaviour.""" @add_test_properties( partially_verifies=["feat_req__lifecycle__parallel_launch_support"], @@ -543,24 +492,13 @@ def test_independent_processes_launch_without_waiting_on_each_other( tmp_path_factory: pytest.TempPathFactory, version: str, ) -> None: - """Verify two independent components launch in parallel, not one-after-the-other. - - `lifecycle_daemon_config.json` has rust_supervised_app depend on - cpp_supervised_app, so it cannot demonstrate parallel launch - both apps - eventually running there is equally consistent with strict serialization. - - Renders that config with `independent_apps=True`, so neither app depends on - the other, and stalls one app at a time: it is replaced by a - stub that runs but never reports Running, so a serialized launcher would sit - on it for the full `ready_timeout` (10 s, plus retries) before starting the - next. The other app must be up within 4 s of daemon startup - - well under one `ready_timeout` - regardless of which one is stalled, so the - pass window cannot be met by strictly sequential launch in either order. - - `version` is unused but required by the module-scope parametrize. + """With no `depends_on` between the apps and one of them stalled (runs, never reports + Running), the other must be running within 4 s of the 1 s startup window. A serialized + launcher would wait out `ready_timeout` (10 s) on the stalled app first. Both stall + orders are tried, and the stalled stub must also be running (it was launched, not skipped). """ ready_timeout_s = 10.0 # rendered over the base config's 2.0 s - parallel_window_s = 4.0 # + ~1 s daemon startup grace, still well under ready_timeout + parallel_window_s = 4.0 # plus the 1 s startup window, still well under ready_timeout assert parallel_window_s < ready_timeout_s / 2 for stalled, other in (("cpp", "rust"), ("rust", "cpp")): daemon_info = start_launch_manager_daemon( @@ -573,36 +511,34 @@ def test_independent_processes_launch_without_waiting_on_each_other( try: stalled_path = str(daemon_info["apps"][stalled]) other_path = str(daemon_info["apps"][other]) - other_started = wait_until(lambda: is_running(other_path), timeout_s=parallel_window_s) + other_started = wait_until(lambda p=other_path: is_running(p), timeout_s=parallel_window_s) assert other_started, ( f"{other}_supervised_app did not start within {parallel_window_s}s while " f"{stalled}_supervised_app was stalled (ready_timeout={ready_timeout_s}s), even " "though neither depends on the other - launch is serialized, not parallel" ) - # Rules out a vacuous pass: the stalled stub must be up too, i.e. both were - # in flight concurrently rather than the stub simply never being launched. + # Both in flight at once: rules out the stalled stub never being launched. assert is_running(stalled_path), f"stalled {stalled}_supervised_app stub was never launched" finally: stop_launch_manager_daemon(daemon_info) class TestHealthMonitoringWithDaemon: - """Health monitoring / watchdog tests with daemon.""" + """Alive-supervision (watchdog) detection, with its own daemon per `version`.""" + # No `smart_watchdog_config` claim: alive supervision is configured only in `defaults`, and + # the test neither configures it per process nor varies it. @add_test_properties( - partially_verifies=[ - "feat_req__lifecycle__liveliness_detection", - "feat_req__lifecycle__smart_watchdog_config", - ], + partially_verifies=["feat_req__lifecycle__liveliness_detection"], test_type="requirements-based", derivation_technique="requirements-analysis", ) def test_watchdog_detection(self, tmp_path_factory: pytest.TempPathFactory, version: str) -> None: - """Verify watchdog detects an unresponsive app (stopped, not reporting health) and reacts. + """SIGSTOP the app so it stops reporting alive indications; launch_manager must log its + Alive Supervision switching to FAILED or EXPIRED within 8 s. - Uses its own daemon per version: the rust run's watchdog failure triggers recovery - (restart, then a run-target switch), so a shared daemon would hand the cpp run a - restarting or relaunched app. + Limitation: checks detection only, not the reaction that follows. Uses its own daemon + so the recovery triggered by one version's failure cannot affect the other's run. """ daemon_info = start_launch_manager_daemon(tmp_path_factory) try: @@ -615,10 +551,8 @@ def _check_watchdog_detection(daemon_info: dict[str, Any], version: str) -> None daemon = daemon_info["daemon"] app_name = "rust_supervised_app" if version == "rust" else "cpp_supervised_app" - # Stop the supervised process to emulate a non-reporting workload. app_path = str(daemon_info["apps"][version]) - # start_launch_manager_daemon already waited for Running, so a missing process here - # means it died under supervision - a genuine failure, not a skip. + # start_launch_manager_daemon already waited for the app, so a missing process is a failure. pid = first_pid(app_path) assert pid is not None, f"{app_name} died before the watchdog check" @@ -626,19 +560,10 @@ def _check_watchdog_detection(daemon_info: dict[str, Any], version: str) -> None sent, reason = signal_process(pid, "-STOP", sandbox_privileged=sandbox_privileged) assert sent, f"Could not signal {app_name} (pid={pid}): {reason}" try: - watchdog_patterns = [ - rf"Got kRunning timeout for process.*\(\s*{re.escape(app_name)}\s*\)", - rf"unexpected termination of process.*\(\s*{re.escape(app_name)}\s*\)", - rf"Alive Supervision \(\s*{re.escape(app_name)}\s*\) switched to FAILED", - rf"Alive Supervision \(\s*{re.escape(app_name)}\s*\) switched to EXPIRED", - ] + # Only alive-supervision verdicts count; a startup timeout or crash is not liveliness detection. + liveliness_lost = rf"Alive Supervision \(\s*{re.escape(app_name)}\s*\) switched to (FAILED|EXPIRED)" # Poll rather than sleep: detection latency varies under CI load. - detected = wait_until( - lambda: any(re.search(pattern, daemon.get_logs()) for pattern in watchdog_patterns), - timeout_s=8.0, - ) - assert detected, ( - f"No target-specific watchdog diagnostics found for {app_name}.\nDaemon logs:\n{daemon.get_logs()}" - ) + detected = wait_until(lambda: re.search(liveliness_lost, daemon.get_logs()), timeout_s=8.0) + assert detected, f"No Alive Supervision failure logged for {app_name}.\nDaemon logs:\n{daemon.get_logs()}" finally: signal_process(pid, "-CONT", sandbox_privileged=sandbox_privileged) diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py b/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py index b43f52aa480..7f3f233f434 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py @@ -10,23 +10,22 @@ # # SPDX-License-Identifier: Apache-2.0 # ******************************************************************************* -"""Real retry-exhaustion coverage for `feat_req__lifecycle__retries_configurable`. - -Drives a dedicated `flaky_startup_app` (see support_apps/flaky_startup_app) whose -crash-before-success count is fixed in the launch_manager config, so both sides of -`ready_recovery_action.restart.number_of_attempts` are exercised deterministically -instead of relying on a real, racy startup failure: - -- `TestRetrySucceedsWithinConfiguredAttempts`: the component crashes fewer times - than the configured attempts allow, so it must recover and reach Running. -- `TestRetryExhaustionTriggersRecovery`: the component always crashes, so the - daemon must give up after exactly the configured attempts and execute the run - target's `recovery_action` (switch to `fallback_run_target`) instead of - restarting forever. +"""Startup-retry FITs for `ready_recovery_action.restart.number_of_attempts`. + +`flaky_startup_app` aborts on its first `crashes_before_success` launches (rendered per class) +and counts every launch in a file, so the daemon's retry count is observed exactly: + +- `TestRetrySucceedsWithinConfiguredAttempts`: crashes use up all retries, then the app runs. +- `TestRetryExhaustionTriggersRecovery`: the app always crashes; the daemon must stop after + 1 + `number_of_attempts` launches and stay alive. + +Limitation for `retries_configurable`: only the single configured value (2) is exercised; the +count is not varied across runs. """ from __future__ import annotations +import json from typing import Any import pytest @@ -38,12 +37,16 @@ from lifecycle_scenario import RetryDaemonScenario from test_properties import add_test_properties -# Must match flaky_startup_app's "number_of_attempts" in lifecycle_daemon_retry_config.json. -_NUMBER_OF_ATTEMPTS = 2 + +def _number_of_attempts(retry_daemon: dict[str, Any]) -> int: + """flaky_startup_app's `number_of_attempts`, read from the rendered config.""" + config = json.loads(retry_daemon["runtime_config"].read_text(encoding="utf-8")) + deployment = config["components"]["flaky_startup_app"]["deployment_config"] + return int(deployment["ready_recovery_action"]["restart"]["number_of_attempts"]) class TestRetrySucceedsWithinConfiguredAttempts(RetryDaemonScenario): - """The component crashes fewer times than `number_of_attempts` allows.""" + """The component crashes exactly `number_of_attempts` times, then succeeds.""" crashes_before_success = 2 @@ -53,17 +56,16 @@ class TestRetrySucceedsWithinConfiguredAttempts(RetryDaemonScenario): derivation_technique="requirements-analysis", ) def test_component_recovers_within_configured_attempts(self, retry_daemon: dict[str, Any]) -> None: - """Daemon retries a failing component up to `number_of_attempts` and lets - it reach Running once it stops crashing. - """ + """Exactly `crashes_before_success + 1` launches happen, the last one stays running (pgrep), + and no further launch follows within 1.5 s.""" app_path = retry_daemon["app_path"] counter_path = retry_daemon["counter_path"] expected_attempts = retry_daemon["crashes_before_success"] + 1 + assert retry_daemon["crashes_before_success"] <= _number_of_attempts(retry_daemon), ( + "crashes_before_success must fit within number_of_attempts for this test to be meaningful" + ) - # Check the attempt counter before is_running(): a crashing attempt is still - # technically "running" for the microseconds before it aborts, so polling - # is_running() first can catch that transient window rather than the - # eventual successful attempt. + # Wait on the counter, not is_running(): a crashing attempt is briefly visible to pgrep. reached = wait_until(lambda: read_retry_attempt_count(counter_path) >= expected_attempts, timeout_s=8.0) assert reached, "flaky_startup_app never reached the expected number of launch attempts" @@ -75,14 +77,13 @@ def test_component_recovers_within_configured_attempts(self, retry_daemon: dict[ started = wait_until(lambda: is_running(app_path), timeout_s=2.0) assert started, "flaky_startup_app never reached Running after its last launch attempt" - # Once healthy it should stay up: no further restarts. relaunched = wait_until(lambda: read_retry_attempt_count(counter_path) != attempts, timeout_s=1.5) assert not relaunched, "Component was relaunched again after it was already Running" assert is_running(app_path), "flaky_startup_app stopped running after recovering" class TestRetryExhaustionTriggersRecovery(RetryDaemonScenario): - """The component always crashes, exceeding `number_of_attempts`.""" + """The component always crashes, exhausting `number_of_attempts`.""" crashes_before_success = 999 @@ -92,34 +93,34 @@ class TestRetryExhaustionTriggersRecovery(RetryDaemonScenario): derivation_technique="requirements-analysis", ) def test_daemon_gives_up_after_configured_attempts(self, retry_daemon: dict[str, Any]) -> None: - """Daemon stops restarting a component once `number_of_attempts` is - exhausted, instead of retrying forever, and executes the run target's - `recovery_action` (switch to `fallback_run_target`). + """Exactly `1 + number_of_attempts` launches happen, none follows within 2 s, the app is + not running and the daemon is still up. + + Limitation: the run target's `recovery_action` (switch to `fallback_run_target`) is only + inferred from the daemon staying up without relaunching; it is not asserted directly. """ app_path = retry_daemon["app_path"] counter_path = retry_daemon["counter_path"] + number_of_attempts = _number_of_attempts(retry_daemon) settled = wait_until( - lambda: read_retry_attempt_count(counter_path) >= _NUMBER_OF_ATTEMPTS + 1, + lambda: read_retry_attempt_count(counter_path) >= number_of_attempts + 1, timeout_s=8.0, ) assert settled, "flaky_startup_app never reached the configured number of launch attempts" - # Give the daemon a chance to keep retrying, if it hasn't actually given up. + # Would catch a daemon that keeps retrying past the budget. attempts_after_exhaustion = read_retry_attempt_count(counter_path) kept_retrying = wait_until( lambda: read_retry_attempt_count(counter_path) != attempts_after_exhaustion, timeout_s=2.0, ) assert not kept_retrying, ( - f"Daemon kept restarting the component past the configured number_of_attempts={_NUMBER_OF_ATTEMPTS}" + f"Daemon kept restarting the component past the configured number_of_attempts={number_of_attempts}" ) - assert attempts_after_exhaustion == _NUMBER_OF_ATTEMPTS + 1, ( - f"Expected exactly {_NUMBER_OF_ATTEMPTS + 1} launch attempts before giving up, got " + assert attempts_after_exhaustion == number_of_attempts + 1, ( + f"Expected exactly {number_of_attempts + 1} launch attempts before giving up, got " f"{attempts_after_exhaustion}" ) - assert not is_running(app_path), ( - "flaky_startup_app is still running after exhausting retries; recovery_action " - "(switch_run_target -> fallback_run_target) should have stopped further attempts" - ) - assert retry_daemon["daemon"].is_running(), "Launch Manager daemon crashed instead of switching run target" + assert not is_running(app_path), "flaky_startup_app is still running after exhausting its retries" + assert retry_daemon["daemon"].is_running(), "launch_manager exited after the component exhausted its retries" diff --git a/feature_integration_tests/test_scenarios/cpp/src/scenarios/lifecycle/conditional_launching.cpp b/feature_integration_tests/test_scenarios/cpp/src/scenarios/lifecycle/conditional_launching.cpp index 3ba73ab2dd9..ef8450729d5 100644 --- a/feature_integration_tests/test_scenarios/cpp/src/scenarios/lifecycle/conditional_launching.cpp +++ b/feature_integration_tests/test_scenarios/cpp/src/scenarios/lifecycle/conditional_launching.cpp @@ -47,15 +47,10 @@ bool env_condition_met(const std::string& name) { return std::getenv(name.c_str()) != nullptr; } -// Best-effort check whether a process matching `process_name` is currently running, by scanning -// /proc//comm (the kernel-truncated 15-char command name) and /proc//cmdline (the full -// argv[0], which covers names comm truncates). +// True if some process's /proc//comm (15-char truncated) or argv[0] basename equals +// `process_name`. Best effort: a process exiting mid-scan counts as "not found this pass". bool process_condition_met(const std::string& process_name) { - // Iterating /proc races with processes exiting mid-scan (ENOENT on a just-vanished pid's - // subdirectory); std::filesystem surfaces that as filesystem_error even with the - // non-throwing error_code constructor, since only construction/increment on the top-level - // directory is covered, not opening files underneath. Treat it as "not found this pass" - // rather than letting a race abort the whole wait loop. + // Iteration can still throw filesystem_error when a pid vanishes, despite the error_code overload. try { std::error_code ec; for (const auto& entry : std::filesystem::directory_iterator( @@ -79,9 +74,7 @@ bool process_condition_met(const std::string& process_name) { if (!argv0.empty()) { const auto argv0_end = argv0.find('\0'); const std::string first_arg = argv0.substr(0, argv0_end); - // Compare the basename only (portion after the last '/'), not a raw suffix of - // the full path: a plain suffix match would also accept e.g. "/usr/bin/oversleep" - // as satisfying process_name="sleep". + // Basename, not suffix: "/usr/bin/oversleep" must not match "sleep". const auto slash_pos = first_arg.find_last_of('/'); const std::string basename = slash_pos == std::string::npos ? first_arg : first_arg.substr(slash_pos + 1); @@ -97,42 +90,42 @@ bool process_condition_met(const std::string& process_name) { } template -std::vector parse_string_array_field(const std::string& input, - const std::string& field_name, - Converter convert) { +std::optional> parse_string_array_field(const std::string& input, + const std::string& field_name, + Converter convert) { std::vector values; const score::json::JsonParser parser; const auto root_any_res = parser.FromBuffer(input); if (!root_any_res.has_value()) { - return values; + return std::nullopt; } const auto root_object_res = root_any_res.value().As(); if (!root_object_res.has_value()) { - return values; + return std::nullopt; } const auto& root = root_object_res.value().get(); const auto test_it = root.find("test"); if (test_it == root.end()) { - return values; + return std::nullopt; } const auto test_object_res = test_it->second.As(); if (!test_object_res.has_value()) { - return values; + return std::nullopt; } const auto& test = test_object_res.value().get(); const auto field_it = test.find(field_name); if (field_it == test.end()) { - return values; + return std::nullopt; } const auto array_res = field_it->second.As(); if (!array_res.has_value()) { - return values; + return std::nullopt; } for (const auto& element : array_res.value().get()) { @@ -146,7 +139,7 @@ std::vector parse_string_array_field(const std::string& input, return values; } -std::vector parse_wait_conditions(const std::string& input) { +std::optional> parse_wait_conditions(const std::string& input) { return parse_string_array_field(input, "wait_conditions", [](const score::json::Any& element) { const auto value = element.As(); if (!value.has_value()) { @@ -156,6 +149,9 @@ std::vector parse_wait_conditions(const std::string& input) { }); } +// FIT stub (not launch_manager): validates `test.wait_conditions`, then polls each condition every +// `polling_interval_ms` until all are met or `timeout_ms` expires. A met condition stays latched, +// even if it later becomes false. Mirrors the Rust scenario, including its error messages. class ConditionalLaunching : public Scenario { public: std::string name() const override { return "conditional_launching"; } @@ -169,7 +165,7 @@ class ConditionalLaunching : public Scenario { uint64_t polling_interval = 50; uint64_t timeout = 5000; - const auto wait_conditions = parse_wait_conditions(input); + const auto wait_conditions_res = parse_wait_conditions(input); const auto root_object_res = root_any_res.value().As(); if (root_object_res.has_value()) { @@ -199,9 +195,15 @@ class ConditionalLaunching : public Scenario { } } + // Same messages as the Rust scenario: "missing" and "empty" are distinct errors. + if (!wait_conditions_res.has_value()) { + throw std::runtime_error( + "Wait conditions were not provided: missing 'test.wait_conditions' in scenario input"); + } + const auto& wait_conditions = *wait_conditions_res; if (wait_conditions.empty()) { throw std::runtime_error( - "Wait conditions were not provided: missing or empty 'test.wait_conditions' in scenario input"); + "Wait conditions were not provided: empty 'test.wait_conditions' in scenario input"); } log_info("Testing conditional launching"); diff --git a/feature_integration_tests/test_scenarios/rust/src/scenarios/lifecycle/conditional_launching.rs b/feature_integration_tests/test_scenarios/rust/src/scenarios/lifecycle/conditional_launching.rs index 2481b027379..f3fc0d8fe2b 100644 --- a/feature_integration_tests/test_scenarios/rust/src/scenarios/lifecycle/conditional_launching.rs +++ b/feature_integration_tests/test_scenarios/rust/src/scenarios/lifecycle/conditional_launching.rs @@ -18,6 +18,9 @@ use std::time::{Duration, Instant}; use test_scenarios_rust::scenario::Scenario; use tracing::info; +/// FIT stub (not launch_manager): validates `test.wait_conditions`, then polls each condition every +/// `polling_interval_ms` until all are met or `timeout_ms` expires. A met condition stays latched, +/// even if it later becomes false. pub struct ConditionalLaunching; fn path_condition_met(path: &str) -> bool { @@ -28,9 +31,8 @@ fn env_condition_met(name: &str) -> bool { std::env::var_os(name).is_some() } -/// Best-effort check whether a process matching `process_name` is currently running, by -/// scanning /proc//comm (kernel-truncated to 15 chars) and /proc//cmdline (full -/// argv[0], which covers names `comm` truncates). +/// True if some process's /proc//comm (15-char truncated) or argv[0] basename equals +/// `process_name`. Best effort: unreadable entries are skipped. fn process_condition_met(process_name: &str) -> bool { let Ok(entries) = fs::read_dir("/proc") else { return false; @@ -52,8 +54,7 @@ fn process_condition_met(process_name: &str) -> bool { if let Ok(cmdline) = fs::read(entry.path().join("cmdline")) { let argv0 = cmdline.split(|&b| b == 0).next().unwrap_or(&[]); if let Ok(argv0) = std::str::from_utf8(argv0) { - // Compare the basename only: a raw suffix match on the full path would also - // accept e.g. "/usr/bin/oversleep" as satisfying process_name="sleep". + // Basename, not suffix: "/usr/bin/oversleep" must not match "sleep". let basename = argv0.rsplit('/').next().unwrap_or(argv0); if basename == process_name { return true; From 57843f665679412256c015b1157ac6a8145dc7fc Mon Sep 17 00:00:00 2001 From: Saumya-R Date: Wed, 30 Sep 2026 10:06:25 +0530 Subject: [PATCH 8/9] added the BUILD file in patches of lifecylce --- patches/lifecycle/BUILD | 0 1 file changed, 0 insertions(+), 0 deletions(-) create mode 100644 patches/lifecycle/BUILD diff --git a/patches/lifecycle/BUILD b/patches/lifecycle/BUILD new file mode 100644 index 00000000000..e69de29bb2d From 11c359376b676ea02bfb20e6fa2cdba2a9e49933 Mon Sep 17 00:00:00 2001 From: Saumya-R Date: Wed, 7 Oct 2026 11:50:39 +0530 Subject: [PATCH 9/9] updating the requirement Ids based on the new feature requirements --- feature_integration_tests/test_cases/BUILD | 5 ++--- .../tests/lifecycle/test_conditional_launching.py | 8 +++----- .../tests/lifecycle/test_process_launching_with_daemon.py | 8 ++++---- .../test_cases/tests/lifecycle/test_retry_exhaustion.py | 8 ++++---- 4 files changed, 13 insertions(+), 16 deletions(-) diff --git a/feature_integration_tests/test_cases/BUILD b/feature_integration_tests/test_cases/BUILD index 36bbcab4b09..cee628acfdf 100644 --- a/feature_integration_tests/test_cases/BUILD +++ b/feature_integration_tests/test_cases/BUILD @@ -188,9 +188,8 @@ score_py_pytest( deps = all_requirements, ) -# Dedicated, isolated coverage for ready_recovery_action.restart.number_of_attempts -# (feat_req__lifecycle__retries_configurable). Each daemon invocation generates its own -# single-component config and runtime tree under TEST_TMPDIR. +# Dedicated, isolated coverage for ready_recovery_action.restart.number_of_attempts. Each daemon +# invocation generates its own single-component config and runtime tree under TEST_TMPDIR. score_py_pytest( name = "fit_lifecycle_retries", timeout = "long", diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py b/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py index dcca057842c..b9149ace601 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_conditional_launching.py @@ -58,10 +58,8 @@ class TestConditionalLaunchingBlocksOnMissingDependency: @add_test_properties( partially_verifies=[ - "feat_req__lifecycle__waitfor_support", - "feat_req__lifecycle__dependency_check", + "feat_req__lifecycle__conditional_startup", "feat_req__lifecycle__process_ordering", - "feat_req__lifecycle__define_swc_dependencies", ], test_type="requirements-based", derivation_technique="requirements-analysis", @@ -73,8 +71,8 @@ def test_rust_stays_down_until_cpp_dependency_becomes_available( launch failure at least twice (it keeps retrying rather than aborting). After cpp is made executable, cpp and then rust must be running within 8 s each. - Limitations: "running" is pgrep process existence; rust has a single dependency, so - `dependency_check` ("all dependencies") cannot be told apart from "any dependency". + Limitations: "running" is pgrep process existence; rust has a single dependency, so this + does not distinguish single-edge gating from multi-dependency behavior. No `cond_process_start` claim: that requirement is about starting on the return value of earlier processes, which this config does not use. """ diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py b/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py index 1862d530f51..d1e11779007 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_process_launching_with_daemon.py @@ -148,7 +148,7 @@ def test_startup_declares_and_launches_multiple_processes( assert daemon.is_running(), "Launch Manager daemon stopped unexpectedly" @add_test_properties( - partially_verifies=["feat_req__lifecycle__process_launch_args"], + partially_verifies=["feat_req__lifecycle__launch_support"], test_type="requirements-based", derivation_technique="requirements-analysis", ) @@ -182,7 +182,7 @@ def test_launch_process_arguments_are_applied( ) @add_test_properties( - partially_verifies=["feat_req__lifecycle__process_launch_args"], + partially_verifies=["feat_req__lifecycle__launch_support"], test_type="requirements-based", derivation_technique="requirements-analysis", ) @@ -214,7 +214,7 @@ def test_launch_process_environment_is_applied( f"{key} mismatch for {app_name}: expected {expected_value!r}, got {proc_env.get(key)!r}" ) - # No requirement claim (feat_req__lifecycle__uid_gid_support): skips in CI, which neither sets + # No requirement claim: skips in CI, which neither sets # FIT_ENABLE_SETCAP=1 nor runs unsandboxed, both needed for the setcap grant. See README.md. def test_launched_process_uid_gid_matches_config_when_applied( self, @@ -371,7 +371,7 @@ def test_scheduling_policy_is_non_default_and_applied( ) @add_test_properties( - partially_verifies=["feat_req__lifecycle__secpol_non_root"], + partially_verifies=["feat_req__lifecycle__launch_support"], test_type="requirements-based", derivation_technique="requirements-analysis", ) diff --git a/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py b/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py index 7f3f233f434..ae052030fee 100644 --- a/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py +++ b/feature_integration_tests/test_cases/tests/lifecycle/test_retry_exhaustion.py @@ -19,8 +19,8 @@ - `TestRetryExhaustionTriggersRecovery`: the app always crashes; the daemon must stop after 1 + `number_of_attempts` launches and stay alive. -Limitation for `retries_configurable`: only the single configured value (2) is exercised; the -count is not varied across runs. +Limitation: only the single configured retry value (2) is exercised; the count is not varied +across runs. """ from __future__ import annotations @@ -51,7 +51,7 @@ class TestRetrySucceedsWithinConfiguredAttempts(RetryDaemonScenario): crashes_before_success = 2 @add_test_properties( - partially_verifies=["feat_req__lifecycle__retries_configurable"], + partially_verifies=["feat_req__lifecycle__launch_support"], test_type="requirements-based", derivation_technique="requirements-analysis", ) @@ -88,7 +88,7 @@ class TestRetryExhaustionTriggersRecovery(RetryDaemonScenario): crashes_before_success = 999 @add_test_properties( - partially_verifies=["feat_req__lifecycle__retries_configurable"], + partially_verifies=["feat_req__lifecycle__launch_support"], test_type="requirements-based", derivation_technique="requirements-analysis", )