diff --git a/.github/workflows/README.md b/.github/workflows/README.md index b046333..0c8e96d 100644 --- a/.github/workflows/README.md +++ b/.github/workflows/README.md @@ -17,20 +17,22 @@ This directory contains GitHub Actions workflows for automated testing. - Column alias tests (8) - Run-time mode tests (7) - Unit tests (7) -- Feature tests (6) +- Feature tests (8) - Conservative backfilling tests (2) - Replay tests (5: one reclamation-boundary binary and four CLI comparisons) - Resource-history tests (5) - Job-store tests (6) -- Config tests, including power-usage `trace_type` coverage -- Native CTest suite (12 tests, plus MPI streaming when MPI is available), - including the custom FCFS scheduler's focused and 2,000-job comparisons +- Config tests (12), including power-usage `trace_type`, capacity-schedule, + and warm-start coverage +- Native CTest suite (15 native registrations plus the Python trace-tools + registration, and MPI streaming when MPI is available), including the + custom FCFS scheduler's focused and 2,000-job comparisons - Ser20-disabled native serialization build and tests - Sphinx and Doxygen documentation build with warnings treated as errors -- Python API tests (17) +- Python API tests (18) - gRPC client/server tests (2) -- Append-job tests (19 C++ + 3 gRPC) -- FCFS/EASY backfill-window focused rerun of the three-case gRPC binary +- Append-job tests (20 C++ + 5 optional gRPC checks) +- FCFS/EASY backfill-window focused rerun of the five-check gRPC binary - Synchronized single-coordinator gRPC test - Progressive-loading tests (C++ + CLI) - Queue-input schema test @@ -69,25 +71,26 @@ Total tests referenced by the full suite: | Category | Count | Verified in this doc pass? | |----------|-------|------------------------------| | Scheduler correctness | 34 | CI runner | -| Custom FCFS | 7 | CTest; five focused checks and two golden schedules, including 2,000 jobs | +| Custom FCFS | 8 | CTest; six focused checks and two golden schedules, including warm-start accounting and 2,000 jobs | | Queue implementation differential | 34 fixtures × 4 implementations | CI runner | | Column aliases | 8 | CI runner | | Run-time mode | 7 | CI runner | | Unit | 7 | CI runner | -| Feature | 6 | CI runner | +| Feature | 8 | CI runner; includes time-varying capacity and native warm start | | Conservative | 2 | CI runner | | Replay | 5 | CI runner; reclamation safety plus resource equivalence | | Resource history | 5 | CI runner | | Job store | 6 | CI runner | -| Config | 9 | CI runner; includes power-usage `trace_type` coverage | -| Native CTest | 12, plus 1 with MPI | CI runner; RNG and binary serialization, trace policies, replay reclamation, custom scheduling, append/streaming APIs, queues, and CLI dispatch | -| Python API | 17 | CI runner | +| Config | 12 | CI runner; includes power-usage, capacity-schedule, and warm-start configuration coverage | +| Native CTest | 15, plus 1 with MPI | CI runner; RNG and binary serialization, trace policies, replay reclamation, custom scheduling, append/streaming APIs, warm starts, capacity parsing, queues, and CLI dispatch | +| Trace tools | 2 checks in 1 CTest registration | CI runner; capacity inference and warm-start boundary/output behavior | +| Python API | 18 | CI runner | | gRPC client/server | 2 | CI runner | -| Append-job | 22: 19 C++ + 3 gRPC | CI runner | -| FCFS/EASY backfill-window gRPC | 3 repeated checks; 1 targeted | CI runner | +| Append-job | 25: 20 C++ + 5 optional gRPC checks | CI runner | +| FCFS/EASY backfill-window gRPC | 5 repeated checks; 1 targeted | CI runner | | Single-coordinator gRPC | 1 | CI runner; synchronized independent systems | | Progressive loading | 11 C++ + 4 CLI | CI runner | -| Queue input schema | 1 binary | CI runner | +| Queue input schema | 1 binary | CI runner; queue variants and replay/simulation runtime validation | | Scale | 7 | CI runner | The workflow summary in `tests.yml` is the authoritative CI-oriented list. diff --git a/.github/workflows/tests.yml b/.github/workflows/tests.yml index 558f9c1..1c977e2 100644 --- a/.github/workflows/tests.yml +++ b/.github/workflows/tests.yml @@ -200,7 +200,7 @@ jobs: id: feature run: | echo "========================================" - echo "Running Feature Tests (6 tests)" + echo "Running Feature Tests (8 tests)" echo "========================================" ./tests/run_feature_tests.sh @@ -246,14 +246,14 @@ jobs: id: config run: | echo "========================================" - echo "Running Config Tests (9 tests)" + echo "Running Config Tests (12 tests)" echo "========================================" ./tests/run_configs_tests.sh - name: Run Core CTest Tests id: core-ctest run: | - echo "Running 12 native CTest registrations plus MPI when available" + echo "Running 15 native CTest registrations, the trace-tools test, plus MPI when available" cd build ctest --output-on-failure @@ -262,7 +262,7 @@ jobs: run: | source venv/bin/activate echo "========================================" - echo "Running Python API Tests (17 tests)" + echo "Running Python API Tests (18 tests)" echo "========================================" PYTHON_EXECUTABLE=$(command -v python) ./tests/run_python_tests.sh @@ -278,7 +278,7 @@ jobs: id: append-job run: | echo "========================================" - echo "Running Append-Job Tests (19 C++ + 3 gRPC)" + echo "Running Append-Job Tests (20 C++ + 5 optional gRPC checks)" echo "========================================" ./tests/run_append_job_tests.sh @@ -334,16 +334,16 @@ jobs: echo " Column Aliases (8 tests): ${{ steps.column-alias.outcome }}" echo " Run Time Mode (7 tests): ${{ steps.run-time-mode.outcome }}" echo " Unit (7 tests): ${{ steps.unit.outcome }}" - echo " Feature (6 tests): ${{ steps.feature.outcome }}" + echo " Feature (8 tests): ${{ steps.feature.outcome }}" echo " Conservative (2 tests): ${{ steps.conservative.outcome }}" echo " Replay (5 tests): ${{ steps.replay.outcome }}" echo " Resource History (5 tests): ${{ steps.resource-history.outcome }}" echo " Job Store (6 tests): ${{ steps.job-store.outcome }}" - echo " Config (9 tests): ${{ steps.config.outcome }}" - echo " Native CTest (12 + MPI when available): ${{ steps.core-ctest.outcome }}" - echo " Python API (17 tests): ${{ steps.python-api.outcome }}" - echo " Append-Job (19 C++ + 3 gRPC): ${{ steps.append-job.outcome }}" - echo " Backfill-window focused rerun (3 checks already counted): ${{ steps.backfill-window.outcome }}" + echo " Config (12 tests): ${{ steps.config.outcome }}" + echo " CTest (15 native + trace tools + MPI when available): ${{ steps.core-ctest.outcome }}" + echo " Python API (18 tests): ${{ steps.python-api.outcome }}" + echo " Append-Job (20 C++ + 5 optional gRPC checks): ${{ steps.append-job.outcome }}" + echo " Backfill-window focused rerun (5 checks already counted): ${{ steps.backfill-window.outcome }}" echo " Single-Coordinator gRPC: ${{ steps.grpc-single-coordinator.outcome }}" echo " Progressive Loading (11 C++ + 4 CLI): ${{ steps.progressive-load.outcome }}" echo " Queue Input Schema: ${{ steps.queue-input-schema.outcome }}" diff --git a/CMakeLists.txt b/CMakeLists.txt index 91d454f..25e0ea7 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -692,6 +692,31 @@ set_target_properties(test_block_queue-bin PROPERTIES CMAKE_INSTALL_RPATH list(APPEND DR_EVT_UNIT_TEST_TARGETS test_block_queue-bin) +# Time-varying capacity schedule parser validation. +add_executable(test_capacity_schedule-bin tests/test_capacity_schedule.cpp) +target_include_directories(test_capacity_schedule-bin PUBLIC + $ + $ + $) +target_link_libraries(test_capacity_schedule-bin PRIVATE dr_evt) +set_target_properties(test_capacity_schedule-bin PROPERTIES + OUTPUT_NAME test_capacity_schedule + CMAKE_INSTALL_RPATH "${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}") +list(APPEND DR_EVT_UNIT_TEST_TARGETS test_capacity_schedule-bin) + +# Native warm-start coverage: boundary classification, capacity transitions, +# statistics, all scheduler policies/queue backends, and differential traces. +add_executable(test_warm_start-bin tests/test_warm_start.cpp) +target_include_directories(test_warm_start-bin PUBLIC + $ + $ + $) +target_link_libraries(test_warm_start-bin PRIVATE dr_evt) +set_target_properties(test_warm_start-bin PROPERTIES + OUTPUT_NAME test_warm_start + CMAKE_INSTALL_RPATH "${CMAKE_INSTALL_PREFIX}/${CMAKE_INSTALL_LIBDIR}") +list(APPEND DR_EVT_UNIT_TEST_TARGETS test_warm_start-bin) + # MPI streaming test (requires MPI) # Only search for MPI for this test - dr_evt library itself has no MPI dependency @@ -714,6 +739,7 @@ if(MPI_CXX_FOUND) endif() if(DR_EVT_WITH_UNIT_TESTING) + find_package(Python3 COMPONENTS Interpreter QUIET) add_test( NAME test_append_job_api COMMAND $) @@ -735,11 +761,38 @@ if(DR_EVT_WITH_UNIT_TESTING) add_test( NAME test_block_queue COMMAND $) + add_test( + NAME test_capacity_schedule + COMMAND $) + add_test( + NAME test_warm_start + COMMAND $) + add_test( + NAME test_warm_start_validation + COMMAND /bin/bash + ${CMAKE_CURRENT_SOURCE_DIR}/tests/run_warm_start_validation_tests.sh + $) + add_test( + NAME test_max_time + COMMAND /bin/bash + ${CMAKE_CURRENT_SOURCE_DIR}/tests/run_max_time_tests.sh + $) + if(Python3_Interpreter_FOUND) + add_test( + NAME test_trace_tools + COMMAND ${CMAKE_COMMAND} -E env + DR_EVT_SIMULATOR=$ + ${Python3_EXECUTABLE} ${CMAKE_CURRENT_SOURCE_DIR}/tests/test_trace_tools.py) + set_tests_properties(test_trace_tools PROPERTIES + LABELS "unit;scripts" + WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}) + endif() set_tests_properties( test_append_job_api test_custom_scheduler test_progressive_load test_queue_input - test_batch_vs_streaming test_block_queue + test_batch_vs_streaming test_block_queue test_capacity_schedule + test_warm_start test_warm_start_validation test_max_time PROPERTIES LABELS "unit;native" WORKING_DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR} diff --git a/README.md b/README.md index 470ff54..af73940 100644 --- a/README.md +++ b/README.md @@ -63,7 +63,8 @@ minimum input fields are `job_submit_time`, `num_nodes`, and `time_limit`. job's execution duration is selected separately with `--run_time_mode`: - `actual` (default) uses `actual_run_time` from the input trace (also accepted as - `duration`, `actual_duration`, or `run_time`). + `actual_runtime`, `duration`, `actual_duration`, or `run_time`). A supplied + value must be finite and no greater than `time_limit`. - `limit` runs each job for exactly its requested `time_limit`. - `distribution` draws a duration from the selected `normal`, `lognormal`, or `uniform` distribution using `--run_time_scale`, `--run_time_stddev`, and @@ -186,9 +187,12 @@ job_submit_time,begin_time,end_time,num_nodes,time_limit 0,100,120,60,20 ``` -The recorded `begin_time` and `end_time` are authoritative. Replay does not -invoke a scheduler or choose new start times; `total_nodes` is used only to -derive the free-node count. +The recorded `begin_time` and `end_time` are authoritative. An optional +`actual_run_time` column (including its accepted aliases) is checked against +`end_time - begin_time`; rows that disagree by more than `1e-6` seconds are +rejected. When the column is absent, replay derives the duration from the two +timestamps. Replay does not invoke a scheduler or choose new start times; +`total_nodes` is used only to derive the free-node count. Replay writes: diff --git a/docs/CLIENT_SERVER_GUIDE.md b/docs/CLIENT_SERVER_GUIDE.md index be7b212..fc5ea9d 100644 --- a/docs/CLIENT_SERVER_GUIDE.md +++ b/docs/CLIENT_SERVER_GUIDE.md @@ -31,6 +31,7 @@ in-process via the streaming API: | Request | Corresponds to | |---|---| | `InitRequest` | Constructing a `Simulation` from a `Sim_Params`-equivalent config | +| `RunRequest` | Calls `Simulation::run()`; initialized input and `sim_start_time` select simulation, full replay, or replay-based warm start, and configured `max_time` supplies an inclusive stopping boundary | | `InitializeTraceRequest` | `Simulation::initialize_trace()` | | `AppendJobRequest` | `Simulation::append_job()` - a genuinely new job the server has never seen before | | `AppendJobsRequest` | `Simulation::append_jobs()` - the batch counterpart, several new jobs in one call | @@ -40,6 +41,14 @@ in-process via the streaming API: | `GetBackfillWindowRequest` | One FCFS/EASY reservation snapshot: current capacity, shadow time, and projected releases | | `GetStatisticsRequest`, `GetCurrentTimeRequest`, etc. | The monitoring/statistics methods | +For a replay-based warm start, set `InitRequest.sim_start_time` to a positive +global simulation boundary, provide a replay-format `infile`, and send +`RunRequest`. +This field is distinct from each job's historical `begin_time`. The server then +applies the same two-stage warm-start classification as the CLI and Python +batch API. A zero simulation start time preserves ordinary batch behavior; +negative and non-finite values are rejected. + Every `ClientMessage` carries a `request_id`, echoed back on the matching `ServerMessage`, so a client can correlate responses even if it pipelines multiple in-flight requests (the provided `dr_evt_client` sends one at a diff --git a/docs/TESTING_GUIDE.md b/docs/TESTING_GUIDE.md index a7dfc0c..ef18a08 100644 --- a/docs/TESTING_GUIDE.md +++ b/docs/TESTING_GUIDE.md @@ -22,12 +22,35 @@ plus golden-output comparisons for a targeted simultaneous-backfill case and a times match the default circular-buffer FCFS scheduler when the callback selects candidates in FCFS order. -The append-job API test also covers Custom-FCFS time-accounted resource area -and prediction-horizon estimation. Its accounting case combines two -allocations and two releases at one timestamp. Its horizon scenario uses four -running and four waiting jobs to exercise successive running-job completion -boundaries, the post-replay full-capacity tail, the ``U=0`` fallback, future- -arrival exclusion, and invalid utilization values. +The append-job API test also covers Custom-FCFS time-accounted resource area, +capacity-aware instantaneous and aggregate utilization, and +prediction-horizon estimation. Its accounting cases include two allocations +and two releases at one timestamp and a scheduled-capacity reduction below +live occupancy. Its horizon scenario uses four running and four waiting jobs +to exercise successive running-job completion boundaries, the post-replay +full-capacity tail, the ``U=0`` fallback, future-arrival exclusion, and invalid +utilization values. The warm-start suite separately checks integrated +effective capacity across capacity changes at and after the start boundary. + +The replay-based warm-start test exercises completed history, live historical jobs, +inherited waiters, boundary and future arrivals, empty tails, fractional and +simultaneous timestamps, every supported priority/backfill/queue combination, +and randomized differential workloads. Runtime-policy cases verify that +``actual``, ``limit``, and ``distribution`` affect only ordinarily scheduled +jobs while warmup jobs retain their recorded end times. Capacity coverage +includes overcommit after a reduction and a capacity change exactly at the +final warmup departure. A separate case protects the documented +``sim_start_time == 0`` behavior: zero disables replay-based warm start and +preserves a traditional full replay. Inclusive ``max_time`` boundaries are checked in +ordinary simulation, replay, and warm-start execution. CLI coverage accepts an ISO simulation start time, +rejects mixed timestamp encodings within one input file, and +rejects negative or non-finite values, simulation-format input, and progressive +file lists. + +The queue-input parser test verifies both queue schemas and runtime validity: +a supplied replay runtime must agree with ``end_time - begin_time`` within the +timestamp tolerance, while a supplied simulation runtime must be finite and +no greater than ``time_limit``. ## Test inventory and commands @@ -39,6 +62,13 @@ CTest discovers tests enabled by the current build configuration. Optional features such as Protobuf, gRPC, MPI, Python bindings, and Catch2 add their corresponding tests only when available. +Run only the replay-based warm-start behavior and CLI validation registrations with: + +```bash +ctest --test-dir build --output-on-failure \ + -R '^(test_warm_start|test_warm_start_validation)$' +``` + ## Expected outputs Scheduler fixtures normally contain: diff --git a/docs/api/PYTHON_API.md b/docs/api/PYTHON_API.md index 8ef9d0b..099bab6 100644 --- a/docs/api/PYTHON_API.md +++ b/docs/api/PYTHON_API.md @@ -51,6 +51,8 @@ or `None`. |---|---| | `infile` | `str` | | `total_nodes` | `int` | +| `capacity_schedule` | `str` | +| `sim_start_time` | `float` | | `trace_format` | `str` | | `timestamp_format` | `str` | | `run_time_mode` | `RunTimeMode` | @@ -59,6 +61,10 @@ or `None`. | `priority_policy` | `PriorityPolicy` | | `verbose` | `bool` | +`sim_start_time` is the global simulation boundary; it is distinct from each +job's historical `begin_time`. A positive value enables replay-based warm start +for replay input, while zero preserves ordinary full replay. + Other C++/CLI configuration fields are not exposed by the binding. Use the `simulator` executable when one of those settings is required; its options are documented in [Command-Line Options](../user-guide/command-line.md). @@ -91,7 +97,7 @@ Their scheduling semantics are documented in | `run_until_exclusive(target_time)` | Process events strictly before the target. | | `get_current_time()` | Return current simulation time. | | `get_nodes_in_use()` | Return allocated nodes. | -| `get_current_utilization()` | Return instantaneous `nodes_in_use / total_nodes`. | +| `get_current_utilization()` | Return instantaneous usage of effective scheduled capacity. | | `get_resource_area()` | Return Custom-FCFS allocated-node area in node-seconds; unavailable for standard schedulers. | | `get_available_nodes()` | Return free nodes. | | `get_active_job_count()` | Return waiting jobs. | @@ -126,12 +132,13 @@ contains `time` and `nodes_released`. For simulations constructed with Custom-FCFS callbacks, `resource_area` is accumulated as `nodes_in_use * interval` between settled scheduling times. -`utilization` divides that area by `total_nodes` and the elapsed accounting -horizon (the snapshot time while work is running, or the last resource event -after it becomes idle). Unlike `get_current_utilization()`, intervals that are -far apart therefore carry proportionally more weight. Standard schedulers -retain the post-hoc scheduled-workload calculation and do not perform live -area bookkeeping. +`utilization` divides that area by integrated effective capacity over the +accounting horizon (the snapshot time while work is running, or the last +resource event after it becomes idle). During non-preemptive draining, +effective capacity is at least the running allocation. Unlike +`get_current_utilization()`, intervals that are far apart therefore carry +proportionally more weight. Standard schedulers retain the post-hoc +scheduled-workload calculation and do not perform live area bookkeeping. Metric definitions are in [Output Trace Files](../user-guide/output-traces.md#cli-summary). diff --git a/docs/api/STREAMING_API.md b/docs/api/STREAMING_API.md index 9155ff3..5646b1a 100644 --- a/docs/api/STREAMING_API.md +++ b/docs/api/STREAMING_API.md @@ -130,7 +130,9 @@ void advance_to(sim_time_t target_time); - Advances through all events up to AND INCLUDING `target_time` - Scheduler makes decisions at each event - Jobs may start/end during advancement -- `current_time` becomes `target_time` after call +- `current_time` becomes a finite `target_time` after the call. Passing the + maximum representable value drains the simulation and leaves `current_time` + at the last real event rather than exposing the sentinel as a timestamp. **Example:** ```cpp @@ -210,16 +212,20 @@ tdiff_t get_resource_area() const; // Custom-FCFS simulations only ``` `get_current_utilization()` is the point-in-time ratio of allocated nodes to -configured nodes. For a simulation created with the Custom-FCFS callback +effective scheduled capacity. During a non-preemptive reduction below live +occupancy, the running allocation is treated as effective capacity until it +drains, keeping utilization bounded by one. For a simulation created with the Custom-FCFS callback constructor, `get_resource_area()` is the area under the allocated-node curve: the time integral of allocated nodes, accumulated once per settled scheduling timestamp and reported in node-seconds. Standard schedulers do not perform this live bookkeeping, and calling `get_resource_area()` for one throws `std::logic_error`. -For Custom FCFS, `Statistics::utilization` divides this area by configured -nodes and the elapsed accounting horizon. Standard schedulers retain their -post-hoc completed-schedule utilization calculation. +For Custom FCFS, `Statistics::utilization` divides this area by the integrated +effective capacity over the accounting horizon. Standard schedulers use the +same capacity-aware denominator with their post-hoc completed-workload area. +Without a capacity schedule, this is equivalent to configured nodes times the +elapsed horizon. **Get count of jobs waiting to be scheduled:** ```cpp diff --git a/docs/api/cpp/source/sim.rst b/docs/api/cpp/source/sim.rst index 95ff197..428970b 100644 --- a/docs/api/cpp/source/sim.rst +++ b/docs/api/cpp/source/sim.rst @@ -16,6 +16,7 @@ function, enum, or file name. .. doxygengroup:: dr_evt_sim :project: dr_evt :content-only: + :no-link: :inner: :members: :protected-members: diff --git a/docs/api/cpp/source/trace.rst b/docs/api/cpp/source/trace.rst index 044ddd7..18dbf4e 100644 --- a/docs/api/cpp/source/trace.rst +++ b/docs/api/cpp/source/trace.rst @@ -16,6 +16,7 @@ function, enum, or file name. .. doxygengroup:: dr_evt_trace :project: dr_evt :content-only: + :no-link: :inner: :members: :protected-members: diff --git a/docs/api/cpp/topics/scheduling-backfill.rst b/docs/api/cpp/topics/scheduling-backfill.rst index 07b8d81..97c9251 100644 --- a/docs/api/cpp/topics/scheduling-backfill.rst +++ b/docs/api/cpp/topics/scheduling-backfill.rst @@ -9,6 +9,7 @@ windows implement priority ordering and EASY or conservative backfilling. .. doxygengroup:: dr_evt_sim :project: dr_evt :content-only: + :no-link: :inner: :members: :protected-members: diff --git a/docs/dev/design-decisions/SIMULATION_VS_REPLAY_MODES.md b/docs/dev/design-decisions/SIMULATION_VS_REPLAY_MODES.md index b5c9edd..9f7b3b8 100644 --- a/docs/dev/design-decisions/SIMULATION_VS_REPLAY_MODES.md +++ b/docs/dev/design-decisions/SIMULATION_VS_REPLAY_MODES.md @@ -32,8 +32,10 @@ a reservation. Once a job starts, its selected duration determines the finish event. In replay mode, `begin_time` and `end_time` are authoritative. Their -difference supplies the duration, and the resource-accounting path processes -those recorded intervals directly. +difference supplies the duration when `actual_run_time` is absent. If an +actual runtime or one of its aliases is supplied, it must be finite and equal +that difference within `1e-6` seconds. The resource-accounting path processes +the recorded intervals directly. ## Record representation @@ -52,9 +54,11 @@ For accepted jobs: - start time is not earlier than submission time; and - runtime is positive. -Simulation additionally requires the recorded finish to equal the computed -start plus the selected runtime. Replay preserves the supplied start and finish -times. +For a runtime loaded from simulation input, the value must be finite and no +greater than `time_limit`; simulation later requires the recorded finish to +equal the computed start plus the selected runtime. Replay preserves the +supplied start and finish times and, when a runtime is also supplied, requires +it to agree with their difference. ## Implementation diff --git a/docs/dev/design-decisions/TIMEZONE_SUPPORT.md b/docs/dev/design-decisions/TIMEZONE_SUPPORT.md index 066a211..6cdb941 100644 --- a/docs/dev/design-decisions/TIMEZONE_SUPPORT.md +++ b/docs/dev/design-decisions/TIMEZONE_SUPPORT.md @@ -22,6 +22,11 @@ the gRPC simulation request. Internally, `Data_Columns` temporarily sets the process `TZ` value while parsing a trace and restores the previous value when the mapping is destroyed. +Input encoding is detected from the first data row of each input file and then +used for the entire file. Mixed encodings are rejected. `--timestamp_format` +is currently retained as a validated compatibility setting but does not select +input parsing or output formatting. Simulator output is always numeric. + The supported user-facing behavior is documented in: - [Trace Formats](../../user-guide/trace-formats.md) diff --git a/docs/index.md b/docs/index.md index 34e96a2..d74a5df 100644 --- a/docs/index.md +++ b/docs/index.md @@ -34,6 +34,7 @@ user-guide/trace-formats user-guide/output-traces user-guide/grpc-setup user-guide/client-server-use-cases +user-guide/maintenance-and-warm-start user-guide/fugaku-power-experiment ``` diff --git a/docs/user-guide/command-line.md b/docs/user-guide/command-line.md index 27db0b6..89132af 100644 --- a/docs/user-guide/command-line.md +++ b/docs/user-guide/command-line.md @@ -11,6 +11,8 @@ Complete reference for all DR_EVT command-line options for the `simulator` binar | Input/output | `-o, --outfile FILENAME` | Write the simulated job schedule. | | Input/output | `-R, --resource_trace FILENAME` | Write resource history. | | System | `-n, --total_nodes COUNT` | Set simulated cluster capacity. | +| System | `--capacity_schedule FILENAME` | Apply time-varying capacity change points. | +| System | `--sim_start_time TIME` | Set the global simulation start time as a nonnegative epoch value or ISO timestamp; a positive value warm-starts replay input. | | Scheduling | `-b, --backfill_policy POLICY` | Select `easy`, `conservative`, or `none`. | | Scheduling | `--num_max_candidates COUNT` | Cap candidates offered to the experimental selector. | | Scheduling | `-p, --priority_policy POLICY` | Select the job-ordering policy. | @@ -25,15 +27,15 @@ Complete reference for all DR_EVT command-line options for the `simulator` binar | Storage | `-H, --resource_history_capacity SIZE` | Set resource-history capacity. | | Trace | `--trace_type TYPE` | Select the standard or experimental record model. | | Trace | `-f, --trace_format FORMAT` | Select the input trace schema. | -| Trace | `-T, --timestamp_format FORMAT` | Select epoch or ISO input/output timestamps. | -| Trace | `-z, --timezone TIMEZONE` | Set the timezone for ISO output timestamps. | +| Trace | `-T, --timestamp_format FORMAT` | Set the retained timestamp-format compatibility value; input is auto-detected and output remains numeric. | +| Trace | `-z, --timezone TIMEZONE` | Interpret calendar timestamps that omit an explicit offset. | | Trace | `-M, --msec_output` | Preserve millisecond precision in output timestamps. | | Runtime | `-r, --run_time_mode MODE` | Select how actual execution lengths are determined. | | Runtime | `-D, --run_time_distribution TYPE` | Select the sampled runtime distribution. | | Runtime | `-S, --run_time_scale FACTOR` | Scale sampled job runtimes. | | Runtime | `-V, --run_time_stddev FACTOR` | Set sampled runtime variation. | | Limits | `-j, --max_jobs COUNT` | Limit the number of simulated jobs. | -| Limits | `-t, --max_time TIME` | Limit simulation time. | +| Limits | `-t, --max_time TIME` | Stop after processing events through this simulation timestamp. | | Other | `-s, --seed VALUE` | Set the random-number seed. | | Other | `-c, --config CONFIGFILE` | Load a Protobuf text configuration. | | Other | `-v, --verbose` | Enable verbose output. | @@ -121,6 +123,45 @@ Total number of nodes in the simulated cluster. ${CMAKE_INSTALL_PREFIX}/bin/simulator traces/jobs.csv --total_nodes 100 ``` +### `--capacity_schedule FILENAME` + +Apply a CSV of capacity change points during simulation. See +[Maintenance, Capacity Changes, and Warm Starts](maintenance-and-warm-start.md) +for the schema, semantics, detection tool, and initialization workflow. + +`--total_nodes` remains the physical maximum and the oversized-job rejection +threshold. Scheduled values may range from zero through that maximum. A job +that exceeds only the current scheduled capacity waits for a later increase; +it is not rejected. Reductions do not preempt running jobs, and a zero value +pauses all new starts for this workload. A schedule row at time zero replaces +the initial available capacity immediately. + +### `--sim_start_time TIME` + +Set the global simulation start time to `TIME`. It accepts the same timestamp +forms as trace columns: nonnegative Unix epoch seconds (including fractional +seconds) or a calendar timestamp such as `2024-01-01T00:00:10`. Calendar +values without an embedded UTC offset use `--timezone`. This parsing is +independent of `--timestamp_format`, and option order does not matter. + +This global boundary is distinct from each replay job's historical +`begin_time` and from job start times written to output. A positive value +enables the recommended replay-based warm start for replay-format input: jobs +with `begin_time < TIME` bypass the wait queue and seed live occupancy. They retain +their historical departures, but are omitted from job output and job +statistics. Jobs beginning at or after the boundary are rescheduled normally +when their submission is also at or after `TIME`. Zero preserves traditional +full replay. + +Only the warm stage tests for historical jobs. After its last departure, the +simulator switches to the ordinary event loop, so the steady-state path has no +per-job warm-start condition. Resource output begins with a baseline at +`TIME`; earlier samples and resource-area accounting are discarded. + +Jobs submitted before `TIME` but not yet running are excluded because +reconstructing an inherited wait queue requires an explicit policy. This mode +is currently incompatible with `--infile_list`. + ## Scheduling Policies ### `-b, --backfill_policy POLICY` @@ -353,15 +394,22 @@ ${CMAKE_INSTALL_PREFIX}/bin/simulator traces/simple.csv --trace_format simple ``` ### `-T, --timestamp_format FORMAT` -Timestamp format used to parse input and write output. +Retained timestamp-format compatibility setting. The current trace parser +detects numeric epoch seconds or calendar timestamps from the first data row +of each input file and uses that encoding for the rest of the file, regardless +of this setting. Simulated-job and resource-trace output is +numeric in both settings; `--msec_output` controls its precision. **Options:** -- `epoch` - Unix epoch seconds (e.g., `1693234567.0`) -- `iso` - ISO 8601 format (e.g., `2026-08-29T14:35:00-07:00`) +- `epoch` +- `iso` -**Default:** `iso` +**Default:** `epoch` -See [Input Trace Files](trace-formats.md) for accepted timestamp forms. +The option currently validates and retains one of these values for API and +configuration compatibility; it does not enforce the input encoding or +select the output encoding. See [Input Trace Files](trace-formats.md) for +accepted timestamp forms. **Example:** ```bash @@ -369,23 +417,31 @@ ${CMAKE_INSTALL_PREFIX}/bin/simulator traces/jobs.csv --timestamp_format epoch ``` ### `-z, --timezone TIMEZONE` -Timezone for ISO timestamp output. +Timezone used to interpret calendar timestamps that do not contain an +embedded UTC offset. It does not affect numeric input or simulator output and +does not depend on `--timestamp_format`. -**Format:** IANA timezone database name (e.g., `"America/Los_Angeles"`, `"UTC"`, `"America/New_York"`) +**Format:** IANA or POSIX timezone value understood by the system `TZ` +implementation (e.g., `"America/Los_Angeles"`, `"UTC"`, +`"America/New_York"`). DR_EVT does not currently validate the name itself. **Default:** `America/Los_Angeles` **Example:** ```bash ${CMAKE_INSTALL_PREFIX}/bin/simulator traces/jobs.csv \ - --timestamp_format iso \ --timezone "America/New_York" ``` +Embedded numeric offsets are recognized but currently have a known conversion +limitation when the configured timezone is not UTC. See +[Timezone Support](../dev/design-decisions/TIMEZONE_SUPPORT.md) before using +offset-bearing input. + ### `-M, --msec_output` Write timestamps in the simulated-job and resource traces with millisecond -precision instead of rounding them to integer seconds. +precision instead of truncating them to integer seconds. **Default:** Disabled @@ -400,7 +456,8 @@ ${CMAKE_INSTALL_PREFIX}/bin/simulator traces/jobs.csv --msec_output How to determine the job's actual, observed execution length in simulation mode. **Options:** -- `actual` - Read the job's actual run time from the input trace (default) +- `actual` - Read the job's finite actual run time from the input trace + (default); the value must not exceed `time_limit` - `distribution` - Sample from statistical distribution (realistic with variation) - `limit` - Jobs run exactly their time_limit (unrealistic, for debugging only) @@ -470,7 +527,15 @@ ${CMAKE_INSTALL_PREFIX}/bin/simulator traces/jobs.csv --max_jobs 100 ``` ### `-t, --max_time TIME` -Maximum simulation time (in trace time units). +Stop a batch run at the nonnegative numeric simulation timestamp `TIME`. +Events exactly at `TIME` are processed; events after it are not. Jobs still +running or waiting at the boundary remain in that state. The simulated-job +output contains only jobs completed through the boundary. + +`TIME` uses the same numeric time coordinate as the trace. It is an absolute +timestamp on that coordinate, not a duration relative to `--sim_start_time`. +When warm start is enabled, `TIME` must therefore be greater than or equal to +`--sim_start_time`. **Default:** Unlimited (run until all jobs complete) @@ -483,7 +548,7 @@ ${CMAKE_INSTALL_PREFIX}/bin/simulator traces/jobs.csv \ ### `-s, --seed VALUE` Random number generator seed for reproducibility. -**Default:** System clock +**Default:** `0` (deterministic) **Example:** ```bash diff --git a/docs/user-guide/grpc-setup.md b/docs/user-guide/grpc-setup.md index 9d0671a..8efbfbc 100644 --- a/docs/user-guide/grpc-setup.md +++ b/docs/user-guide/grpc-setup.md @@ -59,10 +59,10 @@ DR_EVT simulation server listening on 0.0.0.0:50051 and otherwise runs silently, one `Simulation` instance per connected session, until stopped (e.g. `Ctrl-C`, or however your process -supervisor manages it). A server starts with no job samples and does not load -the input file's data rows. The current `InitRequest` does require a -server-readable CSV path so the simulator can validate its header before jobs -are streamed. +supervisor manages it). A streaming session starts with no job samples and +does not load the input file's data rows until requested. `InitRequest` +requires a server-readable CSV path so the simulator can validate its header; +`RunRequest` loads and executes that trace for batch or replay operation. Bind `0.0.0.0` (shown above) so the server is reachable from other machines; bind `127.0.0.1` instead if you only need same-machine access. @@ -94,6 +94,13 @@ same `.proto` service. Its is a complete usage example. See the full guide for what each RPC corresponds to in the in-process [streaming API](../api/STREAMING_API.md). +For batch replay or warm start, set `InitRequest.sim_start_time` (zero for an +ordinary full run), point `infile` at the server-readable trace, and send +`RunRequest`. This is the global simulation start time, not a job's +`begin_time`. A positive simulation start time requires replay columns +including `begin_time` and `end_time`; negative and non-finite values are +rejected. + ## Session identity and completion Every initialization supplies a filename-safe `session_name`. The server adds diff --git a/docs/user-guide/maintenance-and-warm-start.md b/docs/user-guide/maintenance-and-warm-start.md new file mode 100644 index 0000000..2485aa7 --- /dev/null +++ b/docs/user-guide/maintenance-and-warm-start.md @@ -0,0 +1,193 @@ +# Maintenance, Capacity Changes, and Warm Starts + +Historical job traces can contain periods in which the whole machine is down, +only part of it is available, or the workload represented by the trace is +paused while another queue uses the hardware. DR_EVT can model the capacity +available to the simulated workload and initialize a mid-trace run from the +jobs that were active at its starting time. + +## Capacity schedule + +Pass `--capacity_schedule` a CSV with these columns: + +```text +time,total_nodes +1713139200,0 +1713160800,120000 +1713225600,158976 +``` + +Rows are change points and must have strictly increasing times. The first row +selects epoch or calendar encoding for the whole file; `time` accepts the same +forms as job traces. Capacity is +`--total_nodes` before the first row and each value remains effective until +the next row. A row at time zero therefore replaces the initial available +capacity immediately. Values must be between zero and `--total_nodes`. + +`--total_nodes` is the physical maximum, even when a capacity schedule is +present. It—not the currently scheduled capacity—is the oversized-job +rejection threshold. A job larger than the current scheduled capacity but no +larger than `--total_nodes` waits for a later capacity increase instead of +being rejected. + +```bash +simulator warm_start.csv \ + --trace_format simple --timestamp_format epoch \ + --run_time_mode actual --total_nodes 158976 \ + --capacity_schedule capacity.csv +``` + +Capacity reductions are non-preemptive. Existing jobs retain their nodes and +finish normally; no new job starts until it fits within current capacity. The +resource trace reports zero free nodes while a reduction leaves the system +temporarily overcommitted. Capacity increases are scheduling events, so +waiting jobs are reconsidered immediately even if no job arrives or completes +at that time. + +Instantaneous and time-accounted utilization use scheduled capacity rather +than the physical `--total_nodes` maximum. During a non-preemptive drain, +effective capacity is the larger of scheduled capacity and live allocation; +this excludes unavailable nodes without reporting utilization above 100%. + +A scalar capacity schedule describes the resources available to the workload +being simulated. To model a normal queue paused during a split-queue +maintenance period, use capacity zero for the normal workload even when jobs +in another queue continue running. Queue-specific partitions and job routing +are not encoded by this format. + +Protobuf configuration uses the corresponding field: + +```text +total_nodes: 158976 +capacity_schedule: "capacity.csv" +``` + +## Detect candidate periods + +`scripts/detect_capacity_periods.py` sweeps a historical replay trace and +finds sustained periods with a waiting backlog and either low allocation or +no starts. For example: + +```bash +python3 scripts/detect_capacity_periods.py historical.csv candidates.csv \ + --total-nodes 158976 \ + --utilization-threshold 0.25 \ + --minimum-duration 21600 \ + --capacity-schedule capacity.csv +``` + +Repeat `--queue-id ID` to restrict allocation, starts, and backlog detection +to the normal queue or queues. Without it, all jobs are included. The proposed +capacity is the peak observed allocation in each candidate period plus five +percent headroom; change that factor with `--capacity-headroom`. + +For a more comprehensive analysis, use the +`scripts/detect_queue_pause/` toolkit. It reconstructs allocation and pending +demand in configurable time bins, handles multiple trace files, distinguishes +queue pauses, shutdowns, and reduced-capacity periods, and can incorporate an +EASY-backfill opportunity audit. It is still heuristic: the additional +evidence improves classification but does not prove that maintenance or a +queue pause occurred. + +Neither tool is proof of maintenance. A job trace alone cannot distinguish +maintenance from insufficient demand, reservations, dependencies, or scheduler +policy. Review the generated candidates and external maintenance records before +using either capacity schedule. + +## Replay-based warm start (recommended) + +The simulator can consume a replay-format historical schedule directly. Warm +start records always include their historical `begin_time` and `end_time`. For +example, the replay-based warm-start test trace contains: + +```text +job_submit_time,begin_time,end_time,num_nodes,time_limit +0,0,20,40,20 +10,30,80,60,50 +20,60,65,10,5 +50,50,70,40,20 +55,60,90,60,30 +``` + +```bash +simulator tests/test_traces/feature/warm_start_native.csv \ + --trace_format simple --timestamp_format epoch \ + --sim_start_time 50 --run_time_mode actual --total_nodes 100 +``` + +Jobs with `begin_time < t` bypass the wait queue. The simulator replays them +only far enough to reconstruct live occupancy and policy state at `t`, then +discards all earlier resource samples and resets resource-area accounting. +Those seed jobs retain their historical end events so their nodes remain busy +after `t`, but their records are suppressed before completion and therefore +never enter job output or statistics. Jobs submitted at or after `t` are +rescheduled through the normal wait queue; a job beginning exactly at `t` is +not classified as historical. + +Here `t=50` is the global simulation boundary selected by +`--sim_start_time 50`. It is not a per-job `begin_time`. The first +job is warmup history that finishes before `t`. The second starts before `t` +and remains active until its historical end at `80`. The third was submitted +before `t` but had not started, so it is excluded. The last two are submitted +at or after `t` and go through the scheduler. + +Warm start and runtime selection are separate concerns. `--sim_start_time` +classifies the warmup records; every such record always uses its historical +`begin_time` and `end_time`. `--run_time_mode actual` in this command is an +independent choice for the jobs simulated by the scheduler. Those ordinary +jobs follow the same `actual`, `limit`, or `distribution` runtime selection as +any non-warm-start simulation before they enter the wait queue. + +During this temporary warm stage, completion processing checks whether the +job began before `t`. Once the final historical job departs, execution changes +to the ordinary simulation stage, whose compiled event loop contains no such +check. Ordinary jobs can still be scheduled while a warmup job is active; the +two stages select event-processing paths rather than imposing a scheduling +barrier. Jobs submitted before `t` but not yet running are intentionally +excluded because inheriting a historical wait queue is a separate scheduling +policy choice. + +### Synthetic initialization traces (legacy workaround) + +`prepare_warm_start_trace.py` is an alternative workaround for environments +that cannot use the recommended replay-based `--sim_start_time` path. It +creates an ordinary simulation-format trace containing synthetic initialization +jobs. It selects every historical job satisfying +`begin_time <= t < end_time`. A job beginning exactly at `t` is already active; +a job ending at `t` is complete. The helper submits initialization jobs at `t` +with both `actual_run_time` and `time_limit` set to `end_time - t`, then appends +future workload arrivals. + +```bash +python3 scripts/prepare_warm_start_trace.py \ + historical_schedule.csv warm_start.csv \ + --start-time 1713139200 \ + --workload-trace simulation_input.csv \ + --total-nodes 158976 +``` + +The simulator does not interpret the output's `initialization` or +`source_job_index` columns; they are provenance only. Initialization records +enter the ordinary wait queue, scheduler, output, and statistics. Run this +trace as an ordinary simulation with `--run_time_mode actual`, without +simulator `--sim_start_time`. Initialization rows sort before ordinary jobs +submitted at the same timestamp so they normally establish the initial +allocation first. + +`--run_time_mode limit` happens to give initialization records the same +remaining duration because their two runtime fields are equal, but it also +changes every future ordinary job to use its requested limit. +`--run_time_mode distribution` resamples initialization durations and therefore +does not preserve their historical departures. Because runtime mode is global, +this workaround cannot protect historical departures independently of future +job runtime selection as the replay-based method does. + +The helper reports the initialization job and node counts and can reject a +state larger than `--total-nodes`. The capacity schedule effective at `t` must +also be at least that large, or the scheduler cannot start all initialization +jobs at `t`. + +The helper remains available when a standalone simulation-format artifact is +required, but it is not the recommended warm-start method. Its initialization +records are ordinary output jobs, unlike replay-based `--sim_start_time` seed +jobs, which are deliberately omitted from job output and statistics. diff --git a/docs/user-guide/output-traces.md b/docs/user-guide/output-traces.md index 5f1d298..e023078 100644 --- a/docs/user-guide/output-traces.md +++ b/docs/user-guide/output-traces.md @@ -1,8 +1,10 @@ # Output Trace Files The simulator can write two CSV outputs for each run: a simulated-job schedule -and a resource-usage trace. Both use the timestamps selected by -`--timestamp_format`; see [Command-Line Options](command-line.md). +and a resource-usage trace. Both currently write numeric timestamps regardless +of `--timestamp_format`. By default they are truncated to whole seconds; +`--msec_output` writes three decimal places. See +[Command-Line Options](command-line.md). ## Simulated-job schedule @@ -13,8 +15,10 @@ ${CMAKE_INSTALL_PREFIX}/bin/simulator input.csv --outfile results/jobs_sim.csv ``` If `--outfile` is omitted, DR_EVT derives a filename from the input trace -(for example, `jobs.csv` becomes `jobs_sim.csv`). The CSV contains one row for -each scheduled job. In the default ID-input build, an input trace that +(for example, `jobs.csv` becomes `jobs_sim.csv`). Normally, the CSV contains +one row for each scheduled job. When `--max_time` is set, it contains only +jobs completed at or before that inclusive boundary; jobs still running or +waiting are omitted. In the default ID-input build, an input trace that explicitly provides `q_id` retains that column: ```text @@ -48,6 +52,9 @@ ${CMAKE_INSTALL_PREFIX}/bin/simulator input.csv \ ``` The resource trace records the occupancy after each resource-state change. +With `--max_time`, no event after the inclusive boundary is recorded; if no +resource-state change occurs exactly at the boundary, no synthetic final row +is added there. With the default `--trace_type standard`, the format is: ```text diff --git a/docs/user-guide/overview.md b/docs/user-guide/overview.md index 54bb805..5a8f542 100644 --- a/docs/user-guide/overview.md +++ b/docs/user-guide/overview.md @@ -17,6 +17,9 @@ DR_EVT accepts: [command-line options](command-line.md), a [Protobuf text configuration](protobuf-config.md), or both. +Historical maintenance and mid-trace experiments can use +[time-varying capacity and trace-consistent warm starts](maintenance-and-warm-start.md). + It produces: - a scheduled-job trace and a resource-usage trace, documented in @@ -56,5 +59,6 @@ For deployment patterns, see - [Backfilling Algorithms](../BACKFILLING_ALGORITHMS.md) - [Fugaku Power-Usage Experiment](fugaku-power-experiment.md) +- [Maintenance, Capacity Changes, and Warm Starts](maintenance-and-warm-start.md) - [Testing Guide](../TESTING_GUIDE.md) - [Developer Notes](../dev/README.md) diff --git a/docs/user-guide/protobuf-config.md b/docs/user-guide/protobuf-config.md index 5ebd50a..07d8b50 100644 --- a/docs/user-guide/protobuf-config.md +++ b/docs/user-guide/protobuf-config.md @@ -15,7 +15,7 @@ Configuration field names match the long command-line names with the leading | Group | Fields | |---|---| | Input and output | `infile`, `infile_list`, `outfile`, `resource_trace` | -| Limits | `max_jobs`, `max_time`, `total_nodes` | +| Limits/system | `max_jobs`, `max_time`, `sim_start_time`, `total_nodes`, `capacity_schedule` | | Scheduling | `backfill_policy`, `priority_policy`, `num_max_candidates`, `queue_impl`, `block_size` | | Queue storage | `wait_queue_capacity`, `wait_queue_overflow` | | Job storage | `job_store_capacity`, `job_store_overflow`, `job_flush_interval`, `memory_pressure_fraction` | @@ -29,6 +29,10 @@ interactions. Input schemas are documented in [Input Trace Files](trace-formats.md), and generated files in [Output Trace Files](output-traces.md). +`max_time`, when positive, is an inclusive absolute simulation-time boundary: +events at that timestamp are processed and later events are left pending. It +must be greater than or equal to `sim_start_time` when warm start is enabled. + ## File format The file contains fields from the `Simulation_Params` message at the top @@ -39,6 +43,7 @@ infile: "trace.csv" outfile: "results.csv" resource_trace: "resources.csv" total_nodes: 1000 +sim_start_time: 1713139200 backfill_policy: "easy" priority_policy: "fcfs" trace_format: "simple" diff --git a/docs/user-guide/trace-formats.md b/docs/user-guide/trace-formats.md index f0c02d1..214dba7 100644 --- a/docs/user-guide/trace-formats.md +++ b/docs/user-guide/trace-formats.md @@ -11,8 +11,15 @@ The `standard` data model contains scheduling and resource fields. The `pcon` model adds per-job `avgpcon`, `minpcon`, and `maxpcon` values and corresponding resource-trace columns. -Timestamps may be Unix epoch seconds or ISO 8601 strings. ISO input and output -use the timezone selected by `--timezone`. +The first data row of each input file is used to detect Unix epoch seconds or +calendar timestamps, and that encoding is then used for every timestamp in +the file. Mixed encodings in one file are rejected. `--timestamp_format` does +not enforce an encoding. Calendar input without an embedded UTC offset uses +the timezone selected by `--timezone`. +Simulator output is numeric regardless of `--timestamp_format`; use +`--msec_output` to retain milliseconds. Embedded numeric offsets have a known +conversion limitation described in +[Timezone Support](../dev/design-decisions/TIMEZONE_SUPPORT.md). ## Simple Format CSV Structure @@ -24,6 +31,25 @@ job_submit_time,num_nodes,time_limit 120,10,80 ``` +To replay observed execution lengths while still letting the scheduler compute +new start and end times, include `actual_run_time`: + +```text +job_submit_time,num_nodes,time_limit,actual_run_time +0,10,100,75 +50,10,50,30 +120,10,80,60 +``` + +Run this simulation-format trace with `--run_time_mode actual`. The scheduler +uses `time_limit` for reservation planning, while each job releases its nodes +after `actual_run_time` seconds: + +```bash +simulator jobs.csv --trace_format simple --timestamp_format epoch \ + --run_time_mode actual --total_nodes 100 +``` + ### Replay Mode, With Epoch Timestamps ```text job_submit_time,begin_time,end_time,num_nodes,time_limit @@ -67,19 +93,25 @@ determines simulation vs replay mode (see below). | `num_nodes` | Number of nodes requested | Both modes | | `q_id` | Optional one-based queue ID. If absent, the job uses `1` (`Queue1`). | Both modes | | `time_limit` | User-provided time limit (seconds). Accepted column-name aliases: `time_limit`, `timelimit`, `walltime` | Both modes | -| `begin_time` | Historical start time from trace | Replay mode only; must appear together with `end_time` | +| `begin_time` | Historical start time of this individual job; distinct from the global `--sim_start_time` boundary | Replay mode only; must appear together with `end_time` | | `end_time` | Historical end time from trace | Replay mode only; must appear together with `begin_time` | -| `duration` | Accepted alias for `actual_run_time` | Simulation mode, only with `--run_time_mode actual` | +| `duration` | Accepted alias for `actual_run_time`; in replay mode it is only checked against `end_time - begin_time` and does not control execution | Required for simulation `--run_time_mode actual` unless another runtime alias is present; optional in replay mode | | `avgpcon` | Average power usage associated with the job | Required only with `--trace_type pcon`; ignored in standard mode | | `minpcon` | Minimum power usage associated with the job | Required only with `--trace_type pcon`; ignored in standard mode | | `maxpcon` | Maximum power usage associated with the job | Required only with `--trace_type pcon`; ignored in standard mode | | `exit_status` | Output-only compatibility field. The simulator currently writes `0`. | Generated output only | -| `actual_run_time` | The job's real, historical run time (seconds); used by `--run_time_mode actual`. Accepted column-name aliases: `actual_run_time`, `duration`, `actual_duration`, `run_time` | Simulation mode, only with `--run_time_mode actual` | +| `actual_run_time` | The job's real, historical run time in seconds. It determines execution only for simulation `--run_time_mode actual`; in replay mode, `begin_time` and `end_time` determine execution and this field is only checked for consistency. Accepted aliases: `actual_runtime`, `duration`, `actual_duration`, `run_time` | Required for simulation `--run_time_mode actual`; optional in replay mode | `time_limit` and `actual_run_time` accept the aliases listed above. If multiple aliases for one field are present, the first listed match is used. Lassen input uses fixed column positions instead of header names. +A supplied `actual_run_time` must be finite. In simulation input it must not +exceed `time_limit`. In replay input it must equal `end_time - begin_time` +within `1e-6` seconds; if omitted, that timestamp difference supplies the +runtime. Invalid rows are diagnosed and skipped while the rest of the trace is +loaded. + **Ignored input columns**: input fields not used by the selected trace format are ignored. In particular, `exit_status` is accepted only so a generated simulator output can be used as replay input; its value is never read or used diff --git a/experimental/fugaku-power/scripts/analysis/summarize_resource_traces.py b/experimental/fugaku-power/scripts/analysis/summarize_resource_traces.py new file mode 100644 index 0000000..24027bc --- /dev/null +++ b/experimental/fugaku-power/scripts/analysis/summarize_resource_traces.py @@ -0,0 +1,96 @@ +#!/usr/bin/env python3 +"""Stream-merge resource traces and report aggregate peak usage.""" + +import argparse +import csv +import heapq +from contextlib import ExitStack +from pathlib import Path + + +VALUE_COLUMNS = ("allocated_nodes", "avgpcon", "minpcon", "maxpcon") + + +def read_row(reader, indexes, path): + line = next(reader, None) + if line is None: + return None + fields = line.rstrip("\r\n").split(",") + try: + timestamp = int(float(fields[indexes["time"]])) + allocated = int(fields[indexes["allocated_nodes"]]) + values = ( + allocated, + float(fields[indexes["avgpcon"]]), + float(fields[indexes["minpcon"]]), + float(fields[indexes["maxpcon"]]), + ) + except (IndexError, ValueError) as error: + raise ValueError(f"invalid resource row in {path}: {line!r}") from error + return timestamp, values + + +def summarize(paths): + states = [(0, 0.0, 0.0, 0.0) for _ in paths] + heap = [] + readers = [] + + with ExitStack() as stack: + for index, path in enumerate(paths): + stream = stack.enter_context(path.open()) + header = stream.readline().rstrip("\r\n").split(",") + required = ("time",) + VALUE_COLUMNS + missing = [column for column in required if column not in header] + if missing: + raise ValueError(f"{path} is missing: {', '.join(missing)}") + indexes = {column: header.index(column) for column in required} + readers.append((iter(stream), indexes, path)) + row = read_row(readers[-1][0], indexes, path) + if row is not None: + heapq.heappush(heap, (row[0], index, row[1])) + + peak_nodes = 0 + peak_nodes_time = 0 + peak_avgpcon = 0.0 + peak_avgpcon_time = 0 + timestamps = 0 + + while heap: + timestamp = heap[0][0] + while heap and heap[0][0] == timestamp: + _, index, values = heapq.heappop(heap) + states[index] = values + reader, indexes, path = readers[index] + row = read_row(reader, indexes, path) + if row is not None: + heapq.heappush(heap, (row[0], index, row[1])) + + allocated = sum(state[0] for state in states) + avgpcon = sum(state[1] for state in states) + if allocated > peak_nodes: + peak_nodes = allocated + peak_nodes_time = timestamp + if avgpcon > peak_avgpcon: + peak_avgpcon = avgpcon + peak_avgpcon_time = timestamp + timestamps += 1 + + return peak_nodes, peak_nodes_time, peak_avgpcon, peak_avgpcon_time, timestamps + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("resource_traces", nargs="+", type=Path) + args = parser.parse_args() + + result = summarize(args.resource_traces) + print(f"input_files={len(args.resource_traces)}") + print(f"resource_timestamps={result[4]}") + print(f"peak_allocated_nodes={result[0]}") + print(f"peak_allocated_time={result[1]}") + print(f"peak_avgpcon={result[2]:.6f}") + print(f"peak_avgpcon_time={result[3]}") + + +if __name__ == "__main__": + main() diff --git a/python/dr_evt_bindings.cpp b/python/dr_evt_bindings.cpp index 28d8aef..531d8cf 100644 --- a/python/dr_evt_bindings.cpp +++ b/python/dr_evt_bindings.cpp @@ -66,6 +66,11 @@ PYBIND11_MODULE(dr_evt, m) { "initialize_trace().") .def_readwrite("total_nodes", &Sim_Params::m_total_nodes, "int: Total scheduler-managed compute nodes.") + .def_readwrite("capacity_schedule", &Sim_Params::m_capacity_schedule, + "str: Optional capacity-change CSV path.") + .def_readwrite("sim_start_time", &Sim_Params::m_sim_start_time, + "float: Global simulation start time; a positive value " + "enables replay-based warm start for replay input.") .def_readwrite("trace_format", &Sim_Params::m_trace_format, "str: Input trace format identifier.") .def_readwrite("timestamp_format", &Sim_Params::m_timestamp_format, diff --git a/scripts/README.md b/scripts/README.md index 717a191..33f9081 100644 --- a/scripts/README.md +++ b/scripts/README.md @@ -27,6 +27,18 @@ The owning fixtures and runner commands are listed in ## Trace analysis +- `detect_capacity_periods.py` — find sustained low-allocation/backlog or + no-start/backlog periods and optionally emit a candidate capacity schedule. + For more comprehensive analysis, use the `detect_queue_pause/` toolkit, + which reconstructs binned demand and allocation, supports optional EASY- + backfill evidence, and reports richer operating-state evidence. Both tools + are heuristic; validate their candidates against scheduler and maintenance + records. +- `prepare_warm_start_trace.py` — legacy workaround that materializes jobs + crossing `t` as ordinary simulation records with their remaining durations. + Prefer the replay-based `--sim_start_time` method, which preserves historical + departures independently of `--run_time_mode` and suppresses historical seeds + from output and statistics. - `analyze_trace_performance.py` — summary performance analysis. - `calculate_resource_trace.py` — derive resource occupancy from a schedule. - `trace/detect_abnormality.awk` — flag anomalous trace records. @@ -36,3 +48,6 @@ The owning fixtures and runner commands are listed in The separate Fugaku experiment tools are documented in [`experimental/fugaku-power/scripts/README.md`](../experimental/fugaku-power/scripts/README.md). + +The maintenance tools and their assumptions are documented in +[Maintenance, Capacity Changes, and Warm Starts](../docs/user-guide/maintenance-and-warm-start.md). diff --git a/scripts/detect_capacity_periods.py b/scripts/detect_capacity_periods.py new file mode 100755 index 0000000..7c78deb --- /dev/null +++ b/scripts/detect_capacity_periods.py @@ -0,0 +1,192 @@ +#!/usr/bin/env python3 +"""Detect review candidates for reduced capacity or a paused queue.""" + +import argparse +import csv +from datetime import datetime +import math +import os +import sys +import time + + +def parse_time(value, timezone): + value = value.strip() + try: + return float(value) + except ValueError: + pass + parsed = datetime.fromisoformat(value.replace("Z", "+00:00")) + if parsed.tzinfo is not None: + return parsed.timestamp() + old_tz = os.environ.get("TZ") + try: + os.environ["TZ"] = timezone + time.tzset() + return time.mktime(parsed.timetuple()) + parsed.microsecond / 1e6 + finally: + if old_tz is None: + os.environ.pop("TZ", None) + else: + os.environ["TZ"] = old_tz + time.tzset() + + +def pick(row, *names): + for name in names: + if row.get(name, "") != "": + return row[name] + raise ValueError("missing required column (one of: {})".format( + ", ".join(names))) + + +def merge_periods(periods, minimum_duration): + periods.sort(key=lambda item: (item[0], item[1])) + merged = [] + for start, end, reason, peak_allocated, peak_waiting in periods: + if merged and start <= merged[-1][1]: + current = merged[-1] + current[1] = max(current[1], end) + current[2].add(reason) + current[3] = max(current[3], peak_allocated) + current[4] = max(current[4], peak_waiting) + else: + merged.append([start, end, {reason}, peak_allocated, peak_waiting]) + return [item for item in merged if item[1] - item[0] >= minimum_duration] + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("historical_trace") + parser.add_argument("candidates_csv") + parser.add_argument("--capacity-schedule", + help="also write simulator time,total_nodes CSV") + parser.add_argument("--total-nodes", type=int, + help="known normal capacity; defaults to observed peak") + parser.add_argument("--utilization-threshold", type=float, default=0.25) + parser.add_argument("--minimum-duration", type=float, default=3600.0) + parser.add_argument("--minimum-waiting-jobs", type=int, default=1) + parser.add_argument("--capacity-headroom", type=float, default=1.05) + parser.add_argument("--queue-id", action="append", default=[], + help="limit detection to these q_id values (repeatable)") + parser.add_argument("--timezone", default="America/Los_Angeles") + args = parser.parse_args() + if not 0.0 <= args.utilization_threshold <= 1.0: + parser.error("--utilization-threshold must be between 0 and 1") + if args.minimum_duration <= 0 or args.capacity_headroom < 1.0: + parser.error("duration must be positive and headroom must be >= 1") + + selected_queues = set(args.queue_id) + events = {} + with open(args.historical_trace, newline="") as stream: + for row in csv.DictReader(stream): + queue = row.get("q_id", row.get("queue", "")) + if selected_queues and queue not in selected_queues: + continue + submit = parse_time(pick(row, "job_submit_time", "submit_time"), + args.timezone) + begin = parse_time(pick(row, "begin_time", "start_time"), + args.timezone) + end = parse_time(pick(row, "end_time"), args.timezone) + nodes = int(pick(row, "num_nodes", "nodes", "nodes_requested")) + if not submit <= begin <= end: + raise ValueError("job times must satisfy submit <= begin <= end") + events.setdefault(submit, [0, 0, 0])[0] += 1 + events.setdefault(begin, [0, 0, 0])[0] -= 1 + events[begin][1] += nodes + events[begin][2] += 1 + events.setdefault(end, [0, 0, 0])[1] -= nodes + + if not events: + raise ValueError("no matching jobs found") + times = sorted(events) + allocated = 0 + waiting = 0 + peak = 0 + states = [] + waiting_since = None + last_start = None + for pos, point in enumerate(times[:-1]): + wait_delta, allocation_delta, starts = events[point] + old_waiting = waiting + waiting += wait_delta + allocated += allocation_delta + if waiting < 0 or allocated < 0: + raise ValueError("trace produces negative waiting/allocation state") + peak = max(peak, allocated) + if waiting == 0: + waiting_since = None + elif old_waiting == 0 or waiting_since is None: + waiting_since = point + if starts: + last_start = point + states.append((point, times[pos + 1], allocated, waiting, + waiting_since, last_start)) + + normal_capacity = args.total_nodes if args.total_nodes is not None else peak + if normal_capacity <= 0: + raise ValueError("normal capacity must be positive") + threshold_nodes = normal_capacity * args.utilization_threshold + raw = [] + for start, end, used, queued, queue_since, last_started in states: + if queued < args.minimum_waiting_jobs or end <= start: + continue + if used <= threshold_nodes: + raw.append((start, end, "low_allocation_with_backlog", used, queued)) + anchor = queue_since + if last_started is not None and (anchor is None or last_started > anchor): + anchor = last_started + if anchor is not None and end - anchor >= args.minimum_duration: + raw.append((anchor, end, "no_starts_with_backlog", used, queued)) + + periods = merge_periods(raw, args.minimum_duration) + fieldnames = ["start_time", "end_time", "duration_seconds", + "inferred_nodes", "reason", "peak_allocated", + "peak_waiting_jobs"] + with open(args.candidates_csv, "w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=fieldnames) + writer.writeheader() + for start, end, reasons, used, queued in periods: + inferred = min(normal_capacity, + int(math.ceil(used * args.capacity_headroom))) + writer.writerow({ + "start_time": "{:.15g}".format(start), + "end_time": "{:.15g}".format(end), + "duration_seconds": "{:.15g}".format(end - start), + "inferred_nodes": inferred, + "reason": "+".join(sorted(reasons)), + "peak_allocated": used, + "peak_waiting_jobs": queued, + }) + + if args.capacity_schedule: + changes = {} + for start, end, reasons, used, queued in periods: + del reasons, queued + inferred = min(normal_capacity, + int(math.ceil(used * args.capacity_headroom))) + if inferred >= normal_capacity: + continue + changes[start] = min(changes.get(start, normal_capacity), inferred) + changes[end] = normal_capacity + with open(args.capacity_schedule, "w", newline="") as stream: + writer = csv.writer(stream, lineterminator="\n") + writer.writerow(["time", "total_nodes"]) + previous = normal_capacity + for point in sorted(changes): + capacity = changes[point] + if capacity != previous: + writer.writerow(["{:.15g}".format(point), capacity]) + previous = capacity + + print("normal_capacity={}".format(normal_capacity)) + print("candidate_periods={}".format(len(periods))) + print("warning=capacity is inferred from workload evidence; review before simulation") + + +if __name__ == "__main__": + try: + main() + except (OSError, ValueError) as error: + print("error: {}".format(error), file=sys.stderr) + sys.exit(1) diff --git a/scripts/detect_queue_pause/README.md b/scripts/detect_queue_pause/README.md new file mode 100644 index 0000000..49e83a7 --- /dev/null +++ b/scripts/detect_queue_pause/README.md @@ -0,0 +1,422 @@ +# Scheduler capacity and queue-pause inference + +This toolkit converts historical batch-scheduler job traces into hourly +normal-queue capacity schedules. It detects sustained behavior consistent with: + +- a full system shutdown; +- a paused normal queue while limited/special work continues; or +- operation at reduced capacity. + +It distinguishes maintenance-like behavior from ordinary idle time by requiring +queued demand: jobs must have been submitted but not yet started. Node occupancy +is integrated exactly within each time bin, including jobs crossing file and +month boundaries. + +## Input format + +The complete workflow accepts one CSV path or glob. Every input row requires: + +- `job_submit_time`, `begin_time`, and `end_time` as Unix epoch seconds; +- `num_nodes` as a positive integer node request; and +- `time_limit` as the requested runtime in seconds. + +An optional `avgpcon` column improves workload-family separation. Other +columns are preserved by the source files but are not required. Multiple input +files may overlap; choose the overlap policy appropriate for how those exports +were produced. + +## Complete workflow + +Install the Python dependencies first: + +```bash +python3 -m pip install -r requirements.txt +``` + +Configure the machine and trace semantics through environment variables: + +```bash +CAPACITY_NODES=10000 \ +TRACE_TIMEZONE=UTC \ +OVERLAP_POLICY=combine \ +GRACE_MINUTES=60 \ +RELEASE_DELAY_MINUTES=10 \ +./generate_capacity_schedule.sh 'traces/*.csv' output +``` + +`OVERLAP_POLICY=combine` treats overlapping files as distinct concurrent +records. Use `filename-month` only for files named +`YY_MM_scheduling_trace.csv` whose adjacent exports duplicate coverage. Run +`./generate_capacity_schedule.sh --help` for all wrapper settings. + +The workflow produces replay evidence, an hourly timeline, an all-state JSON +report, two capacity-schedule CSVs, and two silhouette plots. + +### Components + +| Script | General role | +| --- | --- | +| `generate_capacity_schedule.sh` | Runs the complete pipeline with explicit machine and trace settings. | +| `discover_maintenance.py` | Reconstructs hourly allocation and demand, then classifies anomalous periods. | +| `backfill_opportunity_audit.py` | Replays conservative EASY scheduling opportunities to filter policy-like delays. | +| `build_resource_capacity_trace.py` | Converts the report into hourly scheduler-capacity scenarios. | +| `plot_capacity_silhouette.py` | Draws non-interpolated capacity-change cases with local-date labels. | + +The Python programs can be used independently. The shell driver is the normal +entry point when starting from raw trace files. + +### Fugaku example + +The included Fugaku traces use JST month boundaries, duplicated adjacent-file +coverage, and a configured capacity of 158,976 nodes: + +```bash +CAPACITY_NODES=158976 \ +TRACE_TIMEZONE=JST \ +OVERLAP_POLICY=filename-month \ +./generate_capacity_schedule.sh '*_scheduling_trace.csv' . +``` + +These are dataset-specific values, not general requirements. + +## Running the detector directly + +```bash +python3 discover_maintenance.py '*_scheduling_trace.csv' \ + --overlap-policy combine --timezone UTC \ + --output maintenance_report.json \ + --bins-output capacity_timeline.csv +``` + +Only periods classified as queue pauses are written by default. This filters +the JSON `periods` array but deliberately keeps `capacity_timeline.csv` +complete. `--only-queue-pause` may still be supplied explicitly: + +```bash +python3 discover_maintenance.py 'traces/*.csv' \ + --overlap-policy combine --timezone UTC \ + --known-capacity 10000 \ + --backfill-evidence backfill_opportunities.csv \ + --only-queue-pause \ + --output queue_pause_report.json \ + --bins-output capacity_timeline.csv +``` + +Use `--all-states` to include reduced-capacity, backfill-suppression, and +shutdown classifications in addition to queue pauses. + +For monthly exports with duplicated coverage, `--overlap-policy +filename-month` prevents double counting by assigning each +`YY_MM_scheduling_trace.csv` file ownership of only its filename month in the +selected timezone. Jobs, pending intervals, and event counts are clipped at +those month boundaries before the files are stitched together. + +The JSON report includes inferred normal capacity, monthly observed-capacity +estimates, detected periods, evidence metrics, confidence, and limitations. If +physical capacity is known, prefer `--known-capacity N`; trace-only inference is +necessarily a lower bound when the workload never fills the machine. + +Timestamps are Unix epoch seconds by default. For readable timestamps, select +the appropriate timezone, for example: + +```bash +python3 discover_maintenance.py 'traces/*.csv' \ + --time-format iso --timezone America/Los_Angeles +``` + +Fixed abbreviations such as `PST` (UTC-08:00) and `JST` (UTC+09:00), numeric +offsets such as `+09:00`, and installed IANA zones are accepted. Prefer +`America/Los_Angeles` over `PST` when daylight-saving transitions should be +applied, and `Asia/Tokyo` is the IANA equivalent of JST. + +The optional timeline CSV has these workload fields: + +- `pending_jobs`: time-weighted average number of submitted jobs that had not + begun during the bin; +- `pending_nodes`: time-weighted average sum of requested nodes for those jobs; +- `jobs_submitted`: number of jobs submitted during the bin; and +- `nodes_started`: sum of `num_nodes` for jobs whose execution began in the bin. +- `instantaneous_running_nodes`: exact allocated nodes at the bin's start, + reconstructed from job start and end events rather than averaged over the + hour. + +Because the first two values are time-weighted averages, they can be fractional. + +## Detection method + +### 1. Reconstruct the timeline + +Each job contributes `num_nodes` to `running_nodes` from `begin_time` up to, but +not including, `end_time`. It contributes one job and `num_nodes` to the pending +values from `job_submit_time` up to `begin_time`. These contributions are +integrated over each bin, so jobs that begin or end partway through a bin are +represented proportionally. The default bin width is one hour. + +`jobs_submitted` is an event count: it counts jobs whose `job_submit_time` falls +inside the bin. It does **not** mean submitted and still waiting; that quantity +is represented by `pending_jobs`. + +### 2. Estimate normal capacity + +Unless `--known-capacity` is provided, normal capacity is estimated as the +99.5th percentile of nonzero `running_nodes` values: + +```text +normal_capacity = quantile(running_nodes, 0.995) +``` + +This estimates the observed schedulable envelope, not necessarily the physical +machine size. It is a lower bound when the workload never fills the machine. +Supplying the known physical or normal-queue capacity is preferable. + +The report also calculates a capacity estimate for each calendar month. It uses +bins with queue pressure when at least 24 such bins exist, and otherwise marks +the monthly estimate as a low-demand or partial-window lower bound. + +### 3. Require queue pressure + +A bin is considered to have demand when either condition is true: + +```text +pending_jobs >= 10 +pending_nodes >= normal_capacity * 0.02 +``` + +Requiring demand is important: a machine with no work to run is ordinarily idle, +not undergoing maintenance. Both thresholds are configurable. + +### 4. Classify individual bins + +For each bin, the detector calculates: + +```text +capacity_fraction = running_nodes / normal_capacity +``` + +It then applies the following default rules: + +| Candidate state | Conditions while queue pressure exists | +| --- | --- | +| Full shutdown | Capacity fraction is at most 0.5% and no jobs start | +| Reduced capacity | Capacity fraction is at most 60% | +| Queue pause or maintenance | Capacity fraction is below 85% and the job-start rate is at most 10% of its typical rate | + +The typical start rate is the median positive number of job starts in pressured +bins operating at or above the reduced-capacity threshold. If that sample is +empty, the median of all nonzero start bins is used. + +### 5. Merge and label periods + +Adjacent candidate bins are merged. One intervening normal bin is bridged by +default, which avoids splitting a maintenance period because of a brief burst of +activity. A merged period must last at least one hour. + +The merged period is labeled as follows: + +- `full_shutdown` when at least 60% of its bins meet the shutdown rule and no + jobs start in the period; +- `queue_pause_or_maintenance` when at least 60% meet the queue-pause rule; or +- `reduced_capacity` when at least 60% meet the reduced-capacity rule; or +- `backfill_suppression` when replay evidence exists without a sufficiently low + allocation level for either preceding classification. + +Confidence is a heuristic from 0 to 0.99. It increases with queue pressure, +severity of the allocation reduction, and duration. It is not a statistical +probability. + +### Interpretation limitations + +Pending jobs are not necessarily ready to run. Dependencies, reservations, +resource-shape constraints, or policy can produce low allocation despite a +large pending queue. Canceled jobs and jobs omitted from the traces are also +invisible. The reported periods should therefore be treated as candidates for +validation against maintenance logs rather than definitive maintenance records. + +Useful tuning controls include `--bin-minutes`, `--min-duration-hours`, +`--reduced-fraction`, and the pending-demand thresholds. Run +`python3 discover_maintenance.py --help` for all options. + +When an input has no queue or partition column, the script can identify a +behavioral queue pause but cannot name or prove which queue caused it. Adding +such a field to future traces would allow direct per-queue analysis. + +## Resource-capacity scenario traces + +`build_resource_capacity_trace.py` produces two hourly normal-queue capacity +scenarios while keeping `structural_capacity_nodes` fixed at the configured +machine capacity. For the Fugaku example: + +```bash +python3 build_resource_capacity_trace.py \ + --report maintenance_report.json \ + --timeline capacity_timeline.csv \ + --capacity 158976 +``` + +By default these files retain the detailed hourly analysis columns. Add +`--simulator-format` to write compact files accepted directly by +`simulator --capacity_schedule`: + +```bash +python3 build_resource_capacity_trace.py \ + --report maintenance_report.json \ + --timeline capacity_timeline.csv \ + --capacity 158976 \ + --simulator-format +``` + +That mode writes only `time,total_nodes`, removes consecutive duplicate +capacities, and restores full capacity at the end of the analyzed timeline if +its final interval was reduced or paused. The timestamps are Unix epoch +seconds. Pass the same value used for `--capacity` to simulator +`--total_nodes`. + +For the complete wrapper, set `SIMULATOR_FORMAT=1` to select this output mode. + +- `resource_capacity_with_reduced_capacity.csv` sets normal-queue capacity to + zero during inferred queue pauses or shutdowns and applies the report's + evidence-derived capacity value during inferred reduced-capacity periods. +- `resource_capacity_without_reduced_capacity.csv` sets normal-queue capacity + to zero during inferred queue pauses or shutdowns but assumes full capacity + during periods labeled `reduced_capacity`. +- Both assume full capacity during normal time and `backfill_suppression`. + +These are scheduler-facing scenario traces, not claims that a machine's +physical node count changed. The reduced-capacity version is explicitly a +sensitivity case derived from behavioral evidence; the without-reduction +version reflects the assumption that structural capacity remained fully +installed. + +Silhouette plots: + +```bash +python3 plot_capacity_silhouette.py \ + resource_capacity_with_reduced_capacity.csv \ + --output inferred_capacity_silhouette_labeled_jst.png \ + --title 'Inferred normal-queue capacity (reductions and queue pauses)' \ + --cases-per-row 21 --timezone JST + +python3 plot_capacity_silhouette.py \ + resource_capacity_without_reduced_capacity.csv \ + --output queue_pause_capacity_silhouette_labeled_jst.png \ + --title 'Queue-pause-only normal-queue capacity' \ + --cases-per-row 21 --timezone JST +``` + +These plots collapse each constant-capacity interval to one equal-width, +contiguous rectangle. Changes are vertical, never interpolated; every case is +labeled by its start date in the selected timezone (JST in this example). The +second CSV is used here only as the queue-pause-only trace: full physical +capacity outside pauses and zero normal-queue capacity during pauses. + +## Reference EASY backfill replay + +For stronger evidence than queue pressure alone, first run the observed-schedule +replay derived from the Python reference EASY scheduler. This example uses the +included Fugaku files and settings: + +```bash +python3 backfill_opportunity_audit.py '*_scheduling_trace.csv' \ + --timeline capacity_timeline.csv \ + --nodes 158976 --timezone JST \ + --grace-minutes 60 --release-delay-minutes 10 \ + --output backfill_opportunities.csv + +python3 discover_maintenance.py '*_scheduling_trace.csv' \ + --overlap-policy filename-month --timezone JST \ + --known-capacity 158976 \ + --backfill-evidence backfill_opportunities.csv \ + --all-states \ + --output maintenance_report.json \ + --bins-output capacity_timeline.csv +``` + +### Conservative controls against scheduler-policy false positives + +The replay does not treat every reference-EASY opportunity that starts late as +evidence of reduced resources. At each hourly snapshot it applies these +controls: + +1. A candidate that actually starts within `--grace-minutes` is successful, + not missed. The default grace window is 60 minutes. +2. A candidate is also discounted when another job that was already pending at + the same snapshot, requested at least as many nodes, and starts within the + grace window. This equal-or-larger successful job is evidence that the node + count was dispatchable and that policy or unobserved eligibility may explain + the candidate's delay. The audit exports `size_controlled_*` counts. +3. Nodes remain unavailable for `--release-delay-minutes` after recorded job + end. The default is a conservative 10-minute reclaim interval. This delay is + included both in current free-node accounting and in EASY reservation safety + tests. The audit exports `reclaiming_nodes` and `unavailable_nodes`. +4. Stable source-qualified job identifiers are exported for opportunity and + miss sets. Merged-period miss fractions use distinct jobs, so one job seen + waiting in several hourly snapshots is counted only once for the period. + +The size control is deliberately one-sided: the trace has no queue, partition, +user, project, application, dependency, reservation, or node-topology fields. +Consequently it can disqualify weak evidence but cannot prove that two jobs had +identical scheduler eligibility. A detected period remains behavioral evidence, +not proof that physical nodes were removed. + +### Workload-family diversity + +A workload family is the fingerprint +`(num_nodes, time_limit_seconds, rounded average power per node)`. Average power +per node is rounded to the nearest 5 units. It is not an identity for a user, +project, queue, or executable. Requiring two families prevents several +identical-shaped jobs from satisfying the evidence gate by themselves. + +Families with at least three regularly spaced submissions are discounted when +their inter-arrival-gap coefficient of variation is at most 0.20 and the mean +gap is at least 30 minutes. A family is also discounted as habitually held when +at least three distinct jobs miss opportunities and at least half of that +family's jobs do so. These filters reduce recurring-policy artifacts; they do +not recover scheduler fields absent from the input. + +At each candidate hour, the audit reconstructs the jobs actually running and +waiting. It applies the reference scheduler's EASY rules without changing the +observed schedule: + +1. Start consecutive FCFS head jobs while they fit. +2. Reserve enough nodes for the first blocked head using running jobs' requested + time limits. +3. Find later jobs whose node request fits currently free nodes and whose time + limit ends before that reservation. +4. Count an opportunity as missed only when the observed job does not start + within the one-hour grace period. + +Repeated holds are discounted in two ways. A workload family is fingerprinted +from node count, time limit, and a coarse per-node power bucket. Regularly +submitted families are treated as recurring. A repeated family is also treated +as habitually held when at least three distinct instances miss opportunities and +at least half of that family's instances do so. Neither family type contributes +to the non-recurring evidence count. + +When `--backfill-evidence` is supplied, a reduced-capacity bin must have at least +three missed non-recurring EASY backfills from at least two distinct workload +families, and strictly more than half of its adjusted eligible backfills must be +missed. The adjusted denominator is all replayed backfill opportunities minus +misses attributed to recurring or habitually held families. Successful starts +remain in the denominator. Full-shutdown and queue-pause bins use the +corresponding combined direct-start and backfill evidence. These defaults are +configurable with `--min-missed-jobs`, `--min-missed-families`, and +`--min-backfill-miss-fraction`. + +With replay evidence enabled, any hourly bin meeting that multi-job, +multi-family, strict-majority backfill rule is treated as a maintenance +candidate. A candidate period must last at least one hour by default. Its state +describes the observed effect: shutdown, queue pause, reduced capacity, or +backfill suppression without a large allocation reduction. + +For every resulting period, the report also gives potential effective-capacity +bounds. For each qualifying evidence hour, it calculates +`instantaneous_running_nodes + smallest_missed_backfill_job_nodes - 1`. The +period lower bound is the maximum of those values and the maximum observed +occupancy. The trace does not establish a meaningful effective-capacity upper +bound, so none is reported; a user-supplied physical capacity is shown only as +context. The lower bound is conditional on the replay assumptions and is not a +physical node-health measurement. + +This remains a conservative behavioral inference. The trace has no explicit +hold, queue eligibility, dependency, reservation, or node-topology fields, so +the replay cannot prove that a job was operationally eligible to run. diff --git a/scripts/detect_queue_pause/backfill_opportunity_audit.py b/scripts/detect_queue_pause/backfill_opportunity_audit.py new file mode 100755 index 0000000..daed686 --- /dev/null +++ b/scripts/detect_queue_pause/backfill_opportunity_audit.py @@ -0,0 +1,533 @@ +#!/usr/bin/env python3 +"""Replay observed traces and find conservative unused EASY opportunities. + +This adapts the reservation and backfill tests from the DR_EVT Python reference +scheduler. It does not reschedule the trace. At each sampled timestamp it +reconstructs the jobs actually running and waiting, then records jobs that the +reference EASY decision would have started but that remained waiting beyond a +configurable grace period. +""" + +import argparse +import bisect +import csv +import heapq +import math +import os +import sys +from collections import Counter + +from discover_maintenance import TimestampFormatter, expand_paths, filename_month_window + + +class Job: + __slots__ = ( + "idx", "submit", "begin", "end", "nodes", "limit", "family", + ) + + def __init__(self, idx, submit, begin, end, nodes, limit_seconds, family): + self.idx = idx + self.submit = submit + self.begin = begin + self.end = end + self.nodes = nodes + self.limit = limit_seconds + self.family = family + + +class FenwickTree: + """Dynamic prefix sums over compressed pessimistic release times.""" + + def __init__(self, size): + self.values = [0] * (size + 1) + + def add(self, index, amount): + index += 1 + while index < len(self.values): + self.values[index] += amount + index += index & -index + + def prefix(self, end): + total = 0 + while end: + total += self.values[end] + end -= end & -end + return total + + def lower_bound(self, target): + """Return the first zero-based index whose prefix reaches target.""" + if target <= 0: + return 0 + index = 0 + step = 1 << (len(self.values).bit_length() - 1) + while step: + candidate = index + step + if candidate < len(self.values) and self.values[candidate] < target: + index = candidate + target -= self.values[candidate] + step >>= 1 + return index if index < len(self.values) - 1 else None + + +def family_key(row, names, nodes, limit_seconds): + """Build a conservative workload-family fingerprint. + + Power per node helps avoid treating all jobs with the same common resource + shape as one recurring workload. It is used only as a retrospective + fingerprint, never for backfill feasibility. + """ + power_bucket = None + power_index = names.get("avgpcon") + if power_index is not None and power_index < len(row): + try: + power_per_node = float(row[power_index]) / nodes + power_bucket = int(round(power_per_node / 5.0) * 5) + except (ValueError, ZeroDivisionError): + pass + return nodes, limit_seconds, power_bucket + + +def recurring_families(jobs, minimum_occurrences, maximum_gap_cv): + """Find regularly submitted families using inter-arrival consistency.""" + submissions = {} + for job in jobs: + submissions.setdefault(job.family, []).append(job.submit) + recurring = set() + for family, times in submissions.items(): + if len(times) < minimum_occurrences: + continue + times.sort() + gaps = [right - left for left, right in zip(times, times[1:]) if right > left] + if len(gaps) < minimum_occurrences - 1: + continue + mean = sum(gaps) / float(len(gaps)) + if mean < 1800: + continue + variance = sum((gap - mean) ** 2 for gap in gaps) / len(gaps) + coefficient = math.sqrt(variance) / mean + if coefficient <= maximum_gap_cv: + recurring.add(family) + return recurring + + +def load_jobs(path, window, release_delay_seconds=0): + jobs = [] + with open(path, "r", newline="", encoding="utf-8-sig") as stream: + reader = csv.reader(stream) + header = next(reader) + names = {name.strip(): index for index, name in enumerate(header)} + required = ("job_submit_time", "begin_time", "end_time", "num_nodes", "time_limit") + missing = [name for name in required if name not in names] + if missing: + raise ValueError("{}: missing columns: {}".format(path, ", ".join(missing))) + for index, row in enumerate(reader): + try: + submit = int(float(row[names["job_submit_time"]])) + begin = int(float(row[names["begin_time"]])) + end = int(float(row[names["end_time"]])) + nodes = int(float(row[names["num_nodes"]])) + limit_seconds = int(float(row[names["time_limit"]])) + except (ValueError, IndexError, OverflowError): + continue + if nodes <= 0 or limit_seconds <= 0 or begin < submit or end < begin: + continue + # Keep jobs that are pending or running at some point in this file's + # ownership window. Completely external spillover is duplicated + # coverage and must not affect the replay. + if submit >= window[1] or end + release_delay_seconds <= window[0]: + continue + jobs.append( + Job( + index, submit, begin, end, nodes, limit_seconds, + family_key(row, names, nodes, limit_seconds), + ) + ) + jobs.sort(key=lambda job: (job.submit, job.idx)) + return jobs + + +def reservation_time( + now, needed_nodes, free_nodes, release_tree, release_times, virtual_jobs, + release_delay_seconds=0, +): + required_release = needed_nodes - free_nodes + if required_release <= 0: + return now + virtual_releases = sorted( + (now + job.limit + release_delay_seconds, job.nodes) + for job in virtual_jobs + ) + virtual_freed = 0 + for release, nodes in virtual_releases: + active_needed = required_release - virtual_freed + active_index = release_tree.lower_bound(active_needed) + if active_index is not None and release_times[active_index] <= release: + return release_times[active_index] + virtual_freed += nodes + active_through_release = release_tree.prefix( + bisect.bisect_right(release_times, release) + ) + if active_through_release + virtual_freed >= required_release: + return release + active_index = release_tree.lower_bound(required_release - virtual_freed) + return None if active_index is None else release_times[active_index] + + +def has_successful_size_control(job, successful_controls): + """Return whether an equal/larger pending job starts within the grace window. + + Such a start demonstrates that at least the candidate's node count could be + dispatched contemporaneously. It does not prove equivalent queue or + project eligibility because those fields are absent from the trace. + """ + return any(control.nodes >= job.nodes for control in successful_controls) + + +def note_miss( + job, now, grace_seconds, metrics, missed_records, kind, successful_controls, +): + if job.begin <= now + grace_seconds: + return + if has_successful_size_control(job, successful_controls): + metrics["size_controlled_{}_jobs".format(kind)] += 1 + metrics["size_controlled_opportunity_jobs"] += 1 + return + metrics["missed_{}_jobs".format(kind)] += 1 + metrics["missed_opportunity_jobs"] += 1 + missed_records.append((job, kind)) + + +def replay_file( + path, window, target_times, capacity, grace_seconds, + recurrence_minimum, recurrence_cv, habitual_fraction, scan_limit, + release_delay_seconds=0, +): + jobs = load_jobs(path, window, release_delay_seconds) + recurring = recurring_families(jobs, recurrence_minimum, recurrence_cv) + start_order = sorted(range(len(jobs)), key=lambda index: (jobs[index].begin, index)) + release_times = sorted(set( + max(job.begin + job.limit, job.end) + release_delay_seconds for job in jobs + )) + release_indices = {value: index for index, value in enumerate(release_times)} + release_tree = FenwickTree(len(release_times)) + unavailable = {} + unavailable_end_heap = [] + running = {} + running_end_heap = [] + start_cursor = 0 + submit_cursor = 0 + head_cursor = 0 + used_nodes = 0 + results = {} + missed_by_time = {} + opportunities_by_time = {} + source_id = os.path.basename(path).replace("_scheduling_trace.csv", "") + ordered_begin_times = [jobs[index].begin for index in start_order] + + for now in target_times: + while start_cursor < len(start_order) and jobs[start_order[start_cursor]].begin <= now: + job = jobs[start_order[start_cursor]] + if job.end > now: + running[job.idx] = job + heapq.heappush(running_end_heap, (job.end, job.idx)) + if job.end + release_delay_seconds > now: + unavailable[job.idx] = job + heapq.heappush( + unavailable_end_heap, + (job.end + release_delay_seconds, job.idx), + ) + release_tree.add( + release_indices[ + max(job.begin + job.limit, job.end) + release_delay_seconds + ], + job.nodes, + ) + used_nodes += job.nodes + start_cursor += 1 + while running_end_heap and running_end_heap[0][0] <= now: + _, index = heapq.heappop(running_end_heap) + running.pop(index, None) + while unavailable_end_heap and unavailable_end_heap[0][0] <= now: + _, index = heapq.heappop(unavailable_end_heap) + job = unavailable.pop(index, None) + if job is not None: + release_tree.add( + release_indices[ + max(job.begin + job.limit, job.end) + release_delay_seconds + ], + -job.nodes, + ) + used_nodes -= job.nodes + while submit_cursor < len(jobs) and jobs[submit_cursor].submit <= now: + submit_cursor += 1 + while head_cursor < submit_cursor and jobs[head_cursor].begin <= now: + head_cursor += 1 + + free_nodes = max(0, capacity - used_nodes) + metrics = Counter() + metrics["grace_seconds"] = grace_seconds + metrics["release_delay_seconds"] = release_delay_seconds + metrics["successful_size_control_enabled"] = 1 + observed_running_nodes = sum(job.nodes for job in running.values()) + metrics["observed_running_nodes"] = observed_running_nodes + metrics["reclaiming_nodes"] = used_nodes - observed_running_nodes + metrics["unavailable_nodes"] = used_nodes + metrics["observed_free_nodes"] = free_nodes + missed_records = [] + opportunity_records = [] + virtual_jobs = [] + cursor = head_cursor + examined = 0 + control_begin = bisect.bisect_right(ordered_begin_times, now) + control_end = bisect.bisect_right( + ordered_begin_times, now + grace_seconds + ) + successful_controls = [ + jobs[start_order[index]] + for index in range(control_begin, control_end) + if jobs[start_order[index]].submit <= now + ] + + # Reference EASY step 1: start consecutive FCFS heads while they fit. + blocked_cursor = None + while cursor < submit_cursor: + job = jobs[cursor] + cursor += 1 + if job.begin <= now: + continue + examined += 1 + if job.nodes <= free_nodes: + metrics["direct_opportunity_jobs"] += 1 + opportunity_records.append((job, "direct")) + note_miss( + job, now, grace_seconds, metrics, missed_records, "direct", + successful_controls, + ) + free_nodes -= job.nodes + virtual_jobs.append(job) + else: + blocked_cursor = cursor - 1 + break + + # Reference EASY steps 2/3: reserve for the blocked head, then scan + # later jobs for safe backfill candidates. + reservation = None + if blocked_cursor is not None: + head = jobs[blocked_cursor] + reservation = reservation_time( + now, head.nodes, free_nodes, release_tree, release_times, + virtual_jobs, release_delay_seconds, + ) + cursor = blocked_cursor + 1 + while cursor < submit_cursor and examined < scan_limit: + job = jobs[cursor] + cursor += 1 + if job.begin <= now: + continue + examined += 1 + if ( + reservation is not None + and job.nodes <= free_nodes + and now + job.limit + release_delay_seconds < reservation + ): + metrics["backfill_opportunity_jobs"] += 1 + opportunity_records.append((job, "backfill")) + note_miss( + job, now, grace_seconds, metrics, missed_records, "backfill", + successful_controls, + ) + free_nodes -= job.nodes + + metrics["queue_jobs_examined"] = examined + metrics["queue_scan_truncated"] = int(examined >= scan_limit and cursor < submit_cursor) + metrics["reservation_time"] = "" if reservation is None else reservation + results[now] = metrics + missed_by_time[now] = missed_records + opportunities_by_time[now] = opportunity_records + + # A repeated family is treated as policy/hold-like only when distinct jobs + # from that family habitually show the same missed-opportunity behavior. + # This is deliberately learned after replay so mere use of a common job + # shape does not by itself suppress evidence. + family_totals = Counter(job.family for job in jobs) + family_missed_jobs = {} + for records in missed_by_time.values(): + for job, _ in records: + family_missed_jobs.setdefault(job.family, set()).add(job.idx) + habitual = set() + for family, indices in family_missed_jobs.items(): + total = family_totals[family] + if ( + total >= recurrence_minimum + and len(indices) >= recurrence_minimum + and len(indices) / float(total) >= habitual_fraction + ): + habitual.add(family) + excluded_families = recurring | habitual + + for now, records in missed_by_time.items(): + metrics = results[now] + families = set() + families_by_kind = {"direct": set(), "backfill": set()} + nonrecurring_nodes = [] + nonrecurring_backfill_nodes = [] + opportunity_ids = {"backfill": set()} + missed_ids = {"backfill": set()} + nonrecurring_missed_ids = {"backfill": set()} + for job, kind in opportunities_by_time[now]: + if kind == "backfill": + opportunity_ids[kind].add("{}:{}".format(source_id, job.idx)) + for job, kind in records: + identifier = "{}:{}".format(source_id, job.idx) + if kind == "backfill": + missed_ids[kind].add(identifier) + if job.family in excluded_families: + continue + if kind == "backfill": + nonrecurring_missed_ids[kind].add(identifier) + metrics["missed_nonrecurring_{}_jobs".format(kind)] += 1 + metrics["missed_nonrecurring_jobs"] += 1 + families.add(job.family) + families_by_kind[kind].add(job.family) + nonrecurring_nodes.append(job.nodes) + if kind == "backfill": + nonrecurring_backfill_nodes.append(job.nodes) + metrics["missed_nonrecurring_families"] = len(families) + metrics["missed_nonrecurring_direct_families"] = len( + families_by_kind["direct"] + ) + metrics["missed_nonrecurring_backfill_families"] = len( + families_by_kind["backfill"] + ) + metrics["min_missed_nonrecurring_nodes"] = ( + min(nonrecurring_nodes) if nonrecurring_nodes else 0 + ) + metrics["min_missed_nonrecurring_backfill_nodes"] = ( + min(nonrecurring_backfill_nodes) if nonrecurring_backfill_nodes else 0 + ) + metrics["recurring_family_count"] = len(recurring) + metrics["habitually_missed_family_count"] = len(habitual) + metrics["backfill_opportunity_job_ids"] = ";".join( + sorted(opportunity_ids["backfill"]) + ) + metrics["missed_backfill_job_ids"] = ";".join( + sorted(missed_ids["backfill"]) + ) + metrics["missed_nonrecurring_backfill_job_ids"] = ";".join( + sorted(nonrecurring_missed_ids["backfill"]) + ) + return results + + +def read_target_bins(path, capacity, threshold): + targets = [] + with open(path, newline="", encoding="utf-8") as stream: + for row in csv.DictReader(stream): + running = float(row["running_nodes"]) + pending_jobs = float(row["pending_jobs"]) + pending_nodes = float(row["pending_nodes"]) + if ( + running < capacity * threshold + and (pending_jobs >= 10 or pending_nodes >= capacity * 0.02) + ): + targets.append(int(row["start"])) + return targets + + +def parse_args(): + parser = argparse.ArgumentParser( + description="Replay observed scheduling state and audit unused EASY opportunities" + ) + parser.add_argument("inputs", nargs="+", help="YY_MM scheduling trace paths/globs") + parser.add_argument("--timeline", default="capacity_timeline.csv") + parser.add_argument("--output", default="backfill_opportunities.csv") + parser.add_argument("--nodes", type=int, required=True) + parser.add_argument("--timezone", default="UTC") + parser.add_argument("--candidate-fraction", type=float, default=0.85) + parser.add_argument("--grace-minutes", type=int, default=60) + parser.add_argument( + "--release-delay-minutes", type=int, default=10, + help="post-end node reclaim delay used by the replay (default: 10)", + ) + parser.add_argument("--recurring-minimum", type=int, default=3) + parser.add_argument("--recurring-gap-cv", type=float, default=0.20) + parser.add_argument( + "--habitual-miss-fraction", type=float, default=0.50, + help="fraction of a repeated family's jobs required to discount it (default: 0.50)", + ) + parser.add_argument( + "--scan-limit", type=int, default=10000, + help="maximum queued jobs examined per sampled hour (default: 10000)", + ) + return parser.parse_args() + + +def main(): + args = parse_args() + if args.nodes <= 0 or args.scan_limit <= 0 or args.recurring_minimum < 2: + raise SystemExit("--nodes and --scan-limit must be positive") + if args.grace_minutes < 0 or args.release_delay_minutes < 0: + raise SystemExit("--grace-minutes and --release-delay-minutes cannot be negative") + if not 0 <= args.habitual_miss_fraction <= 1 or args.recurring_gap_cv < 0: + raise SystemExit("recurrence fractions must be between zero and one") + timestamps = TimestampFormatter("epoch", args.timezone) + paths = expand_paths(args.inputs) + targets = read_target_bins(args.timeline, args.nodes, args.candidate_fraction) + target_set = set(targets) + all_results = {} + owned_targets = set() + + for path in paths: + window = filename_month_window(path, timestamps) + file_targets = sorted( + value for value in target_set if window[0] <= value < window[1] + ) + owned_targets.update(file_targets) + if not file_targets: + continue + print( + "Auditing {} ({} candidate bins)".format(os.path.basename(path), len(file_targets)), + file=sys.stderr, + ) + all_results.update( + replay_file( + path, window, file_targets, args.nodes, args.grace_minutes * 60, + args.recurring_minimum, args.recurring_gap_cv, + args.habitual_miss_fraction, args.scan_limit, + args.release_delay_minutes * 60, + ) + ) + + fields = [ + "start", "observed_running_nodes", "observed_free_nodes", + "direct_opportunity_jobs", "backfill_opportunity_jobs", + "missed_direct_jobs", "missed_backfill_jobs", "missed_opportunity_jobs", + "size_controlled_direct_jobs", "size_controlled_backfill_jobs", + "size_controlled_opportunity_jobs", + "missed_nonrecurring_direct_jobs", "missed_nonrecurring_backfill_jobs", + "missed_nonrecurring_jobs", "missed_nonrecurring_families", + "missed_nonrecurring_direct_families", + "missed_nonrecurring_backfill_families", + "min_missed_nonrecurring_nodes", + "min_missed_nonrecurring_backfill_nodes", + "recurring_family_count", "habitually_missed_family_count", + "reclaiming_nodes", "unavailable_nodes", + "grace_seconds", "release_delay_seconds", + "successful_size_control_enabled", + "queue_jobs_examined", "queue_scan_truncated", + "reservation_time", + "backfill_opportunity_job_ids", "missed_backfill_job_ids", + "missed_nonrecurring_backfill_job_ids", + ] + with open(args.output, "w", newline="", encoding="utf-8") as stream: + writer = csv.DictWriter(stream, fieldnames=fields) + writer.writeheader() + for start in sorted(owned_targets): + values = dict(all_results.get(start, {})) + values["start"] = start + writer.writerow({field: values.get(field, 0) for field in fields}) + print("Wrote {} audited bins to {}".format(len(owned_targets), args.output), file=sys.stderr) + + +if __name__ == "__main__": + main() diff --git a/scripts/detect_queue_pause/build_resource_capacity_trace.py b/scripts/detect_queue_pause/build_resource_capacity_trace.py new file mode 100755 index 0000000..4283ee7 --- /dev/null +++ b/scripts/detect_queue_pause/build_resource_capacity_trace.py @@ -0,0 +1,165 @@ +#!/usr/bin/env python3 +"""Build hourly normal-queue capacity scenarios from maintenance evidence. + +Structural capacity is held constant. Queue-pause/shutdown periods expose zero +capacity to the normal queue. The optional sensitivity scenario also applies +the report's evidence-derived value during reduced-capacity periods. +""" + +import argparse +import csv +import json +import math + +from discover_maintenance import TimestampFormatter + + +def build_rows(timeline, periods, capacity, include_reduced, timezone_name): + timestamps = TimestampFormatter("iso", timezone_name) + by_hour = {} + for period in periods: + start = int(period["start"]) + end = int(period["end"]) + for hour in range(start, end, 3600): + by_hour[hour] = period + + rows = [] + for item in timeline: + start = int(item["start"]) + end = int(item["end"]) + period = by_hour.get(start) + state = "full_capacity" if period is None else period["state"] + available = capacity + assumption = "full_structural_capacity" + if state in ("queue_pause_or_maintenance", "full_shutdown"): + available = 0 + assumption = "normal_queue_paused" + elif state == "reduced_capacity" and include_reduced: + available = min( + capacity, + max(0, int(math.ceil( + float(period["potential_effective_capacity_lower_nodes"]) + ))), + ) + assumption = "inferred_reduced_capacity_sensitivity" + elif state == "reduced_capacity": + assumption = "reduced_capacity_ignored_full_assumed" + elif state == "backfill_suppression": + assumption = "no_capacity_reduction_full_assumed" + + rows.append( + { + "start": start, + "end": end, + "start_local": timestamps.value(start), + "end_local": timestamps.value(end), + "timezone": timezone_name, + "structural_capacity_nodes": capacity, + "normal_queue_capacity_nodes": available, + "state": state, + "assumption": assumption, + } + ) + return rows + + +def simulator_rows(rows, capacity): + """Convert hourly analysis rows to DR_EVT capacity change points.""" + if not rows: + raise ValueError("timeline contains no capacity rows") + + changes = [] + previous = None + for row in rows: + available = int(row["normal_queue_capacity_nodes"]) + if available != previous: + changes.append({"time": int(row["start"]), "total_nodes": available}) + previous = available + + # A simulator change point remains effective indefinitely. Restore normal + # capacity after a reduced final interval instead of accidentally extending + # that inferred state beyond the analyzed timeline. + final_end = int(rows[-1]["end"]) + if previous != capacity: + changes.append({"time": final_end, "total_nodes": capacity}) + return changes + + +def write_rows(path, rows, simulator_format=False, capacity=None): + if simulator_format: + fields = ["time", "total_nodes"] + rows = simulator_rows(rows, capacity) + else: + fields = [ + "start", "end", "start_local", "end_local", "timezone", + "structural_capacity_nodes", "normal_queue_capacity_nodes", + "state", "assumption", + ] + with open(path, "w", newline="", encoding="utf-8") as stream: + writer = csv.DictWriter(stream, fieldnames=fields, lineterminator="\n") + writer.writeheader() + writer.writerows(rows) + + +def main(): + parser = argparse.ArgumentParser( + description="Build normal-queue capacity traces from a detector report" + ) + parser.add_argument("--report", default="maintenance_report.json") + parser.add_argument("--timeline", default="capacity_timeline.csv") + parser.add_argument( + "--capacity", type=int, + help="structural node capacity (default: report normal capacity)", + ) + parser.add_argument("--timezone", default="UTC") + parser.add_argument( + "--with-reduced-output", + default="resource_capacity_with_reduced_capacity.csv", + ) + parser.add_argument( + "--without-reduced-output", + default="resource_capacity_without_reduced_capacity.csv", + ) + parser.add_argument( + "--simulator-format", action="store_true", + help=( + "write compact DR_EVT --capacity_schedule files with columns " + "time,total_nodes instead of the detailed hourly analysis format" + ), + ) + args = parser.parse_args() + if args.capacity is not None and args.capacity <= 0: + parser.error("--capacity must be positive") + + with open(args.report, encoding="utf-8") as stream: + report = json.load(stream) + with open(args.timeline, newline="", encoding="utf-8") as stream: + timeline = list(csv.DictReader(stream)) + + capacity = args.capacity + if capacity is None: + capacity = int(round(float(report["normal_capacity"]["inferred_nodes"]))) + if capacity <= 0: + raise ValueError("report normal capacity must be positive") + + with_reduced = build_rows( + timeline, report["periods"], capacity, True, args.timezone + ) + without_reduced = build_rows( + timeline, report["periods"], capacity, False, args.timezone + ) + write_rows( + args.with_reduced_output, with_reduced, args.simulator_format, capacity + ) + write_rows( + args.without_reduced_output, + without_reduced, + args.simulator_format, + capacity, + ) + print(args.with_reduced_output) + print(args.without_reduced_output) + + +if __name__ == "__main__": + main() diff --git a/scripts/detect_queue_pause/discover_maintenance.py b/scripts/detect_queue_pause/discover_maintenance.py new file mode 100755 index 0000000..c99b395 --- /dev/null +++ b/scripts/detect_queue_pause/discover_maintenance.py @@ -0,0 +1,1042 @@ +#!/usr/bin/env python3 +"""Infer maintenance, queue pauses, and normal capacity from job traces. + +The input must contain at least these columns: + job_submit_time, begin_time, end_time, num_nodes + +Times are Unix seconds. Only the Python standard library is required. +""" + +import argparse +import csv +import glob +import json +import math +import os +import re +import sys +from collections import defaultdict +from datetime import datetime, timedelta, timezone +from pathlib import Path +from typing import Dict, Iterable, List, Optional, Sequence, Tuple + + +REQUIRED_COLUMNS = ("job_submit_time", "begin_time", "end_time", "num_nodes") +FIXED_TIMEZONES = { + "UTC": 0, + "GMT": 0, + "PST": -8, + "PDT": -7, + "MST": -7, + "MDT": -6, + "CST": -6, + "CDT": -5, + "EST": -5, + "EDT": -4, + "JST": 9, + "KST": 9, +} + + +def resolve_timezone(name): + """Resolve fixed abbreviations and, when available, IANA zone names.""" + normalized = name.upper() + if normalized in FIXED_TIMEZONES: + return timezone.utc if normalized in ("UTC", "GMT") else timezone( + timedelta(hours=FIXED_TIMEZONES[normalized]), normalized + ) + try: + from zoneinfo import ZoneInfo + return ZoneInfo(name) + except (ImportError, KeyError): + pass + try: + import pytz + except ImportError: + pytz = None + if pytz is not None: + try: + return pytz.timezone(name) + except pytz.UnknownTimeZoneError: + pass + raise ValueError( + "unknown timezone {!r}; use UTC, PST, PDT, JST, a numeric offset " + "such as +09:00, or an installed IANA name such as Asia/Tokyo".format(name) + ) + + +def parse_numeric_timezone(name): + if len(name) != 6 or name[0] not in "+-" or name[3] != ":": + return None + try: + hours = int(name[1:3]) + minutes = int(name[4:6]) + except ValueError: + return None + if hours > 23 or minutes > 59: + return None + offset = (hours * 60 + minutes) * (1 if name[0] == "+" else -1) + return timezone(timedelta(minutes=offset), name) + + +class TimestampFormatter: + def __init__(self, output_format, timezone_name): + self.output_format = output_format + numeric = parse_numeric_timezone(timezone_name) + self.timezone = numeric if numeric is not None else resolve_timezone(timezone_name) + self.timezone_name = timezone_name + + def value(self, epoch): + if self.output_format == "epoch": + return int(epoch) + value = datetime.fromtimestamp(epoch, self.timezone).isoformat() + return value.replace("+00:00", "Z") + + def month(self, epoch): + return datetime.fromtimestamp(epoch, self.timezone).strftime("%Y-%m") + + def month_bounds(self, year, month): + next_year, next_month = (year + 1, 1) if month == 12 else (year, month + 1) + start_naive = datetime(year, month, 1) + end_naive = datetime(next_year, next_month, 1) + if hasattr(self.timezone, "localize"): + start = self.timezone.localize(start_naive) + end = self.timezone.localize(end_naive) + else: + start = start_naive.replace(tzinfo=self.timezone) + end = end_naive.replace(tzinfo=self.timezone) + return int(start.timestamp()), int(end.timestamp()) + + +class TraceStats: + def __init__( + self, + rows=0, + invalid_rows=0, + invalid_intervals=0, + min_submit=None, + max_submit=None, + ): + self.rows = rows + self.invalid_rows = invalid_rows + self.invalid_intervals = invalid_intervals + self.min_submit = min_submit + self.max_submit = max_submit + self.analysis_start = min_submit + self.analysis_end = None if max_submit is None else max_submit + 1 + + +class TimelineBuilder: + """Accumulate exact node/job seconds without expanding every interval.""" + + def __init__(self, bin_seconds: int) -> None: + self.bin_seconds = bin_seconds + self.run_edge = defaultdict(float) # type: Dict[int, float] + self.run_full_delta = defaultdict(float) # type: Dict[int, float] + self.run_boundary_delta = defaultdict(float) # type: Dict[int, float] + self.pending_job_edge = defaultdict(float) # type: Dict[int, float] + self.pending_job_full_delta = defaultdict(float) # type: Dict[int, float] + self.pending_node_edge = defaultdict(float) # type: Dict[int, float] + self.pending_node_full_delta = defaultdict(float) # type: Dict[int, float] + self.starts = defaultdict(int) # type: Dict[int, int] + self.start_nodes = defaultdict(int) # type: Dict[int, int] + self.submits = defaultdict(int) # type: Dict[int, int] + + def _add_interval( + self, + start: int, + end: int, + weight: float, + edges: Dict[int, float], + full_delta: Dict[int, float], + ) -> None: + if end <= start: + return + width = self.bin_seconds + first = start // width + last = (end - 1) // width + if first == last: + edges[first] += (end - start) * weight + return + edges[first] += ((first + 1) * width - start) * weight + edges[last] += (end - last * width) * weight + if last > first + 1: + full_delta[first + 1] += weight + full_delta[last] -= weight + + def add_job( + self, submit: int, begin: int, end: int, nodes: int, + window: Optional[Tuple[int, int]] = None, + ) -> None: + width = self.bin_seconds + window_start, window_end = window if window is not None else (-sys.maxsize, sys.maxsize) + if window_start <= submit < window_end: + self.submits[submit // width] += 1 + if window_start <= begin < window_end: + self.starts[begin // width] += 1 + self.start_nodes[begin // width] += nodes + run_start = max(begin, window_start) + run_end = min(end, window_end) + self._add_interval( + run_start, run_end, nodes, + self.run_edge, self.run_full_delta, + ) + # Exact occupancy at each bin boundary. Unlike running_nodes, this is + # an instantaneous sample and is not averaged over the bin. + if run_end > run_start: + first_boundary = (run_start + width - 1) // width + after_last_boundary = (run_end + width - 1) // width + if first_boundary < after_last_boundary: + self.run_boundary_delta[first_boundary] += nodes + self.run_boundary_delta[after_last_boundary] -= nodes + self._add_interval( + max(submit, window_start), + min(begin, window_end), + 1.0, + self.pending_job_edge, + self.pending_job_full_delta, + ) + self._add_interval( + max(submit, window_start), + min(begin, window_end), + nodes, + self.pending_node_edge, + self.pending_node_full_delta, + ) + + +def expand_paths(patterns: Sequence[str]) -> List[str]: + paths = [] # type: List[str] + for pattern in patterns: + matches = sorted(glob.glob(pattern)) + if matches: + paths.extend(matches) + elif os.path.isfile(pattern): + paths.append(pattern) + else: + raise FileNotFoundError(f"no input files match {pattern!r}") + # Preserve order while eliminating overlap between explicit paths and globs. + return list(dict.fromkeys(os.path.abspath(path) for path in paths)) + + +def filename_month_window(path, timestamps): + match = re.match(r"^(\d{2}|\d{4})_(\d{2})_scheduling_trace\.csv$", os.path.basename(path)) + if not match: + raise ValueError( + "{}: --overlap-policy filename-month requires a " + "YY_MM_scheduling_trace.csv filename".format(path) + ) + year = int(match.group(1)) + if year < 100: + year += 2000 + month = int(match.group(2)) + if not 1 <= month <= 12: + raise ValueError("{}: invalid month in filename".format(path)) + return timestamps.month_bounds(year, month) + + +def read_traces( + paths: Sequence[str], builder: TimelineBuilder, + overlap_policy="combine", timestamps=None, +) -> TraceStats: + stats = TraceStats() + for path in paths: + window = None + if overlap_policy == "filename-month": + window = filename_month_window(path, timestamps) + stats.analysis_start = ( + window[0] if stats.analysis_start is None + else min(stats.analysis_start, window[0]) + ) + stats.analysis_end = ( + window[1] if stats.analysis_end is None + else max(stats.analysis_end, window[1]) + ) + with open(path, "r", newline="", encoding="utf-8-sig") as stream: + reader = csv.reader(stream) + try: + header = next(reader) + except StopIteration: + continue + names = {name.strip(): idx for idx, name in enumerate(header)} + missing = [name for name in REQUIRED_COLUMNS if name not in names] + if missing: + raise ValueError(f"{path}: missing columns: {', '.join(missing)}") + indices = [names[name] for name in REQUIRED_COLUMNS] + max_index = max(indices) + for row in reader: + stats.rows += 1 + try: + if len(row) <= max_index: + raise ValueError + submit, begin, end, nodes = ( + int(float(row[index])) for index in indices + ) + except (ValueError, OverflowError): + stats.invalid_rows += 1 + continue + if nodes <= 0 or begin < submit or end < begin: + stats.invalid_intervals += 1 + continue + stats.min_submit = submit if stats.min_submit is None else min(stats.min_submit, submit) + stats.max_submit = submit if stats.max_submit is None else max(stats.max_submit, submit) + builder.add_job(submit, begin, end, nodes, window) + if not stats.rows or stats.min_submit is None or stats.max_submit is None: + raise ValueError("no valid job records found") + if overlap_policy == "combine": + stats.analysis_start = stats.min_submit + stats.analysis_end = stats.max_submit + 1 + return stats + + +def quantile(values: Iterable[float], q: float) -> float: + ordered = sorted(values) + if not ordered: + return 0.0 + position = (len(ordered) - 1) * q + lower = math.floor(position) + upper = math.ceil(position) + if lower == upper: + return ordered[lower] + return ordered[lower] + (ordered[upper] - ordered[lower]) * (position - lower) + + +def build_bins(builder: TimelineBuilder, stats: TraceStats) -> List[Dict[str, object]]: + width = builder.bin_seconds + first = stats.analysis_start // width + # In combine mode, stop at the last submission so the drained tail of a + # truncated trace does not look like a shutdown. Filename-month mode instead + # uses the nonoverlapping ownership windows established from the filenames. + last = (stats.analysis_end - 1) // width + run_full = pending_jobs_full = pending_nodes_full = 0.0 + instantaneous_running = 0.0 + bins = [] # type: List[Dict[str, object]] + for index in range(first, last + 1): + run_full += builder.run_full_delta.get(index, 0.0) + pending_jobs_full += builder.pending_job_full_delta.get(index, 0.0) + pending_nodes_full += builder.pending_node_full_delta.get(index, 0.0) + instantaneous_running += builder.run_boundary_delta.get(index, 0.0) + bins.append( + { + "index": index, + "start_epoch": index * width, + "running_nodes": run_full + builder.run_edge.get(index, 0.0) / width, + "instantaneous_running_nodes": instantaneous_running, + "pending_jobs": pending_jobs_full + + builder.pending_job_edge.get(index, 0.0) / width, + "pending_nodes": pending_nodes_full + + builder.pending_node_edge.get(index, 0.0) / width, + "jobs_started": builder.starts.get(index, 0), + "nodes_started": builder.start_nodes.get(index, 0), + "jobs_submitted": builder.submits.get(index, 0), + } + ) + return bins + + +def infer_capacity( + bins: Sequence[Dict[str, object]], q: float, known_capacity: Optional[int] +) -> Tuple[float, str]: + if known_capacity is not None: + return float(known_capacity), "user supplied" + running = [float(item["running_nodes"]) for item in bins if item["running_nodes"]] + capacity = quantile(running, q) + if capacity <= 0: + raise ValueError("could not infer capacity: no running jobs in analysis interval") + return capacity, f"{q:g} quantile of observed hourly-equivalent node allocation" + + +def capacity_windows( + bins: Sequence[Dict[str, object]], q: float, global_capacity: float, + timestamps: TimestampFormatter, +) -> List[Dict[str, object]]: + grouped = defaultdict(list) # type: Dict[str, List[Dict[str, object]]] + for item in bins: + grouped[timestamps.month(int(item["start_epoch"]))].append(item) + result = [] # type: List[Dict[str, object]] + for month, items in sorted(grouped.items()): + pressure_floor = max(1.0, global_capacity * 0.02) + pressured = [ + item + for item in items + if float(item["pending_nodes"]) >= pressure_floor + or float(item["pending_jobs"]) >= 10 + ] + sample = pressured if len(pressured) >= 24 else items + estimate = quantile((float(item["running_nodes"]) for item in sample), q) + partial_month = len(items) < 24 * 27 + if partial_month: + confidence = "partial-window lower bound" + elif len(pressured) >= 24: + confidence = "high" + else: + confidence = "low-demand lower bound" + result.append( + { + "month": month, + "observed_capacity_nodes": round(estimate, 1), + "fraction_of_global": round(estimate / global_capacity, 4), + "bin_count": len(items), + "pressured_bin_count": len(pressured), + "confidence": confidence, + } + ) + return result + + +def bridge_candidates(candidate: List[bool], gap_bins: int) -> List[bool]: + if gap_bins <= 0: + return candidate + bridged = candidate[:] + previous = None + for i, value in enumerate(candidate): + if not value: + continue + if previous is not None and i - previous - 1 <= gap_bins: + for j in range(previous + 1, i): + bridged[j] = True + previous = i + return bridged + + +def load_backfill_evidence(path, bins): + by_start = {int(item["start_epoch"]): item for item in bins} + fields = ( + "direct_opportunity_jobs", "backfill_opportunity_jobs", + "missed_opportunity_jobs", "missed_nonrecurring_jobs", + "missed_nonrecurring_families", "missed_backfill_jobs", + "missed_nonrecurring_backfill_jobs", "queue_scan_truncated", + "missed_nonrecurring_direct_jobs", + "missed_nonrecurring_direct_families", + "missed_nonrecurring_backfill_families", + "min_missed_nonrecurring_nodes", + "min_missed_nonrecurring_backfill_nodes", + "observed_running_nodes", "observed_free_nodes", + "reclaiming_nodes", "unavailable_nodes", + "size_controlled_direct_jobs", "size_controlled_backfill_jobs", + "size_controlled_opportunity_jobs", + "grace_seconds", "release_delay_seconds", + "successful_size_control_enabled", + ) + id_fields = ( + "backfill_opportunity_job_ids", "missed_backfill_job_ids", + "missed_nonrecurring_backfill_job_ids", + ) + matched = 0 + with open(path, newline="", encoding="utf-8") as stream: + reader = csv.DictReader(stream) + if "start" not in (reader.fieldnames or []): + raise ValueError("{}: backfill evidence has no start column".format(path)) + for row in reader: + try: + item = by_start.get(int(row["start"])) + except (TypeError, ValueError): + continue + if item is None: + continue + for field in fields: + try: + item[field] = int(row.get(field, 0) or 0) + except ValueError: + item[field] = 0 + for field in id_fields: + value = row.get(field) + item[field] = set(value.split(";")) if value else set() + item["has_distinct_backfill_job_ids"] = all( + field in (reader.fieldnames or []) for field in id_fields + ) + matched += 1 + return matched + + +def detect_periods( + bins: Sequence[Dict[str, object]], + capacity: float, + width: int, + min_duration_hours: float, + min_pending_jobs: float, + min_pending_fraction: float, + shutdown_fraction: float, + reduced_fraction: float, + pause_start_fraction: float, + bridge_bins_count: int, + timestamps: TimestampFormatter, + require_backfill_evidence=False, + min_missed_jobs=3, + min_missed_families=2, + min_backfill_miss_fraction=0.5, +) -> List[Dict[str, object]]: + demand_start_counts = [ + float(item["jobs_started"]) + for item in bins + if ( + float(item["pending_jobs"]) >= min_pending_jobs + or float(item["pending_nodes"]) >= capacity * min_pending_fraction + ) + and float(item["jobs_started"]) > 0 + and float(item["running_nodes"]) >= capacity * reduced_fraction + ] + typical_starts = quantile(demand_start_counts, 0.5) + if typical_starts == 0: + typical_starts = quantile( + (float(item["jobs_started"]) for item in bins if item["jobs_started"]), 0.5 + ) + + raw = [] # type: List[bool] + flags = [] # type: List[Tuple[bool, bool, bool]] + for item in bins: + running_fraction = float(item["running_nodes"]) / capacity + demand = ( + float(item["pending_jobs"]) >= min_pending_jobs + or float(item["pending_nodes"]) >= capacity * min_pending_fraction + ) + general_opportunity_evidence = ( + not require_backfill_evidence + or ( + int(item.get("missed_nonrecurring_jobs", 0)) >= min_missed_jobs + and int(item.get("missed_nonrecurring_families", 0)) >= min_missed_families + ) + ) + has_distinct_ids = bool(item.get("has_distinct_backfill_job_ids", False)) + if has_distinct_ids: + opportunity_ids = set(item.get("backfill_opportunity_job_ids", set())) + missed_ids = set(item.get("missed_backfill_job_ids", set())) + nonrecurring_missed_ids = set( + item.get("missed_nonrecurring_backfill_job_ids", set()) + ) + total_backfill_opportunities = len(opportunity_ids) + total_missed_backfills = len(missed_ids) + nonrecurring_missed_backfills = len(nonrecurring_missed_ids) + else: + opportunity_ids = set() + missed_ids = set() + nonrecurring_missed_ids = set() + total_backfill_opportunities = int( + item.get("backfill_opportunity_jobs", 0) + ) + total_missed_backfills = int(item.get("missed_backfill_jobs", 0)) + nonrecurring_missed_backfills = int( + item.get("missed_nonrecurring_backfill_jobs", 0) + ) + # Remove known recurring/habitually-held misses from the denominator. + # Successful starts by those families remain, making this ratio + # conservative when family identity is uncertain. + adjusted_eligible_backfills = max( + 0, + total_backfill_opportunities + - max(0, total_missed_backfills - nonrecurring_missed_backfills), + ) + backfill_miss_fraction = ( + nonrecurring_missed_backfills / float(adjusted_eligible_backfills) + if adjusted_eligible_backfills else 0.0 + ) + item["eligible_nonheld_backfill_jobs"] = adjusted_eligible_backfills + if has_distinct_ids: + item["eligible_nonheld_backfill_job_ids"] = opportunity_ids - ( + missed_ids - nonrecurring_missed_ids + ) + item["backfill_miss_fraction"] = backfill_miss_fraction + backfill_opportunity_evidence = ( + not require_backfill_evidence + or ( + nonrecurring_missed_backfills >= min_missed_jobs + and int(item.get("missed_nonrecurring_backfill_families", 0)) + >= min_missed_families + # "Most" is strict: exactly half is not a majority. + and backfill_miss_fraction > min_backfill_miss_fraction + ) + ) + shutdown = ( + demand and general_opportunity_evidence + and running_fraction <= shutdown_fraction and item["jobs_started"] == 0 + ) + reduced = ( + demand and backfill_opportunity_evidence + and running_fraction <= reduced_fraction + ) + paused = ( + demand and general_opportunity_evidence + and running_fraction < 0.85 + and float(item["jobs_started"]) <= typical_starts * pause_start_fraction + ) + flags.append((shutdown, paused, reduced)) + # Shutdowns and queue pauses use combined direct-start and backfill + # evidence, while reduced-capacity and backfill-suppression bins use + # the stricter backfill evidence. Do not globally replace the + # state-specific gates with the backfill-only gate when an evidence + # file is present. + raw.append( + shutdown or paused or reduced + or (require_backfill_evidence and backfill_opportunity_evidence) + ) + + candidate = bridge_candidates(raw, bridge_bins_count) + minimum_bins = max(1, math.ceil(min_duration_hours * 3600 / width)) + periods = [] # type: List[Dict[str, object]] + i = 0 + while i < len(bins): + if not candidate[i]: + i += 1 + continue + j = i + 1 + while j < len(bins) and candidate[j]: + j += 1 + if j - i < minimum_bins: + i = j + continue + segment = bins[i:j] + real_flags = flags[i:j] + shutdown_share = sum(flag[0] for flag in real_flags) / len(real_flags) + pause_share = sum(flag[1] for flag in real_flags) / len(real_flags) + reduced_share = sum(flag[2] for flag in real_flags) / len(real_flags) + avg_running = sum(float(item["running_nodes"]) for item in segment) / len(segment) + avg_pending_jobs = sum(float(item["pending_jobs"]) for item in segment) / len(segment) + avg_pending_nodes = sum(float(item["pending_nodes"]) for item in segment) / len(segment) + avg_missed_jobs = sum( + float(item.get("missed_nonrecurring_jobs", 0)) for item in segment + ) / len(segment) + peak_missed_jobs = max( + int(item.get("missed_nonrecurring_jobs", 0)) for item in segment + ) + avg_missed_families = sum( + float(item.get("missed_nonrecurring_families", 0)) for item in segment + ) / len(segment) + use_distinct_jobs = all( + bool(item.get("has_distinct_backfill_job_ids", False)) + for item in segment + ) + if use_distinct_jobs: + eligible_job_ids = set() + missed_job_ids = set() + for item in segment: + eligible_job_ids.update( + item.get("eligible_nonheld_backfill_job_ids", set()) + ) + missed_job_ids.update( + item.get("missed_nonrecurring_backfill_job_ids", set()) + ) + eligible_backfills = len(eligible_job_ids) + missed_backfills = len(missed_job_ids) + else: + eligible_backfills = sum( + int(item.get("eligible_nonheld_backfill_jobs", 0)) for item in segment + ) + missed_backfills = sum( + int(item.get("missed_nonrecurring_backfill_jobs", 0)) for item in segment + ) + period_backfill_miss_fraction = ( + missed_backfills / float(eligible_backfills) if eligible_backfills else 0.0 + ) + jobs_started = sum(int(item["jobs_started"]) for item in segment) + capacity_fraction = avg_running / capacity + if shutdown_share >= 0.6 and jobs_started == 0: + state = "full_shutdown" + evidence = "queued demand, almost no node allocation, and no job starts" + elif pause_share >= 0.6: + state = "queue_pause_or_maintenance" + evidence = "queued demand and unusually few starts while limited work continued" + elif reduced_share >= 0.6: + state = "reduced_capacity" + evidence = "sustained allocation collapse despite queued demand" + else: + state = "backfill_suppression" + evidence = "multiple eligible backfill jobs were not started" + # The aggregate strict-majority check protects inferences based on + # backfill evidence. Shutdown and queue-pause classifications instead + # use the combined opportunity evidence already enforced per bin. + if ( + require_backfill_evidence + and state in ("reduced_capacity", "backfill_suppression") + and period_backfill_miss_fraction <= min_backfill_miss_fraction + ): + i = j + continue + if require_backfill_evidence: + evidence += "; multiple nonrecurring jobs missed reference EASY opportunities" + observed_peak = max( + float( + item.get( + "instantaneous_running_nodes", + item.get("observed_running_nodes", item["running_nodes"]), + ) + ) + for item in segment + ) + lower_candidates = [] + for item in segment: + missed_jobs = int(item.get("missed_nonrecurring_backfill_jobs", 0)) + missed_families = int( + item.get("missed_nonrecurring_backfill_families", 0) + ) + if ( + missed_jobs < min_missed_jobs + or missed_families < min_missed_families + or float(item.get("backfill_miss_fraction", 0)) + <= min_backfill_miss_fraction + ): + continue + minimum_job_nodes = int( + item.get("min_missed_nonrecurring_backfill_nodes", 0) + ) + if minimum_job_nodes <= 0: + continue + occupancy = float( + item.get( + "instantaneous_running_nodes", + item.get("observed_running_nodes", item["running_nodes"]), + ) + ) + lower_candidates.append(occupancy + minimum_job_nodes - 1) + effective_lower = max([observed_peak] + lower_candidates) + bound_status = "evidence_lower_bound_only" + duration_hours = len(segment) * width / 3600 + pressure = min(1.0, max(avg_pending_jobs / max(min_pending_jobs, 1), avg_pending_nodes / max(capacity * min_pending_fraction, 1)) / 5) + severity = min(1.0, max(0.0, 1.0 - capacity_fraction)) + longevity = min(1.0, duration_hours / 24) + if require_backfill_evidence: + opportunity_strength = min( + 1.0, + max( + avg_missed_jobs / max(min_missed_jobs * 5.0, 1), + avg_missed_families / max(min_missed_families * 5.0, 1), + ), + ) + confidence = min( + 0.99, + 0.25 + 0.15 * pressure + 0.2 * severity + + 0.15 * longevity + 0.25 * opportunity_strength, + ) + else: + confidence = min( + 0.99, 0.4 + 0.2 * pressure + 0.25 * severity + 0.15 * longevity + ) + periods.append( + { + "start": timestamps.value(int(segment[0]["start_epoch"])), + "end": timestamps.value(int(segment[-1]["start_epoch"]) + width), + "duration_hours": round(duration_hours, 2), + "state": state, + "maintenance_candidate": bool(require_backfill_evidence), + "confidence": round(confidence, 3), + "average_running_nodes": round(avg_running, 1), + "peak_bin_running_nodes": round(max(float(x["running_nodes"]) for x in segment), 1), + "fraction_of_normal_capacity": round(capacity_fraction, 4), + "average_pending_jobs": round(avg_pending_jobs, 1), + "average_pending_nodes": round(avg_pending_nodes, 1), + "jobs_started": jobs_started, + "average_missed_nonrecurring_jobs": round(avg_missed_jobs, 2), + "peak_missed_nonrecurring_jobs": peak_missed_jobs, + "average_missed_nonrecurring_families": round(avg_missed_families, 2), + "eligible_nonheld_backfill_jobs": eligible_backfills, + "missed_nonrecurring_backfill_jobs": missed_backfills, + "backfill_miss_fraction": round(period_backfill_miss_fraction, 4), + "backfill_evidence_count_basis": ( + "distinct_jobs" if use_distinct_jobs else "hourly_observations" + ), + "potential_effective_capacity_lower_nodes": round(effective_lower, 1), + "effective_capacity_bound_status": bound_status, + "capacity_bound_evidence_bins": len(lower_candidates), + "capacity_bound_conflicting_bins": 0, + "evidence": evidence, + } + ) + i = j + return periods + + +def write_bins( + path: str, + bins: Sequence[Dict[str, object]], + width: int, + capacity: float, + timestamps: TimestampFormatter, +) -> None: + fields = [ + "start", "end", "running_nodes", "capacity_fraction", "pending_jobs", + "pending_nodes", "jobs_submitted", "jobs_started", "nodes_started", + "direct_opportunity_jobs", "backfill_opportunity_jobs", + "missed_opportunity_jobs", "missed_nonrecurring_jobs", + "missed_nonrecurring_families", "missed_backfill_jobs", + "missed_nonrecurring_backfill_jobs", "queue_scan_truncated", + "missed_nonrecurring_direct_jobs", + "missed_nonrecurring_direct_families", + "missed_nonrecurring_backfill_families", + "min_missed_nonrecurring_nodes", + "min_missed_nonrecurring_backfill_nodes", + "instantaneous_running_nodes", + "observed_running_nodes", "observed_free_nodes", + "reclaiming_nodes", "unavailable_nodes", + "size_controlled_direct_jobs", "size_controlled_backfill_jobs", + "size_controlled_opportunity_jobs", + "eligible_nonheld_backfill_jobs", "backfill_miss_fraction", + ] + with open(path, "w", newline="", encoding="utf-8") as stream: + writer = csv.DictWriter(stream, fieldnames=fields) + writer.writeheader() + for item in bins: + epoch = int(item["start_epoch"]) + writer.writerow( + { + "start": timestamps.value(epoch), + "end": timestamps.value(epoch + width), + "running_nodes": round(float(item["running_nodes"]), 3), + "capacity_fraction": round(float(item["running_nodes"]) / capacity, 6), + "pending_jobs": round(float(item["pending_jobs"]), 3), + "pending_nodes": round(float(item["pending_nodes"]), 3), + "jobs_submitted": item["jobs_submitted"], + "jobs_started": item["jobs_started"], + "nodes_started": item["nodes_started"], + "instantaneous_running_nodes": round( + float(item.get("instantaneous_running_nodes", 0)), 3 + ), + "direct_opportunity_jobs": item.get("direct_opportunity_jobs", 0), + "backfill_opportunity_jobs": item.get("backfill_opportunity_jobs", 0), + "missed_opportunity_jobs": item.get("missed_opportunity_jobs", 0), + "missed_nonrecurring_jobs": item.get("missed_nonrecurring_jobs", 0), + "missed_nonrecurring_families": item.get("missed_nonrecurring_families", 0), + "missed_backfill_jobs": item.get("missed_backfill_jobs", 0), + "missed_nonrecurring_backfill_jobs": item.get( + "missed_nonrecurring_backfill_jobs", 0 + ), + "missed_nonrecurring_direct_jobs": item.get( + "missed_nonrecurring_direct_jobs", 0 + ), + "missed_nonrecurring_direct_families": item.get( + "missed_nonrecurring_direct_families", 0 + ), + "missed_nonrecurring_backfill_families": item.get( + "missed_nonrecurring_backfill_families", 0 + ), + "min_missed_nonrecurring_nodes": item.get( + "min_missed_nonrecurring_nodes", 0 + ), + "min_missed_nonrecurring_backfill_nodes": item.get( + "min_missed_nonrecurring_backfill_nodes", 0 + ), + "observed_running_nodes": item.get("observed_running_nodes", 0), + "observed_free_nodes": item.get("observed_free_nodes", 0), + "reclaiming_nodes": item.get("reclaiming_nodes", 0), + "unavailable_nodes": item.get("unavailable_nodes", 0), + "size_controlled_direct_jobs": item.get( + "size_controlled_direct_jobs", 0 + ), + "size_controlled_backfill_jobs": item.get( + "size_controlled_backfill_jobs", 0 + ), + "size_controlled_opportunity_jobs": item.get( + "size_controlled_opportunity_jobs", 0 + ), + "eligible_nonheld_backfill_jobs": item.get( + "eligible_nonheld_backfill_jobs", 0 + ), + "backfill_miss_fraction": round( + float(item.get("backfill_miss_fraction", 0)), 6 + ), + "queue_scan_truncated": item.get("queue_scan_truncated", 0), + } + ) + + +def parse_args(argv: Optional[Sequence[str]] = None) -> argparse.Namespace: + parser = argparse.ArgumentParser( + description="Infer downtime, queue pauses, and normal node capacity from job traces." + ) + parser.add_argument("inputs", nargs="+", help="CSV paths or glob patterns") + parser.add_argument("-o", "--output", help="JSON report (default: stdout)") + parser.add_argument("--bins-output", help="optional CSV containing the reconstructed timeline") + output_filter = parser.add_mutually_exclusive_group() + output_filter.add_argument( + "--only-queue-pause", dest="only_queue_pause", action="store_true", + default=True, + help=( + "write only queue_pause_or_maintenance periods (default); the " + "optional timeline CSV remains complete" + ), + ) + output_filter.add_argument( + "--all-states", dest="only_queue_pause", action="store_false", + help="write every detected period state to the JSON report", + ) + parser.add_argument( + "--backfill-evidence", + help="CSV from backfill_opportunity_audit.py; requires multi-job evidence", + ) + parser.add_argument( + "--min-missed-jobs", type=int, default=3, + help="minimum nonrecurring missed opportunities per bin (default: 3)", + ) + parser.add_argument( + "--min-missed-families", type=int, default=2, + help="minimum distinct nonrecurring families per bin (default: 2)", + ) + parser.add_argument( + "--min-backfill-miss-fraction", type=float, default=0.50, + help=( + "maintenance requires a miss fraction greater than this value " + "(default: 0.50, i.e. a strict majority)" + ), + ) + parser.add_argument( + "--time-format", choices=("epoch", "iso"), default="epoch", + help="timestamp representation (default: epoch)", + ) + parser.add_argument( + "--timezone", default="UTC", + help="timezone for ISO timestamps/month boundaries, e.g. PST, JST, or Asia/Tokyo (default: UTC)", + ) + parser.add_argument( + "--overlap-policy", choices=("combine", "filename-month"), default="combine", + help="combine all jobs, or stitch each YY_MM trace to its filename month (default: combine)", + ) + parser.add_argument("--bin-minutes", type=int, default=60, help="timeline resolution (default: 60)") + parser.add_argument("--known-capacity", type=int, help="normal node capacity; skips inference") + parser.add_argument("--capacity-quantile", type=float, default=0.995, help="capacity quantile (default: 0.995)") + parser.add_argument("--min-duration-hours", type=float, default=1.0, help="shortest reported anomaly (default: 1)") + parser.add_argument("--min-pending-jobs", type=float, default=10.0, help="demand threshold (default: 10)") + parser.add_argument("--min-pending-fraction", type=float, default=0.02, help="pending-node demand as capacity fraction (default: 0.02)") + parser.add_argument("--shutdown-fraction", type=float, default=0.005, help="allocation fraction considered shut down (default: 0.005)") + parser.add_argument("--reduced-fraction", type=float, default=0.60, help="allocation fraction considered reduced (default: 0.60)") + parser.add_argument("--pause-start-fraction", type=float, default=0.10, help="start rate fraction considered paused (default: 0.10)") + parser.add_argument("--bridge-bins", type=int, default=1, help="merge anomalies over this many normal bins (default: 1)") + args = parser.parse_args(argv) + if args.bin_minutes <= 0 or args.min_duration_hours <= 0: + parser.error("bin size and minimum duration must be positive") + if args.known_capacity is not None and args.known_capacity <= 0: + parser.error("--known-capacity must be positive") + if ( + args.min_pending_jobs < 0 or args.bridge_bins < 0 + or args.min_missed_jobs < 1 or args.min_missed_families < 1 + ): + parser.error("pending-job and bridge-bin thresholds cannot be negative") + for name in ( + "capacity_quantile", "min_pending_fraction", "shutdown_fraction", + "reduced_fraction", "pause_start_fraction", "min_backfill_miss_fraction", + ): + value = getattr(args, name) + if not 0 <= value <= 1: + parser.error(f"--{name.replace('_', '-')} must be between 0 and 1") + return args + + +def main(argv: Optional[Sequence[str]] = None) -> int: + args = parse_args(argv) + try: + timestamps = TimestampFormatter(args.time_format, args.timezone) + paths = expand_paths(args.inputs) + width = args.bin_minutes * 60 + builder = TimelineBuilder(width) + stats = read_traces(paths, builder, args.overlap_policy, timestamps) + bins = build_bins(builder, stats) + evidence_bins = 0 + evidence_parameters = {} + if args.backfill_evidence: + evidence_bins = load_backfill_evidence(args.backfill_evidence, bins) + evidence_row = next( + (item for item in bins if "successful_size_control_enabled" in item), + None, + ) + if evidence_row is not None: + evidence_parameters = { + "grace_seconds": evidence_row.get("grace_seconds", 0), + "release_delay_seconds": evidence_row.get( + "release_delay_seconds", 0 + ), + "successful_size_control_enabled": bool( + evidence_row.get("successful_size_control_enabled", 0) + ), + "period_count_basis": ( + "distinct_jobs" + if evidence_row.get("has_distinct_backfill_job_ids", False) + else "hourly_observations" + ), + } + capacity, capacity_basis = infer_capacity(bins, args.capacity_quantile, args.known_capacity) + periods = detect_periods( + bins, capacity, width, args.min_duration_hours, + args.min_pending_jobs, args.min_pending_fraction, + args.shutdown_fraction, args.reduced_fraction, + args.pause_start_fraction, args.bridge_bins, + timestamps, + bool(args.backfill_evidence), args.min_missed_jobs, + args.min_missed_families, args.min_backfill_miss_fraction, + ) + if args.only_queue_pause: + periods = [ + period for period in periods + if period["state"] == "queue_pause_or_maintenance" + ] + report = { + "schema_version": 1, + "analysis_interval": { + "start": timestamps.value((stats.analysis_start // width) * width), + "end": timestamps.value(((stats.analysis_end - 1) // width + 1) * width), + "bin_minutes": args.bin_minutes, + "time_format": args.time_format, + "timezone": args.timezone, + "overlap_policy": args.overlap_policy, + }, + "input": { + "files": paths, + "rows": stats.rows, + "invalid_rows": stats.invalid_rows, + "invalid_intervals": stats.invalid_intervals, + "backfill_evidence": ( + None if not args.backfill_evidence else { + "path": os.path.abspath(args.backfill_evidence), + "matched_bins": evidence_bins, + "parameters": evidence_parameters, + } + ), + }, + "normal_capacity": { + "inferred_nodes": round(capacity, 1), + "basis": capacity_basis, + "interpretation": ( + "user-supplied total system capacity" + if args.known_capacity is not None + else "observed schedulable capacity; a lower bound when demand is insufficient" + ), + }, + "detection_parameters": { + "minimum_duration_hours": args.min_duration_hours, + "minimum_pending_jobs": args.min_pending_jobs, + "minimum_pending_node_fraction": args.min_pending_fraction, + "shutdown_capacity_fraction": args.shutdown_fraction, + "reduced_capacity_fraction": args.reduced_fraction, + "pause_start_rate_fraction": args.pause_start_fraction, + "bridged_normal_bins": args.bridge_bins, + "requires_backfill_evidence": bool(args.backfill_evidence), + "minimum_missed_nonrecurring_jobs": args.min_missed_jobs, + "minimum_missed_nonrecurring_families": args.min_missed_families, + "minimum_backfill_miss_fraction": args.min_backfill_miss_fraction, + "output_state_filter": ( + "queue_pause_or_maintenance" if args.only_queue_pause else None + ), + }, + "capacity_by_month": capacity_windows( + bins, args.capacity_quantile, capacity, timestamps + ), + "periods": periods, + "limitations": [ + "The input schema has no queue/partition field, so queue_pause_or_maintenance is behavioral evidence, not a proven queue identity.", + "Pending demand is reconstructed from jobs present in the trace; canceled or omitted jobs are invisible.", + "Node allocation measures reserved resources, not CPU utilization or physical node health.", + "Backfill replay cannot observe queue eligibility, explicit holds, dependencies, reservations, or node topology.", + "Recurring and habitually missed families are inferred from resource, time-limit, power, and submission patterns.", + ], + } + rendered = json.dumps(report, indent=2) + "\n" + if args.output: + Path(args.output).write_text(rendered, encoding="utf-8") + else: + sys.stdout.write(rendered) + if args.bins_output: + write_bins(args.bins_output, bins, width, capacity, timestamps) + except (OSError, ValueError) as exc: + print(f"error: {exc}", file=sys.stderr) + return 2 + return 0 + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/scripts/detect_queue_pause/generate_capacity_schedule.sh b/scripts/detect_queue_pause/generate_capacity_schedule.sh new file mode 100755 index 0000000..ac1fec3 --- /dev/null +++ b/scripts/detect_queue_pause/generate_capacity_schedule.sh @@ -0,0 +1,139 @@ +#!/usr/bin/env bash +set -euo pipefail + +usage() { + cat <<'EOF' +Usage: ./generate_capacity_schedule.sh [TRACE_PATTERN] [OUTPUT_DIRECTORY] + +Generate replay evidence, inferred operating periods, two hourly capacity +schedules, and their silhouette plots. TRACE_PATTERN defaults to +*_scheduling_trace.csv and OUTPUT_DIRECTORY defaults to the current directory. + +Optional environment settings: + CAPACITY_NODES=N (required structural node capacity) + TRACE_TIMEZONE=UTC + OVERLAP_POLICY=combine + GRACE_MINUTES=60 + RELEASE_DELAY_MINUTES=10 + CASES_PER_ROW=21 + SIMULATOR_FORMAT=0 (1 writes time,total_nodes CSVs) +EOF +} + +if [[ "${1:-}" == "-h" || "${1:-}" == "--help" ]]; then + usage + exit 0 +fi + +SCRIPT_DIR=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd -P) +CALL_DIR=$(pwd -P) +TRACE_PATTERN=${1:-"*_scheduling_trace.csv"} +OUTPUT_DIRECTORY=${2:-"$CALL_DIR"} + +CAPACITY_NODES=${CAPACITY_NODES:-} +TRACE_TIMEZONE=${TRACE_TIMEZONE:-UTC} +OVERLAP_POLICY=${OVERLAP_POLICY:-combine} +GRACE_MINUTES=${GRACE_MINUTES:-60} +RELEASE_DELAY_MINUTES=${RELEASE_DELAY_MINUTES:-10} +CASES_PER_ROW=${CASES_PER_ROW:-21} +SIMULATOR_FORMAT=${SIMULATOR_FORMAT:-0} + +if [[ -z "$CAPACITY_NODES" ]]; then + echo "CAPACITY_NODES must be set to the machine's structural node capacity" >&2 + echo "Example: CAPACITY_NODES=158976 $0 '$TRACE_PATTERN' '$OUTPUT_DIRECTORY'" >&2 + exit 2 +fi + +if [[ "$TRACE_PATTERN" != /* ]]; then + TRACE_PATTERN="$CALL_DIR/$TRACE_PATTERN" +fi +if [[ "$OUTPUT_DIRECTORY" != /* ]]; then + OUTPUT_DIRECTORY="$CALL_DIR/$OUTPUT_DIRECTORY" +fi +mkdir -p "$OUTPUT_DIRECTORY" +OUTPUT_DIRECTORY=$(cd "$OUTPUT_DIRECTORY" && pwd -P) + +BOOTSTRAP_REPORT=$(mktemp "$OUTPUT_DIRECTORY/.capacity-bootstrap.XXXXXXXX") +cleanup() { + rm -f "$BOOTSTRAP_REPORT" +} +trap cleanup EXIT + +TIMELINE="$OUTPUT_DIRECTORY/capacity_timeline.csv" +EVIDENCE="$OUTPUT_DIRECTORY/backfill_opportunities.csv" +REPORT="$OUTPUT_DIRECTORY/maintenance_report.json" +WITH_REDUCED="$OUTPUT_DIRECTORY/resource_capacity_with_reduced_capacity.csv" +WITHOUT_REDUCED="$OUTPUT_DIRECTORY/resource_capacity_without_reduced_capacity.csv" +WITH_REDUCED_PLOT="$OUTPUT_DIRECTORY/inferred_capacity_silhouette_labeled_jst.png" +QUEUE_PAUSE_PLOT="$OUTPUT_DIRECTORY/queue_pause_capacity_silhouette_labeled_jst.png" + +echo "[1/6] Building the bootstrap hourly timeline" +python3 "$SCRIPT_DIR/discover_maintenance.py" "$TRACE_PATTERN" \ + --overlap-policy "$OVERLAP_POLICY" \ + --timezone "$TRACE_TIMEZONE" \ + --known-capacity "$CAPACITY_NODES" \ + --all-states \ + --output "$BOOTSTRAP_REPORT" \ + --bins-output "$TIMELINE" + +echo "[2/6] Replaying EASY backfill opportunities" +python3 "$SCRIPT_DIR/backfill_opportunity_audit.py" "$TRACE_PATTERN" \ + --timeline "$TIMELINE" \ + --nodes "$CAPACITY_NODES" \ + --timezone "$TRACE_TIMEZONE" \ + --grace-minutes "$GRACE_MINUTES" \ + --release-delay-minutes "$RELEASE_DELAY_MINUTES" \ + --output "$EVIDENCE" + +echo "[3/6] Detecting final operating-state periods" +python3 "$SCRIPT_DIR/discover_maintenance.py" "$TRACE_PATTERN" \ + --overlap-policy "$OVERLAP_POLICY" \ + --timezone "$TRACE_TIMEZONE" \ + --known-capacity "$CAPACITY_NODES" \ + --backfill-evidence "$EVIDENCE" \ + --all-states \ + --output "$REPORT" \ + --bins-output "$TIMELINE" + +echo "[4/6] Building capacity schedule CSV files" +BUILDER_FORMAT=() +if [[ "$SIMULATOR_FORMAT" == "1" ]]; then + BUILDER_FORMAT=(--simulator-format) +fi +python3 "$SCRIPT_DIR/build_resource_capacity_trace.py" \ + --report "$REPORT" \ + --timeline "$TIMELINE" \ + --capacity "$CAPACITY_NODES" \ + --timezone "$TRACE_TIMEZONE" \ + --with-reduced-output "$WITH_REDUCED" \ + --without-reduced-output "$WITHOUT_REDUCED" \ + "${BUILDER_FORMAT[@]}" + +CAPACITY_MPL_DIR=${MPLCONFIGDIR:-"$OUTPUT_DIRECTORY/.matplotlib-cache"} +mkdir -p "$CAPACITY_MPL_DIR" + +echo "[5/6] Plotting inferred reductions and queue pauses" +MPLCONFIGDIR="$CAPACITY_MPL_DIR" python3 "$SCRIPT_DIR/plot_capacity_silhouette.py" \ + "$WITH_REDUCED" \ + --output "$WITH_REDUCED_PLOT" \ + --title "Inferred normal-queue capacity (reductions and queue pauses)" \ + --timezone "$TRACE_TIMEZONE" \ + --cases-per-row "$CASES_PER_ROW" + +echo "[6/6] Plotting the queue-pause-only schedule" +MPLCONFIGDIR="$CAPACITY_MPL_DIR" python3 "$SCRIPT_DIR/plot_capacity_silhouette.py" \ + "$WITHOUT_REDUCED" \ + --output "$QUEUE_PAUSE_PLOT" \ + --title "Queue-pause-only normal-queue capacity" \ + --timezone "$TRACE_TIMEZONE" \ + --cases-per-row "$CASES_PER_ROW" + +echo "Generated outputs in $OUTPUT_DIRECTORY" +printf ' %s\n' \ + "$TIMELINE" \ + "$EVIDENCE" \ + "$REPORT" \ + "$WITH_REDUCED" \ + "$WITHOUT_REDUCED" \ + "$WITH_REDUCED_PLOT" \ + "$QUEUE_PAUSE_PLOT" diff --git a/scripts/detect_queue_pause/plot_capacity_silhouette.py b/scripts/detect_queue_pause/plot_capacity_silhouette.py new file mode 100755 index 0000000..e75b241 --- /dev/null +++ b/scripts/detect_queue_pause/plot_capacity_silhouette.py @@ -0,0 +1,163 @@ +#!/usr/bin/env python3 +"""Plot a capacity trace as wrapped, uniformly spaced rectangular segments.""" + +import argparse +import csv +import math +from datetime import datetime + +import matplotlib +matplotlib.use("Agg") +import matplotlib.pyplot as plt +from matplotlib.lines import Line2D +from matplotlib.patches import Patch +from matplotlib.ticker import StrMethodFormatter + +from discover_maintenance import TimestampFormatter + + +COLORS = { + "full": "#b9d4e8", + "reduced": "#d17a22", + "paused": "#7a0177", +} + + +def load_cases(path, timezone): + with open(path, newline="", encoding="utf-8") as stream: + rows = list(csv.DictReader(stream)) + if not rows: + raise ValueError("{} is empty".format(path)) + simulator_format = "time" in rows[0] and "total_nodes" in rows[0] + if simulator_format: + full_capacity = max(int(row["total_nodes"]) for row in rows) + values = [int(row["total_nodes"]) for row in rows] + time_field = "time" + else: + full_capacity = max(int(row["structural_capacity_nodes"]) for row in rows) + values = [int(row["normal_queue_capacity_nodes"]) for row in rows] + time_field = "start" + changes = [0] + for index in range(1, len(rows)): + if values[index] != values[index - 1]: + changes.append(index) + cases = [] + for case_index, row_index in enumerate(changes): + end_index = changes[case_index + 1] if case_index + 1 < len(changes) else len(rows) + capacity = values[row_index] + if capacity == 0: + kind = "paused" + elif capacity < full_capacity: + kind = "reduced" + else: + kind = "full" + cases.append( + { + "date": datetime.fromtimestamp( + int(rows[row_index][time_field]), timezone + ), + "capacity": capacity, + "kind": kind, + "duration_hours": end_index - row_index, + } + ) + return cases, full_capacity + + +def plot_cases(cases, full_capacity, output, title, timezone_name, cases_per_row, dpi): + row_count = int(math.ceil(len(cases) / float(cases_per_row))) + figure, axes = plt.subplots( + row_count, 1, figsize=(18, max(3.2 * row_count, 5)), squeeze=False + ) + axes = [row[0] for row in axes] + for row_index, axis in enumerate(axes): + first = row_index * cases_per_row + last = min(len(cases), first + cases_per_row) + block = cases[first:last] + positions = list(range(len(block))) + values = [case["capacity"] for case in block] + colors = [COLORS[case["kind"]] for case in block] + + # Width 1 and align=edge make adjacent cases a continuous bar silhouette. + axis.bar( + positions, values, width=1.0, align="edge", color=colors, + edgecolor="#333333", linewidth=0.45, + ) + for position, case in zip(positions, block): + if case["kind"] == "paused": + # A zero-height capacity bar is otherwise invisible. Keep the + # value at zero and emphasize only its baseline. + axis.hlines( + 0, position, position + 1, colors=COLORS["paused"], + linewidth=6.0, zorder=5, + ) + outline_x = list(range(len(block) + 1)) + outline_y = values + [values[-1]] + axis.step(outline_x, outline_y, where="post", color="#222222", linewidth=0.75) + axis.axhline( + full_capacity, color="#555555", linestyle="--", linewidth=1.0 + ) + axis.set_xlim(0, len(block)) + axis.set_ylim(-0.02 * full_capacity, 1.06 * full_capacity) + axis.set_xticks([value + 0.5 for value in positions]) + axis.set_xticklabels( + [case["date"].strftime("%Y-%m-%d") for case in block], + rotation=60, ha="right", fontsize=8, + ) + axis.set_ylabel("Capacity\n(nodes)") + axis.yaxis.set_major_formatter(StrMethodFormatter("{x:,.0f}")) + axis.grid(axis="y", alpha=0.20) + axis.set_title( + "Change cases {:d}–{:d}".format(first + 1, last), + loc="left", fontsize=10, + ) + axes[0].legend( + handles=[ + Patch(facecolor=COLORS["full"], edgecolor="#333333", label="Full capacity"), + Patch(facecolor=COLORS["reduced"], edgecolor="#333333", label="Inferred reduction"), + Line2D( + [0], [0], color=COLORS["paused"], linewidth=6.0, + label="Queue paused (capacity 0 line)", + ), + ], + loc="lower right", fontsize=8, ncol=3, + ) + axes[-1].set_xlabel( + "Capacity-change date ({}, uniform spacing; duration not to scale)".format( + timezone_name + ) + ) + figure.suptitle(title, fontsize=15) + figure.subplots_adjust( + top=0.96, bottom=0.08, left=0.08, right=0.99, hspace=0.80 + ) + figure.savefig(output, dpi=dpi) + plt.close(figure) + + +def parse_args(): + parser = argparse.ArgumentParser(description="Plot a capacity silhouette") + parser.add_argument("input") + parser.add_argument("--output", required=True) + parser.add_argument("--title", required=True) + parser.add_argument("--timezone", default="UTC") + parser.add_argument("--cases-per-row", type=int, default=21) + parser.add_argument("--dpi", type=int, default=180) + return parser.parse_args() + + +def main(): + args = parse_args() + if args.cases_per_row <= 0: + raise ValueError("--cases-per-row must be positive") + timezone = TimestampFormatter("epoch", args.timezone).timezone + cases, full_capacity = load_cases(args.input, timezone) + plot_cases( + cases, full_capacity, args.output, args.title, args.timezone, + args.cases_per_row, args.dpi, + ) + print("{} ({} change cases)".format(args.output, len(cases))) + + +if __name__ == "__main__": + main() diff --git a/scripts/detect_queue_pause/requirements.txt b/scripts/detect_queue_pause/requirements.txt new file mode 100644 index 0000000..dd69cd5 --- /dev/null +++ b/scripts/detect_queue_pause/requirements.txt @@ -0,0 +1,2 @@ +matplotlib>=3.3 +tzdata>=2024.1 diff --git a/scripts/detect_queue_pause/test_backfill_opportunity_audit.py b/scripts/detect_queue_pause/test_backfill_opportunity_audit.py new file mode 100755 index 0000000..5157dc2 --- /dev/null +++ b/scripts/detect_queue_pause/test_backfill_opportunity_audit.py @@ -0,0 +1,89 @@ +import csv +import os +import tempfile +import unittest + +from backfill_opportunity_audit import Job, recurring_families, replay_file + + +class BackfillAuditTests(unittest.TestCase): + def test_detects_missed_easy_backfill(self): + handle, path = tempfile.mkstemp(suffix="_trace.csv") + os.close(handle) + try: + with open(path, "w", newline="") as stream: + writer = csv.writer(stream) + writer.writerow( + ["job_submit_time", "begin_time", "end_time", "time_limit", + "num_nodes", "avgpcon"] + ) + # Eight nodes are occupied. The five-node FCFS head is blocked, + # while the later two-node job safely fits before its reservation. + writer.writerow([-200, -100, 1000, 1100, 8, 800]) + writer.writerow([-90, 500, 600, 100, 5, 500]) + writer.writerow([-80, 400, 500, 100, 2, 200]) + result = replay_file( + path, (0, 3600), [0], 10, 60, 3, 0.2, 0.5, 1000 + )[0] + self.assertEqual(result["backfill_opportunity_jobs"], 1) + self.assertEqual(result["missed_nonrecurring_backfill_jobs"], 1) + self.assertEqual(result["missed_nonrecurring_backfill_families"], 1) + finally: + os.unlink(path) + + def test_recognizes_regular_family(self): + family = (2, 600, 100) + jobs = [ + Job(index, index * 86400, index * 86400 + 100, index * 86400 + 200, + 2, 600, family) + for index in range(4) + ] + self.assertEqual(recurring_families(jobs, 3, 0.2), {family}) + + def test_equal_or_larger_successful_start_discounts_miss(self): + handle, path = tempfile.mkstemp(suffix="_trace.csv") + os.close(handle) + try: + with open(path, "w", newline="") as stream: + writer = csv.writer(stream) + writer.writerow( + ["job_submit_time", "begin_time", "end_time", "time_limit", + "num_nodes", "avgpcon"] + ) + writer.writerow([-200, -100, 1000, 1100, 8, 800]) + writer.writerow([-90, 500, 600, 100, 5, 500]) + # This two-node backfill appears schedulable but remains queued. + writer.writerow([-80, 400, 500, 100, 2, 200]) + # A larger job already pending at the snapshot starts promptly. + writer.writerow([-70, 30, 40, 10, 3, 300]) + result = replay_file( + path, (0, 3600), [0], 10, 60, 3, 0.2, 0.5, 1000 + )[0] + self.assertGreaterEqual(result["size_controlled_backfill_jobs"], 1) + self.assertEqual(result["missed_nonrecurring_backfill_jobs"], 0) + finally: + os.unlink(path) + + def test_release_delay_keeps_nodes_unavailable(self): + handle, path = tempfile.mkstemp(suffix="_trace.csv") + os.close(handle) + try: + with open(path, "w", newline="") as stream: + writer = csv.writer(stream) + writer.writerow( + ["job_submit_time", "begin_time", "end_time", "time_limit", + "num_nodes", "avgpcon"] + ) + writer.writerow([-200, -100, -30, 70, 8, 800]) + result = replay_file( + path, (-100, 3600), [0], 10, 60, 3, 0.2, 0.5, 1000, 60 + )[0] + self.assertEqual(result["observed_running_nodes"], 0) + self.assertEqual(result["reclaiming_nodes"], 8) + self.assertEqual(result["observed_free_nodes"], 2) + finally: + os.unlink(path) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/detect_queue_pause/test_build_resource_capacity_trace.py b/scripts/detect_queue_pause/test_build_resource_capacity_trace.py new file mode 100755 index 0000000..decdf30 --- /dev/null +++ b/scripts/detect_queue_pause/test_build_resource_capacity_trace.py @@ -0,0 +1,39 @@ +import unittest + +from build_resource_capacity_trace import build_rows + + +class ResourceCapacityTraceTests(unittest.TestCase): + def test_capacity_scenarios(self): + timeline = [ + {"start": str(hour * 3600), "end": str((hour + 1) * 3600)} + for hour in range(4) + ] + periods = [ + { + "start": 3600, "end": 7200, "state": "reduced_capacity", + "potential_effective_capacity_lower_nodes": 60, + }, + { + "start": 7200, "end": 10800, + "state": "queue_pause_or_maintenance", + "potential_effective_capacity_lower_nodes": 20, + }, + ] + with_reduced = build_rows(timeline, periods, 100, True, "JST") + without_reduced = build_rows(timeline, periods, 100, False, "JST") + self.assertEqual( + [row["normal_queue_capacity_nodes"] for row in with_reduced], + [100, 60, 0, 100], + ) + self.assertEqual( + [row["normal_queue_capacity_nodes"] for row in without_reduced], + [100, 100, 0, 100], + ) + self.assertEqual(with_reduced[0]["timezone"], "JST") + self.assertIn("start_local", with_reduced[0]) + self.assertIn("end_local", with_reduced[0]) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/detect_queue_pause/test_discover_maintenance.py b/scripts/detect_queue_pause/test_discover_maintenance.py new file mode 100755 index 0000000..e4717e4 --- /dev/null +++ b/scripts/detect_queue_pause/test_discover_maintenance.py @@ -0,0 +1,330 @@ +import unittest + +from discover_maintenance import ( + TimelineBuilder, + TimestampFormatter, + TraceStats, + build_bins, + detect_periods, + filename_month_window, + parse_args, +) + + +class DetectorTests(unittest.TestCase): + def test_only_queue_pause_option(self): + self.assertTrue(parse_args(["trace.csv"]).only_queue_pause) + self.assertTrue( + parse_args(["trace.csv", "--only-queue-pause"]).only_queue_pause + ) + self.assertFalse(parse_args(["trace.csv", "--all-states"]).only_queue_pause) + + def test_interval_accounting_across_bins(self): + builder = TimelineBuilder(3600) + builder.add_job(0, 1800, 5400, 10) + bins = build_bins(builder, TraceStats(rows=1, min_submit=0, max_submit=7200)) + self.assertEqual([x["running_nodes"] for x in bins], [5, 5, 0]) + self.assertEqual( + [x["instantaneous_running_nodes"] for x in bins], [0, 10, 0] + ) + self.assertEqual(bins[0]["pending_jobs"], 0.5) + + def test_period_majority_includes_bridged_hour(self): + bins = [] + for hour, opportunities, misses in ((0, 4, 3), (1, 100, 0), (2, 4, 3)): + bins.append( + { + "start_epoch": hour * 3600, + "running_nodes": 20, + "pending_jobs": 20, + "pending_nodes": 200, + "jobs_started": 5, + "backfill_opportunity_jobs": opportunities, + "missed_backfill_jobs": misses, + "missed_nonrecurring_backfill_jobs": misses, + "missed_nonrecurring_backfill_families": 2 if misses else 0, + "missed_nonrecurring_jobs": misses, + "missed_nonrecurring_families": 2 if misses else 0, + } + ) + periods = detect_periods( + bins, 100, 3600, 1, 10, 0.02, 0.005, 0.6, 0.1, 1, + TimestampFormatter("epoch", "UTC"), True, 3, 2, 0.5, + ) + self.assertEqual(periods, []) + + def test_detects_backlogged_shutdown(self): + bins = [] + for hour in range(24): + shutdown = 8 <= hour < 16 + bins.append( + { + "start_epoch": hour * 3600, + "running_nodes": 0 if shutdown else 100, + "pending_jobs": 50 if shutdown else 0, + "pending_nodes": 500 if shutdown else 0, + "jobs_started": 0 if shutdown else 20, + } + ) + periods = detect_periods( + bins, 100, 3600, 6, 10, 0.02, 0.005, 0.6, 0.1, 0, + TimestampFormatter("epoch", "UTC"), + ) + self.assertEqual(len(periods), 1) + self.assertEqual(periods[0]["state"], "full_shutdown") + self.assertEqual(periods[0]["duration_hours"], 8) + + def test_queue_pause_uses_combined_evidence_when_backfill_is_enabled(self): + item = { + "start_epoch": 0, + "running_nodes": 20, + "pending_jobs": 20, + "pending_nodes": 200, + "jobs_started": 0, + "missed_nonrecurring_jobs": 3, + "missed_nonrecurring_families": 2, + "backfill_opportunity_jobs": 0, + "missed_backfill_jobs": 0, + "missed_nonrecurring_backfill_jobs": 0, + "missed_nonrecurring_backfill_families": 0, + } + periods = detect_periods( + [item], 100, 3600, 1, 10, 0.02, 0.005, 0.6, 0.1, 0, + TimestampFormatter("epoch", "UTC"), True, 3, 2, 0.5, + ) + self.assertEqual(len(periods), 1) + self.assertEqual(periods[0]["state"], "queue_pause_or_maintenance") + self.assertEqual(periods[0]["backfill_miss_fraction"], 0.0) + + def test_shutdown_uses_combined_evidence_when_backfill_is_enabled(self): + item = { + "start_epoch": 0, + "running_nodes": 0, + "pending_jobs": 20, + "pending_nodes": 200, + "jobs_started": 0, + "missed_nonrecurring_jobs": 3, + "missed_nonrecurring_families": 2, + "backfill_opportunity_jobs": 0, + "missed_backfill_jobs": 0, + "missed_nonrecurring_backfill_jobs": 0, + "missed_nonrecurring_backfill_families": 0, + } + periods = detect_periods( + [item], 100, 3600, 1, 10, 0.02, 0.005, 0.6, 0.1, 0, + TimestampFormatter("epoch", "UTC"), True, 3, 2, 0.5, + ) + self.assertEqual(len(periods), 1) + self.assertEqual(periods[0]["state"], "full_shutdown") + self.assertEqual(periods[0]["backfill_miss_fraction"], 0.0) + + def test_idle_without_backlog_is_not_maintenance(self): + bins = [ + { + "start_epoch": hour * 3600, + "running_nodes": 0, + "pending_jobs": 0, + "pending_nodes": 0, + "jobs_started": 0, + } + for hour in range(12) + ] + periods = detect_periods( + bins, 100, 3600, 6, 10, 0.02, 0.005, 0.6, 0.1, 1, + TimestampFormatter("epoch", "UTC"), + ) + self.assertEqual(periods, []) + + def test_reduced_capacity_requires_multi_family_backfill_evidence(self): + item = { + "start_epoch": 0, + "running_nodes": 20, + "pending_jobs": 20, + "pending_nodes": 200, + "jobs_started": 5, + "missed_nonrecurring_backfill_jobs": 2, + "missed_nonrecurring_backfill_families": 2, + "missed_nonrecurring_jobs": 2, + "missed_nonrecurring_families": 2, + "backfill_opportunity_jobs": 3, + "missed_backfill_jobs": 2, + } + arguments = ( + [item], 100, 3600, 1, 10, 0.02, 0.005, 0.6, 0.1, 0, + TimestampFormatter("epoch", "UTC"), True, 3, 2, + ) + self.assertEqual(detect_periods(*arguments), []) + item["missed_nonrecurring_backfill_jobs"] = 3 + item["missed_backfill_jobs"] = 3 + periods = detect_periods(*arguments) + self.assertEqual(len(periods), 1) + self.assertEqual(periods[0]["state"], "reduced_capacity") + + def test_backfill_evidence_is_maintenance_above_reduced_threshold(self): + item = { + "start_epoch": 0, + "running_nodes": 80, + "pending_jobs": 20, + "pending_nodes": 200, + "jobs_started": 5, + "missed_nonrecurring_backfill_jobs": 3, + "missed_nonrecurring_backfill_families": 2, + "missed_nonrecurring_jobs": 3, + "missed_nonrecurring_families": 2, + "backfill_opportunity_jobs": 5, + "missed_backfill_jobs": 3, + "min_missed_nonrecurring_backfill_nodes": 2, + "observed_running_nodes": 80, + "instantaneous_running_nodes": 90, + } + periods = detect_periods( + [item], 100, 3600, 1, 10, 0.02, 0.005, 0.6, 0.1, 0, + TimestampFormatter("epoch", "UTC"), True, 3, 2, + ) + self.assertEqual(len(periods), 1) + self.assertEqual(periods[0]["state"], "backfill_suppression") + self.assertEqual(periods[0]["potential_effective_capacity_lower_nodes"], 91) + self.assertNotIn("potential_effective_capacity_upper_nodes", periods[0]) + + def test_capacity_lower_bound_is_maximum_missed_fit_threshold(self): + bins = [] + for hour, occupancy, smallest_job in ((0, 20, 10), (1, 30, 2)): + bins.append( + { + "start_epoch": hour * 3600, + "running_nodes": occupancy, + "instantaneous_running_nodes": occupancy, + "pending_jobs": 20, + "pending_nodes": 200, + "jobs_started": 5, + "backfill_opportunity_jobs": 5, + "missed_backfill_jobs": 3, + "missed_nonrecurring_backfill_jobs": 3, + "missed_nonrecurring_backfill_families": 2, + "missed_nonrecurring_jobs": 3, + "missed_nonrecurring_families": 2, + "min_missed_nonrecurring_backfill_nodes": smallest_job, + } + ) + periods = detect_periods( + bins, 100, 3600, 1, 10, 0.02, 0.005, 0.6, 0.1, 0, + TimestampFormatter("epoch", "UTC"), True, 3, 2, 0.5, + ) + self.assertEqual(len(periods), 1) + self.assertEqual(periods[0]["potential_effective_capacity_lower_nodes"], 31) + self.assertNotIn("potential_effective_capacity_upper_nodes", periods[0]) + + def test_backfill_minimum_counts_without_majority_are_not_maintenance(self): + item = { + "start_epoch": 0, + "running_nodes": 20, + "pending_jobs": 20, + "pending_nodes": 200, + "jobs_started": 5, + "backfill_opportunity_jobs": 10, + "missed_backfill_jobs": 3, + "missed_nonrecurring_backfill_jobs": 3, + "missed_nonrecurring_backfill_families": 2, + "missed_nonrecurring_jobs": 3, + "missed_nonrecurring_families": 2, + } + periods = detect_periods( + [item], 100, 3600, 1, 10, 0.02, 0.005, 0.6, 0.1, 0, + TimestampFormatter("epoch", "UTC"), True, 3, 2, 0.5, + ) + self.assertEqual(periods, []) + self.assertEqual(item["eligible_nonheld_backfill_jobs"], 10) + self.assertEqual(item["backfill_miss_fraction"], 0.3) + + def test_strict_backfill_majority_is_maintenance(self): + item = { + "start_epoch": 0, + "running_nodes": 20, + "pending_jobs": 20, + "pending_nodes": 200, + "jobs_started": 5, + "backfill_opportunity_jobs": 10, + "missed_backfill_jobs": 6, + "missed_nonrecurring_backfill_jobs": 6, + "missed_nonrecurring_backfill_families": 2, + "missed_nonrecurring_jobs": 6, + "missed_nonrecurring_families": 2, + } + periods = detect_periods( + [item], 100, 3600, 1, 10, 0.02, 0.005, 0.6, 0.1, 0, + TimestampFormatter("epoch", "UTC"), True, 3, 2, 0.5, + ) + self.assertEqual(len(periods), 1) + self.assertEqual(periods[0]["backfill_miss_fraction"], 0.6) + + def test_exactly_half_of_backfills_is_not_a_majority(self): + item = { + "start_epoch": 0, + "running_nodes": 20, + "pending_jobs": 20, + "pending_nodes": 200, + "jobs_started": 5, + "backfill_opportunity_jobs": 6, + "missed_backfill_jobs": 3, + "missed_nonrecurring_backfill_jobs": 3, + "missed_nonrecurring_backfill_families": 2, + "missed_nonrecurring_jobs": 3, + "missed_nonrecurring_families": 2, + } + periods = detect_periods( + [item], 100, 3600, 1, 10, 0.02, 0.005, 0.6, 0.1, 0, + TimestampFormatter("epoch", "UTC"), True, 3, 2, 0.5, + ) + self.assertEqual(periods, []) + + def test_period_majority_counts_distinct_jobs_not_hourly_duplicates(self): + bins = [] + for hour, successful_ids in enumerate((("d", "e"), ("f", "g"))): + opportunity_ids = {"a", "b", "c"} | set(successful_ids) + bins.append( + { + "start_epoch": hour * 3600, + "running_nodes": 20, + "pending_jobs": 20, + "pending_nodes": 200, + "jobs_started": 5, + "backfill_opportunity_jobs": 5, + "missed_backfill_jobs": 3, + "missed_nonrecurring_backfill_jobs": 3, + "missed_nonrecurring_backfill_families": 2, + "missed_nonrecurring_jobs": 3, + "missed_nonrecurring_families": 2, + "has_distinct_backfill_job_ids": True, + "backfill_opportunity_job_ids": opportunity_ids, + "missed_backfill_job_ids": {"a", "b", "c"}, + "missed_nonrecurring_backfill_job_ids": {"a", "b", "c"}, + } + ) + periods = detect_periods( + bins, 100, 3600, 1, 10, 0.02, 0.005, 0.6, 0.1, 0, + TimestampFormatter("epoch", "UTC"), True, 3, 2, 0.5, + ) + # Each hour is 3/5, but the merged period is only 3/7 distinct jobs. + self.assertEqual(periods, []) + + def test_timestamp_formats_and_fixed_timezones(self): + self.assertEqual(TimestampFormatter("epoch", "JST").value(0), 0) + self.assertEqual( + TimestampFormatter("iso", "JST").value(0), + "1970-01-01T09:00:00+09:00", + ) + self.assertEqual( + TimestampFormatter("iso", "PST").value(0), + "1969-12-31T16:00:00-08:00", + ) + + def test_filename_month_uses_selected_timezone(self): + start, end = filename_month_window( + "23_08_scheduling_trace.csv", TimestampFormatter("epoch", "JST") + ) + self.assertEqual(start, 1690815600) + self.assertEqual(end, 1693494000) + + +if __name__ == "__main__": + unittest.main() diff --git a/scripts/prepare_warm_start_trace.py b/scripts/prepare_warm_start_trace.py new file mode 100755 index 0000000..38bf6f5 --- /dev/null +++ b/scripts/prepare_warm_start_trace.py @@ -0,0 +1,145 @@ +#!/usr/bin/env python3 +"""Build a simulation trace whose initial allocation matches history at t.""" + +import argparse +import csv +from datetime import datetime +import os +import sys +import time + + +def parse_time(value, timezone): + value = value.strip() + try: + return float(value) + except ValueError: + pass + normalized = value.replace("Z", "+00:00") + parsed = datetime.fromisoformat(normalized) + if parsed.tzinfo is not None: + return parsed.timestamp() + old_tz = os.environ.get("TZ") + try: + os.environ["TZ"] = timezone + time.tzset() + return time.mktime(parsed.timetuple()) + parsed.microsecond / 1e6 + finally: + if old_tz is None: + os.environ.pop("TZ", None) + else: + os.environ["TZ"] = old_tz + time.tzset() + + +def value(row, *names): + for name in names: + if name in row and row[name] != "": + return row[name] + raise ValueError("missing required column (one of: {})".format( + ", ".join(names))) + + +def fmt(number): + return "{:.15g}".format(number) + + +def main(): + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("historical_trace") + parser.add_argument("output_trace") + parser.add_argument("--start-time", required=True, + help="epoch seconds or ISO timestamp") + parser.add_argument("--workload-trace", + help="future simulation input; defaults to the historical trace") + parser.add_argument("--timezone", default="America/Los_Angeles", + help="timezone for ISO timestamps without an offset") + parser.add_argument("--total-nodes", type=int, + help="fail if initialization jobs exceed this capacity") + args = parser.parse_args() + + start = parse_time(args.start_time, args.timezone) + with open(args.historical_trace, newline="") as stream: + history = list(csv.DictReader(stream)) + + initial = [] + for index, row in enumerate(history): + begin = parse_time(value(row, "begin_time", "start_time"), args.timezone) + end = parse_time(value(row, "end_time"), args.timezone) + # A job that begins exactly at the boundary is running at that + # boundary and must be installed before ordinary arrivals at the same + # timestamp. Jobs ending at the boundary are already complete. + if begin <= start < end: + nodes = int(value(row, "num_nodes", "nodes", "nodes_requested")) + remaining = end - start + initial.append({ + "job_submit_time": fmt(start), + "num_nodes": str(nodes), + "time_limit": fmt(remaining), + "actual_run_time": fmt(remaining), + "initialization": "1", + "source_job_index": str(index), + }) + + required_nodes = sum(int(row["num_nodes"]) for row in initial) + if args.total_nodes is not None and required_nodes > args.total_nodes: + raise SystemExit( + "initial state requires {} nodes, exceeding --total-nodes={}".format( + required_nodes, args.total_nodes)) + + workload_path = args.workload_trace or args.historical_trace + with open(workload_path, newline="") as stream: + workload = list(csv.DictReader(stream)) + + future = [] + for index, row in enumerate(workload): + submit = parse_time(value(row, "job_submit_time", "submit_time"), + args.timezone) + if submit < start: + continue + nodes = int(value(row, "num_nodes", "nodes", "nodes_requested")) + limit = float(value(row, "time_limit", "requested_time", "wall_time")) + actual = None + for name in ("actual_run_time", "duration", "actual_duration", "run_time"): + if row.get(name, "") != "": + actual = float(row[name]) + break + if actual is None and row.get("begin_time", "") and row.get("end_time", ""): + actual = (parse_time(row["end_time"], args.timezone) - + parse_time(row["begin_time"], args.timezone)) + if actual is None: + actual = limit + future.append({ + "job_submit_time": fmt(submit), + "num_nodes": str(nodes), + "time_limit": fmt(limit), + "actual_run_time": fmt(actual), + "initialization": "0", + "source_job_index": str(index), + }) + + # Initialization rows precede ordinary arrivals at the same timestamp, so + # stable FCFS insertion establishes historical occupancy first. + rows = initial + future + rows.sort(key=lambda row: (float(row["job_submit_time"]), + 0 if row["initialization"] == "1" else 1, + int(row["source_job_index"]))) + fields = ["job_submit_time", "num_nodes", "time_limit", + "actual_run_time", "initialization", "source_job_index"] + with open(args.output_trace, "w", newline="") as stream: + writer = csv.DictWriter(stream, fieldnames=fields) + writer.writeheader() + writer.writerows(rows) + + print("initialization_jobs={}".format(len(initial))) + print("initialization_nodes={}".format(required_nodes)) + print("future_jobs={}".format(len(future))) + print("output={}".format(args.output_trace)) + + +if __name__ == "__main__": + try: + main() + except (OSError, ValueError) as error: + print("error: {}".format(error), file=sys.stderr) + sys.exit(1) diff --git a/src/params/sim_params.cpp b/src/params/sim_params.cpp index 6e68b85..0740031 100644 --- a/src/params/sim_params.cpp +++ b/src/params/sim_params.cpp @@ -10,8 +10,11 @@ */ #include "params/sim_params.hpp" +#include "trace/parse_utils.hpp" #include "utils/file.hpp" +#include #include +#include #include #include #include @@ -27,6 +30,8 @@ namespace dr_evt { static constexpr int OPT_TRACE_TYPE = 1000; static constexpr int OPT_JOB_FLUSH_INTERVAL = 1001; static constexpr int OPT_NUM_MAX_CANDIDATES = 1002; +static constexpr int OPT_CAPACITY_SCHEDULE = 1003; +static constexpr int OPT_SIM_START_TIME = 1004; /** @brief getopt short-option specification for the simulator CLI. */ #define OPTIONS "hi:j:n:o:s:t:b:p:q:Q:A:G:r:f:T:z:D:S:V:vc:R:MK:W:H:L:m:" @@ -37,6 +42,8 @@ static const struct option sim_longopts[] = { {"infile_list", required_argument, 0, 'L'}, {"max_jobs", required_argument, 0, 'j'}, {"total_nodes", required_argument, 0, 'n'}, + {"capacity_schedule", required_argument, 0, OPT_CAPACITY_SCHEDULE}, + {"sim_start_time", required_argument, 0, OPT_SIM_START_TIME}, {"outfile", required_argument, 0, 'o'}, {"seed", required_argument, 0, 's'}, {"max_time", required_argument, 0, 't'}, @@ -69,9 +76,8 @@ static const struct option sim_longopts[] = { Sim_Params::Sim_Params() : m_seed(0u), m_max_jobs(10u), m_max_time(dr_evt::max_sim_time), - m_is_jobs_set(false), m_is_time_set(false), - m_backfill_policy(BackfillPolicy::EASY), - m_num_max_candidates(1), + m_sim_start_time(0.0), m_is_jobs_set(false), m_is_time_set(false), + m_backfill_policy(BackfillPolicy::EASY), m_num_max_candidates(1), m_priority_policy(PriorityPolicy::FCFS), m_queue_impl(QueueImplementation::CIRCULAR), m_block_size(128), m_wait_queue_capacity(0), // 0 = size of job trace (never overflows) @@ -81,8 +87,8 @@ Sim_Params::Sim_Params() m_job_flush_interval(0), m_memory_pressure_fraction(0.0), m_resource_history_capacity(0), m_total_nodes(dr_evt::total_nodes), m_trace_type(TraceType::STANDARD), - m_trace_format("simple"), // Default to simple format - m_timestamp_format("iso"), // Default to ISO/human-readable timestamps + m_trace_format("simple"), // Default to simple format + m_timestamp_format("epoch"), // Retained timestamp compatibility value m_timezone("America/Los_Angeles"), // Default timezone m_run_time_mode(RunTimeMode::ACTUAL), // Default: jobs run actual_run_time // from trace (most realistic) @@ -95,6 +101,57 @@ Sim_Params::Sim_Params() void Sim_Params::getopt(int &argc, char **&argv) { int c; + std::string sim_start_time_arg; + auto apply_sim_start_time = [&]() { + if (sim_start_time_arg.empty()) { + return; + } + const char *old_tz_ptr = std::getenv("TZ"); + const bool had_old_tz = old_tz_ptr != nullptr; + const std::string old_tz = had_old_tz ? old_tz_ptr : ""; + try { + setenv("TZ", m_timezone.c_str(), 1); + tzset(); + size_t parsed_chars = 0; + bool parsed_numeric = false; + try { + const double numeric = std::stod(sim_start_time_arg, &parsed_chars); + if (parsed_chars == sim_start_time_arg.size()) { + m_sim_start_time = numeric; + parsed_numeric = true; + } + } catch (const std::exception &) { + // Fall through to ISO timestamp parsing. + } + if (!parsed_numeric) { + epoch_t parsed; + set_by(parsed, sim_start_time_arg); + m_sim_start_time = convert_epoch(parsed); + } + } catch (const std::exception &e) { + if (had_old_tz) { + setenv("TZ", old_tz.c_str(), 1); + } else { + unsetenv("TZ"); + } + tzset(); + std::cerr << "Error: invalid --sim_start_time: " << e.what() + << std::endl; + print_usage(argv[0], 1); + } + if (had_old_tz) { + setenv("TZ", old_tz.c_str(), 1); + } else { + unsetenv("TZ"); + } + tzset(); + if (!std::isfinite(m_sim_start_time) || m_sim_start_time < 0.0) { + std::cerr << "Error: --sim_start_time must be finite and nonnegative" + << std::endl; + print_usage(argv[0], 1); + } + sim_start_time_arg.clear(); + }; m_is_jobs_set = false; m_is_time_set = false; @@ -116,6 +173,13 @@ void Sim_Params::getopt(int &argc, char **&argv) { case 'n': /* --total_nodes */ m_total_nodes = static_cast(atoi(optarg)); break; + case OPT_CAPACITY_SCHEDULE: /* --capacity_schedule */ + m_capacity_schedule = std::string(optarg); + break; + case OPT_SIM_START_TIME: /* --sim_start_time */ + // Parse after all options so --timezone is order-independent. + sim_start_time_arg = optarg; + break; case 'o': /* --outfile */ m_outfile = std::string(optarg); break; @@ -295,7 +359,7 @@ void Sim_Params::getopt(int &argc, char **&argv) { { std::string format(optarg); if (format.empty()) { - m_timestamp_format = "iso"; + m_timestamp_format = "epoch"; } else if (format == "epoch" || format == "iso") { m_timestamp_format = format; } else { @@ -354,6 +418,9 @@ void Sim_Params::getopt(int &argc, char **&argv) { case 'c': /* --config */ { #if defined(DR_EVT_HAS_PROTOBUF) + // Preserve left-to-right precedence: a following config replaces any + // earlier command-line simulation start time. + apply_sim_start_time(); std::string config_file(optarg); try { read_proto_params(config_file, *this, m_verbose); @@ -377,6 +444,8 @@ void Sim_Params::getopt(int &argc, char **&argv) { } } + apply_sim_start_time(); + // --infile_list mode has no single positional trace file to require - // the file it names lists several instead. Otherwise, unchanged: // exactly one positional argument, which always wins over -i/--infile @@ -398,6 +467,12 @@ void Sim_Params::getopt(int &argc, char **&argv) { } set_outfile(m_outfile); + if (m_is_time_set && (!std::isfinite(m_max_time) || m_max_time < 0.0)) { + std::cerr << "Error: --max_time must be finite and nonnegative" + << std::endl; + print_usage(argv[0], 1); + } + if (!m_is_jobs_set && m_is_time_set) { m_max_jobs = std::numeric_limits::max(); } @@ -442,15 +517,31 @@ void Sim_Params::print_usage(const std::string exec, int code) { " -n, --total_nodes\n" " Specify total number of nodes in the system (default: 795).\n" "\n" + " --capacity_schedule FILENAME\n" + " CSV change points with columns time,total_nodes. Capacity\n" + " defaults to --total_nodes before the first row. Reductions\n" + " are non-preemptive: running jobs finish, and new starts\n" + " wait until they fit. Jobs are rejected only when they\n" + " exceed --total_nodes, not the scheduled capacity. A zero\n" + " value pauses new starts.\n" + "\n" + " --sim_start_time TIME\n" + " Set the global simulation start time as nonnegative epoch\n" + " seconds or an ISO timestamp. With replay-format\n" + " input, jobs that began before TIME seed live resource state\n" + " without entering output or job statistics. Their remaining\n" + " departures still release resources normally.\n" + "\n" " -o, --outfile\n" " Specify the output file name for simulation.\n" "\n" " -s, --seed\n" " Specify the seed for random number generator. Without this,\n" - " it will use a value dependent on the current system clock.\n" + " the deterministic default seed is 0.\n" "\n" " -t, --max_time\n" - " Specify the upper limit of simulation time to run.\n" + " Stop batch simulation after processing all events at this\n" + " nonnegative simulation timestamp.\n" "\n" " -b, --backfill_policy {easy|conservative|none}\n" " Backfilling policy (default: easy).\n" @@ -571,15 +662,15 @@ void Sim_Params::print_usage(const std::string exec, int code) { " lassen: 33-column LLNL Lassen format\n" "\n" " -T, --timestamp_format {epoch|iso}\n" - " Timestamp format in trace file (default: iso).\n" - " epoch: Unix epoch seconds (integers)\n" - " iso: ISO 8601 or human-readable timestamps\n" + " Compatibility setting (default: epoch). Input timestamps\n" + " are auto-detected as epoch seconds or calendar timestamps;\n" + " simulator output is numeric in either setting.\n" "\n" " -z, --timezone TIMEZONE\n" " Timezone for timestamp parsing (default: " "America/Los_Angeles).\n" " Examples: UTC, America/New_York, America/Los_Angeles\n" - " Only used when timestamp_format=iso\n" + " Used for calendar timestamps without an embedded offset.\n" "\n" " -r, --run_time_mode {actual|distribution|limit}\n" " How to determine the job's actual run time in simulation " @@ -653,9 +744,11 @@ void Sim_Params::print() const { msg += " - seed: " + to_string(m_seed) + "\n"; msg += " - max_jobs: " + to_string(m_max_jobs) + "\n"; msg += " - max_time: " + to_string(m_max_time) + "\n"; + msg += " - sim_start_time: " + to_string(m_sim_start_time) + "\n"; msg += " - infile: " + m_infile + "\n"; msg += " - outfile: " + m_outfile + "\n"; msg += " - total_nodes: " + to_string(m_total_nodes) + "\n"; + msg += " - capacity_schedule: " + m_capacity_schedule + "\n"; msg += " - num_max_candidates: " + to_string(m_num_max_candidates) + "\n"; msg += " - job_flush_interval: " + to_string(m_job_flush_interval) + "\n"; msg += " - is_jobs_set: " + string{m_is_jobs_set ? "true" : "false"} + "\n"; diff --git a/src/params/sim_params.hpp b/src/params/sim_params.hpp index d6e708e..25cfba6 100644 --- a/src/params/sim_params.hpp +++ b/src/params/sim_params.hpp @@ -97,6 +97,8 @@ class Sim_Params { dr_evt::num_jobs_t m_max_jobs; /// Maximum simulation time horizon. dr_evt::sim_time_t m_max_time; + /// Global simulation start timestamp. Zero preserves the traditional run. + dr_evt::sim_time_t m_sim_start_time; /// Primary input trace filename. std::string m_infile; @@ -147,11 +149,13 @@ class Sim_Params { size_t m_resource_history_capacity; /// Total nodes available to the simulated scheduler. num_nodes_t m_total_nodes; + /// Optional CSV of time-varying capacity change points. + std::string m_capacity_schedule; /// Job-trace data model: standard or experimental Pcon. TraceType m_trace_type; /// Input trace format name, such as "simple" or "lassen". std::string m_trace_format; - /// Input timestamp format name, such as "epoch" or "iso". + /// Retained epoch/iso compatibility value; file input is auto-detected. std::string m_timestamp_format; /// IANA timezone used for timestamps without an embedded offset. std::string m_timezone; diff --git a/src/proto/dr_evt_params.cpp b/src/proto/dr_evt_params.cpp index e295a1e..7d5d51b 100644 --- a/src/proto/dr_evt_params.cpp +++ b/src/proto/dr_evt_params.cpp @@ -12,6 +12,7 @@ #include "proto/dr_evt_params.hpp" #include "proto/utils.hpp" #include "utils/file.hpp" +#include #include #include #include @@ -36,6 +37,17 @@ set_sim_options(const dr_evt_proto::DR_EVT_Params::Simulation_Params &cfg, sp.m_max_jobs = cfg.max_jobs(); sp.m_max_time = cfg.max_time(); + if (!std::isfinite(sp.m_max_time) || sp.m_max_time < 0.0) { + throw std::runtime_error( + "Invalid max_time in protobuf config (must be finite and " + "nonnegative)"); + } + if (!std::isfinite(cfg.sim_start_time()) || cfg.sim_start_time() < 0.0) { + throw std::runtime_error( + "Invalid sim_start_time in protobuf config (must be finite and " + "nonnegative)"); + } + sp.m_sim_start_time = cfg.sim_start_time(); sp.m_is_jobs_set = (sp.m_max_jobs > 0u); sp.m_is_time_set = (sp.m_max_time > 0.0); @@ -63,6 +75,9 @@ set_sim_options(const dr_evt_proto::DR_EVT_Params::Simulation_Params &cfg, if (cfg.total_nodes() > 0) { sp.m_total_nodes = cfg.total_nodes(); } + if (!cfg.capacity_schedule().empty()) { + sp.m_capacity_schedule = cfg.capacity_schedule(); + } // Candidate limit for the callback-driven custom scheduler. // Proto3's zero value preserves Sim_Params' default of one. @@ -133,7 +148,7 @@ set_sim_options(const dr_evt_proto::DR_EVT_Params::Simulation_Params &cfg, sp.m_trace_format = "simple"; } - // Timestamp format (options: "epoch" or "iso", default: "iso") + // Retained timestamp compatibility value (default: "epoch") if (!cfg.timestamp_format().empty()) { std::string format = cfg.timestamp_format(); if (format == "epoch" || format == "iso") { @@ -143,7 +158,7 @@ set_sim_options(const dr_evt_proto::DR_EVT_Params::Simulation_Params &cfg, format); } } else { - sp.m_timestamp_format = "iso"; + sp.m_timestamp_format = "epoch"; } // Timezone (examples: "UTC", "America/Los_Angeles", "America/New_York", diff --git a/src/proto/dr_evt_params.proto b/src/proto/dr_evt_params.proto index ea32b52..1a96168 100644 --- a/src/proto/dr_evt_params.proto +++ b/src/proto/dr_evt_params.proto @@ -11,8 +11,7 @@ package dr_evt_proto; message DR_EVT_Params { message Simulation_Params { - // The seed for random number generator. Without this, - // it will use a value dependent on the current system clock. + // Random-number seed (default: deterministic zero). uint32 seed = 1; // The maximum number of jobs to run. @@ -44,7 +43,7 @@ message DR_EVT_Params { // Trace format parameters string trace_type = 12; // "standard" or "pcon" (default: "standard") string trace_format = 13; // "simple" or "lassen" (default: "simple") - string timestamp_format = 14; // "epoch" or "iso" (default: "iso") + string timestamp_format = 14; // retained "epoch"/"iso" compatibility value (default: "epoch") string timezone = 15; // e.g., "UTC", "America/Los_Angeles" (default: "America/Los_Angeles") // Run time determination parameters @@ -98,6 +97,14 @@ message DR_EVT_Params { // Maximum feasible jobs passed to an experimental external backfill // selector in one decision. 0 keeps Sim_Params' default (1). uint64 num_max_candidates = 29; + + // Optional CSV with time,total_nodes change points. Values may not + // exceed total_nodes; reductions do not preempt running jobs. + string capacity_schedule = 30; + + // Global simulation start time. A positive value enables replay-based + // warm start; replay jobs already running at the boundary seed resources. + double sim_start_time = 31; } message Tracing_Params { diff --git a/src/proto/dr_evt_server.cpp b/src/proto/dr_evt_server.cpp index 7c9fc95..97d0aff 100644 --- a/src/proto/dr_evt_server.cpp +++ b/src/proto/dr_evt_server.cpp @@ -15,6 +15,7 @@ #include #include #include +#include #include #include #include @@ -195,6 +196,12 @@ class SimulationServiceImpl final : public SimulationService::Service { if (!r.infile().empty()) sp.m_infile = r.infile(); sp.m_msec_output = r.msec_output(); + if (!std::isfinite(r.sim_start_time()) || + r.sim_start_time() < 0.0) { + throw std::runtime_error( + "sim_start_time must be finite and nonnegative"); + } + sp.m_sim_start_time = r.sim_start_time(); if (r.backfill_policy().empty()) sp.m_backfill_policy = dr_evt::BackfillPolicy::EASY; @@ -322,6 +329,12 @@ class SimulationServiceImpl final : public SimulationService::Service { resp.mutable_advance_to(); break; } + case ClientMessage::kRun: { + require_init(sim); + sim->run(); + resp.mutable_run(); + break; + } case ClientMessage::kRunUntilExclusive: { require_init(sim); sim->run_until_exclusive(req.run_until_exclusive().target_time()); diff --git a/src/proto/dr_evt_service.proto b/src/proto/dr_evt_service.proto index 121fdbc..5e38a8d 100644 --- a/src/proto/dr_evt_service.proto +++ b/src/proto/dr_evt_service.proto @@ -13,7 +13,7 @@ syntax = "proto3"; package dr_evt_grpc; // Single bidirectional-streaming RPC exposing DR_EVT's C++ streaming API -// (append_job, advance_to, run_until_exclusive, and the monitoring queries) +// (batch run, append_job, advance_to, run_until_exclusive, and monitoring) // over the network. AppendJobRequest is the genuine streaming case: it adds // a previously unknown job and immediately enqueues it. A Session() // can host successive server-side Simulation instances: each InitRequest @@ -45,6 +45,7 @@ message ClientMessage { FinishSimulationRequest finish_simulation = 15; GetBackfillWindowRequest get_backfill_window = 16; GetCurrentUtilizationRequest get_current_utilization = 17; + RunRequest run = 18; } } @@ -54,7 +55,7 @@ message ClientMessage { message InitRequest { uint32 total_nodes = 1; string trace_format = 2; // "simple" or "lassen" (default: "simple") - string timestamp_format = 3; // "epoch" or "iso" + string timestamp_format = 3; // retained "epoch"/"iso" compatibility value string timezone = 4; string backfill_policy = 5; // "easy", "conservative", or "none" string priority_policy = 6; // "fcfs", "fcfs_alt", "sjf", or "ljf" @@ -85,8 +86,16 @@ message InitRequest { // wait_queue_capacity. Ignored unless queue_impl is "circular". // Empty string keeps Sim_Params' own default (grow). string wait_queue_overflow = 14; // "abort" or "grow" + + // Global simulation start time. A positive value selects native replay + // warm-start execution for RunRequest; zero preserves ordinary execution. + double sim_start_time = 15; } +// Runs the configured input trace to completion. Unlike AdvanceToRequest, +// this invokes Simulation::run(), including replay and warm-start setup. +message RunRequest {} + // Declares this simulation complete. The server processes all remaining // submitted work, writes its reports, returns final statistics, then releases // the simulation so this same Session stream may be initialized again. @@ -178,6 +187,7 @@ message ServerMessage { FinishSimulationResponse finish_simulation = 16; GetBackfillWindowResponse get_backfill_window = 17; GetCurrentUtilizationResponse get_current_utilization = 18; + RunResponse run = 19; } } @@ -213,6 +223,8 @@ message AppendJobsResponse { message AdvanceToResponse {} +message RunResponse {} + message RunUntilExclusiveResponse {} message GetCurrentTimeResponse { diff --git a/src/sim/CMakeLists.txt b/src/sim/CMakeLists.txt index 8e77ac9..dcd9a05 100644 --- a/src/sim/CMakeLists.txt +++ b/src/sim/CMakeLists.txt @@ -3,6 +3,7 @@ set_full_path(THIS_DIR_HEADERS job_submit_common.hpp job_submit_model.hpp multi_platform_runs.hpp + capacity_schedule.hpp scheduler_policies.hpp schedule_windows.hpp scheduler_base.hpp @@ -19,6 +20,7 @@ set_full_path(THIS_DIR_HEADERS ) set_full_path(THIS_DIR_SOURCES + capacity_schedule.cpp schedule_windows.cpp scheduler_base.cpp scheduler_fcfs.cpp diff --git a/src/sim/capacity_schedule.cpp b/src/sim/capacity_schedule.cpp new file mode 100644 index 0000000..308b288 --- /dev/null +++ b/src/sim/capacity_schedule.cpp @@ -0,0 +1,101 @@ +/****************************************************************************** + * Copyright 2023 Lawrence Livermore National Security, LLC * + * See the top-level LICENSE file for details. * + * * + * SPDX-License-Identifier: MIT * + ******************************************************************************/ + +#include "sim/capacity_schedule.hpp" +#include "trace/parse_utils.hpp" +#include +#include +#include + +namespace dr_evt { + +std::vector +load_capacity_schedule(const std::string &filename, + num_nodes_t configured_maximum) { + if (filename.empty()) { + return {}; + } + + std::ifstream input(filename); + if (!input) { + throw std::runtime_error("Failed to open capacity schedule: " + filename); + } + + std::string line; + if (!std::getline(input, line)) { + throw std::invalid_argument("Capacity schedule is empty: " + filename); + } + + std::unordered_map columns; + const auto header = comma_separate(line); + for (size_t i = 0; i < header.size(); ++i) { + columns.emplace(trim(line.substr(header[i].first, header[i].second)), i); + } + const auto time_it = columns.find("time"); + const auto nodes_it = columns.find("total_nodes"); + if (time_it == columns.end() || nodes_it == columns.end()) { + throw std::invalid_argument( + "Capacity schedule requires CSV columns 'time,total_nodes'"); + } + + std::vector changes; + TimestampEncoding timestamp_encoding = TimestampEncoding::EPOCH; + bool timestamp_encoding_detected = false; + size_t row = 1; + while (std::getline(input, line)) { + ++row; + if (trim(line).empty()) { + continue; + } + const auto fields = comma_separate(line); + if (time_it->second >= fields.size() || nodes_it->second >= fields.size()) { + throw std::invalid_argument("Capacity schedule row " + + std::to_string(row) + + " has fewer fields than its header"); + } + + epoch_t parsed_time; + unsigned parsed_nodes = 0; + try { + const std::string time_value = + trim(line.substr(fields[time_it->second].first, + fields[time_it->second].second)); + if (!timestamp_encoding_detected) { + timestamp_encoding = detect_timestamp_encoding(time_value); + timestamp_encoding_detected = true; + } + set_by(parsed_time, time_value, timestamp_encoding); + set_by(parsed_nodes, trim(line.substr(fields[nodes_it->second].first, + fields[nodes_it->second].second))); + } catch (const std::exception &e) { + throw std::invalid_argument("Invalid capacity schedule row " + + std::to_string(row) + ": " + e.what()); + } + + const sim_time_t time = convert_epoch(parsed_time); + if (!changes.empty() && time <= changes.back().time) { + throw std::invalid_argument( + "Capacity schedule times must be strictly increasing (row " + + std::to_string(row) + ")"); + } + if (parsed_nodes > configured_maximum) { + throw std::invalid_argument( + "Capacity schedule row " + std::to_string(row) + " requests " + + std::to_string(parsed_nodes) + " nodes, exceeding --total_nodes=" + + std::to_string(configured_maximum)); + } + changes.push_back({time, static_cast(parsed_nodes)}); + } + + if (changes.empty()) { + throw std::invalid_argument("Capacity schedule contains no data rows: " + + filename); + } + return changes; +} + +} // namespace dr_evt diff --git a/src/sim/capacity_schedule.hpp b/src/sim/capacity_schedule.hpp new file mode 100644 index 0000000..58953af --- /dev/null +++ b/src/sim/capacity_schedule.hpp @@ -0,0 +1,41 @@ +/****************************************************************************** + * Copyright 2023 Lawrence Livermore National Security, LLC * + * See the top-level LICENSE file for details. * + * * + * SPDX-License-Identifier: MIT * + ******************************************************************************/ + +/** @file sim/capacity_schedule.hpp + * @brief Time-varying machine-capacity input. + */ + +#ifndef DR_EVT_SIM_CAPACITY_SCHEDULE_HPP +#define DR_EVT_SIM_CAPACITY_SCHEDULE_HPP + +#include "dr_evt_types.hpp" +#include +#include + +namespace dr_evt { + +/** One capacity change, effective at `time` until the next change. */ +struct Capacity_Change { + sim_time_t time; + num_nodes_t total_nodes; +}; + +/** + * Load a CSV containing `time,total_nodes` change points. + * + * The first data row selects epoch or calendar encoding for the whole file. + * Times accept the same spellings as simple job traces. Rows must be strictly + * increasing in time, and capacities may not exceed the configured maximum. + * An empty filename returns an empty schedule. + */ +std::vector +load_capacity_schedule(const std::string &filename, + num_nodes_t configured_maximum); + +} // namespace dr_evt + +#endif // DR_EVT_SIM_CAPACITY_SCHEDULE_HPP diff --git a/src/sim/scheduler_fcfs_custom.cpp b/src/sim/scheduler_fcfs_custom.cpp index 60a264e..c2f8098 100644 --- a/src/sim/scheduler_fcfs_custom.cpp +++ b/src/sim/scheduler_fcfs_custom.cpp @@ -27,7 +27,8 @@ CustomFCFSScheduler::CustomFCFSScheduler( m_num_max_candidates(num_max_candidates), m_job_cost_function(std::move(cost_function)), m_backfill_selector(std::move(selector)), m_resource_area(0.0), - m_resource_area_time(0.0), m_accounted_available_nodes(total_nodes) { + m_resource_area_time(0.0), m_resource_area_start(0.0), + m_accounted_available_nodes(total_nodes) { if (m_num_max_candidates == 0) { throw std::invalid_argument( "CustomFCFSScheduler requires num_max_candidates > 0"); @@ -69,9 +70,23 @@ void CustomFCFSScheduler::commit_available_nodes(num_nodes_t available_nodes) { } void CustomFCFSScheduler::reset_resource_accounting() { + reset_resource_accounting(0.0, m_total_nodes); +} + +void CustomFCFSScheduler::reset_resource_accounting( + sim_time_t start_time, num_nodes_t available_nodes) { + if (!std::isfinite(start_time) || start_time < 0.0) { + throw std::invalid_argument( + "resource accounting start time must be finite and nonnegative"); + } + if (available_nodes > m_total_nodes) { + throw std::invalid_argument( + "resource accounting available nodes exceed total capacity"); + } m_resource_area = 0.0; - m_resource_area_time = 0.0; - m_accounted_available_nodes = m_total_nodes; + m_resource_area_time = start_time; + m_resource_area_start = start_time; + m_accounted_available_nodes = available_nodes; } tdiff_t @@ -87,11 +102,12 @@ CustomFCFSScheduler::resource_area_through(sim_time_t through_time) const { } double CustomFCFSScheduler::utilization_through(sim_time_t through_time) const { - if (m_total_nodes == 0 || through_time <= 0.0) { + const sim_time_t duration = through_time - m_resource_area_start; + if (m_total_nodes == 0 || duration <= 0.0) { return 0.0; } return resource_area_through(through_time) / - (static_cast(m_total_nodes) * through_time); + (static_cast(m_total_nodes) * duration); } tdiff_t diff --git a/src/sim/scheduler_fcfs_custom.hpp b/src/sim/scheduler_fcfs_custom.hpp index 0d618cb..35ade3d 100644 --- a/src/sim/scheduler_fcfs_custom.hpp +++ b/src/sim/scheduler_fcfs_custom.hpp @@ -70,6 +70,8 @@ class CustomFCFSScheduler : public SchedulerBase { tdiff_t m_resource_area; /// Last simulation-time boundary incorporated into m_resource_area. sim_time_t m_resource_area_time; + /// Boundary from which the current resource-area interval is measured. + sim_time_t m_resource_area_start; /// Free nodes after the most recently settled scheduling cycle. num_nodes_t m_accounted_available_nodes; @@ -79,9 +81,13 @@ class CustomFCFSScheduler : public SchedulerBase { /** Store free capacity after all scheduling at the current time settles. */ void commit_available_nodes(num_nodes_t available_nodes); - /** Reset resource accounting before a new simulation run. */ + /** Reset resource accounting for a traditional empty time-zero start. */ void reset_resource_accounting(); + /** Reset resource accounting at a populated simulation boundary. */ + void reset_resource_accounting(sim_time_t start_time, + num_nodes_t available_nodes); + /** Return allocated-node area through a finite snapshot time. */ tdiff_t resource_area_through(sim_time_t through_time) const; diff --git a/src/sim/sim.cpp b/src/sim/sim.cpp index cb57e26..4d04ba2 100644 --- a/src/sim/sim.cpp +++ b/src/sim/sim.cpp @@ -32,9 +32,17 @@ BasicSimulation::BasicSimulation(const Sim_Params ¶ms) params.m_total_nodes, m_trace.data().size(), params.m_backfill_policy, params.m_priority_policy, params.m_queue_impl, params.m_block_size, params.m_wait_queue_capacity, params.m_wait_queue_overflow)), - m_custom_scheduler(nullptr), m_current_time(0.0), m_jobs_completed(0), - m_jobs_submitted(0), m_rng(params.m_seed), m_queue_length_sum(0), - m_queue_length_samples(0), m_queue_length_peak(0) {} + m_custom_scheduler(nullptr), m_current_time(0.0), + m_job_rejection_capacity(params.m_total_nodes), + m_capacity_changes(load_capacity_schedule(params.m_capacity_schedule, + params.m_total_nodes)), + m_next_capacity_change(0), m_current_capacity(params.m_total_nodes), + m_capacity_area(0.0), m_capacity_area_time(0.0), m_jobs_completed(0), + m_jobs_submitted(0), m_pre_start_jobs(0), m_warm_resource_area(0.0), + m_warm_resource_end(0.0), m_rng(params.m_seed), m_queue_length_sum(0), + m_queue_length_samples(0), m_queue_length_peak(0) { + reset_capacity_schedule(); +} template BasicSimulation::BasicSimulation(const Sim_Params ¶ms, @@ -48,15 +56,33 @@ BasicSimulation::BasicSimulation(const Sim_Params ¶ms, std::move(selector), params.m_wait_queue_capacity, params.m_wait_queue_overflow)), m_custom_scheduler(static_cast(m_scheduler.get())), - m_current_time(0.0), m_jobs_completed(0), m_jobs_submitted(0), - m_rng(params.m_seed), m_queue_length_sum(0), m_queue_length_samples(0), - m_queue_length_peak(0) {} + m_current_time(0.0), m_job_rejection_capacity(params.m_total_nodes), + m_capacity_changes(load_capacity_schedule(params.m_capacity_schedule, + params.m_total_nodes)), + m_next_capacity_change(0), m_current_capacity(params.m_total_nodes), + m_capacity_area(0.0), m_capacity_area_time(0.0), m_jobs_completed(0), + m_jobs_submitted(0), m_pre_start_jobs(0), m_warm_resource_area(0.0), + m_warm_resource_end(0.0), m_rng(params.m_seed), m_queue_length_sum(0), + m_queue_length_samples(0), m_queue_length_peak(0) { + reset_capacity_schedule(); +} template void BasicSimulation::run() { if (m_params.m_verbose) { std::cout << "Starting simulation..." << std::endl; } + if (m_params.m_is_time_set && + (!std::isfinite(m_params.m_max_time) || m_params.m_max_time < 0.0)) { + throw std::invalid_argument( + "--max_time must be finite and nonnegative"); + } + if (m_params.m_is_time_set && + m_params.m_max_time < m_params.m_sim_start_time) { + throw std::invalid_argument( + "--max_time must be greater than or equal to --sim_start_time"); + } + // Must happen before initialize_trace()/run_progressive() (either // resolves m_data's capacity from whatever's set here) - unlike // resource-history's capacity, which is only needed once recording @@ -67,6 +93,11 @@ template void BasicSimulation::run() { m_trace.set_memory_pressure_fraction(m_params.m_memory_pressure_fraction); if (!m_params.m_infile_list.empty()) { + if (m_params.m_sim_start_time != 0.0) { + throw std::runtime_error( + "--sim_start_time is not supported with --infile_list; warm-start " + "classification requires one replay trace loaded at the boundary"); + } // Progressive loading: REPLAY-format input isn't supported here // - REPLAY bypasses the scheduler entirely (begin_time/end_time // already fixed in the trace), so there's no notion of "submit @@ -100,37 +131,70 @@ template void BasicSimulation::run() { std::to_string(m_params.m_total_nodes) + " nodes\n"; } - // Open the resource-trace file early (if one was requested) so - // reclaiming from the now-bounded circular buffer can flush to it - // incrementally during the run, rather than only at the very end. m_trace.set_resource_history_capacity(m_params.m_resource_history_capacity); - m_trace.start_resource_trace(m_params.get_resource_trace(), - m_params.m_total_nodes, m_params.m_msec_output); - - // Same reasoning, for job records: m_data can now reclaim too, so the - // output file needs to be open before that ever happens, not only - // at the very end. - m_trace.start_simulated_trace(m_params.get_outfile(), m_params.m_msec_output); - if (m_trace.dcols().get_trace_mode() == TraceMode::REPLAY) { - // Replay-format input (begin_time/end_time present): don't consult - // the scheduler at all - reuse the same bypass logic the standalone - // tracer binary uses, driven into Trace's own owned context so the rest - // of this class (write_simulated_trace(), write_resource_trace()) - // sees the result exactly as if the scheduler had run. - m_trace.run_job_trace(); + if (m_params.m_sim_start_time != 0.0) { + run_warm_start(); } else { - // Batch mode: Submit all jobs upfront, then advance to infinity - // This uses the streaming API internally - for (num_jobs_t i = 0; i < m_trace.data().size(); ++i) { - const auto &job = m_trace.job_at(i); - sim_time_t submit_time = convert_epoch(job.get_submit_time()); - submit_job(i, submit_time); - } + // Open outputs before processing so bounded buffers can flush + // incrementally. Warm-start opens them later, after discarding pre-boundary + // samples and installing its nonzero baseline. + m_trace.start_resource_trace(m_params.get_resource_trace(), + m_params.m_total_nodes, + m_params.m_msec_output); + m_trace.start_simulated_trace(m_params.get_outfile(), + m_params.m_msec_output); - // Batch mode: advance to infinity to process all jobs - // The loop will exit when both wait_queue and event_queue are empty - advance_to(std::numeric_limits::max()); + if (m_trace.dcols().get_trace_mode() == TraceMode::REPLAY) { + // Replay-format input (begin_time/end_time present): don't consult + // the scheduler at all - reuse the same bypass logic the standalone + // tracer binary uses, driven into Trace's own owned context so the rest + // of this class (write_simulated_trace(), write_resource_trace()) + // sees the result exactly as if the scheduler had run. + const sim_time_t run_limit = + m_params.m_is_time_set + ? m_params.m_max_time + : std::numeric_limits::max(); + sim_time_t replay_end = m_current_time; + job_no_t replay_job_no = + static_cast(m_trace.num_reclaimed()); + for (const auto &job : m_trace.data()) { + const sim_time_t submit = + convert_epoch(job.get_submit_time()); + const sim_time_t begin = + convert_epoch(job.get_begin_time()); + const sim_time_t end = convert_epoch(job.get_end_time()); + if (submit <= run_limit) { + replay_end = std::max(replay_end, end); + if (end <= run_limit) { + ++m_jobs_completed; + } else if (m_params.m_is_time_set && begin <= run_limit) { + m_running_jobs[replay_job_no] = { + begin, static_cast(end - begin), job.get_num_nodes()}; + } + } + ++replay_job_no; + } + m_trace.run_job_trace(std::string(), m_params.m_total_nodes, run_limit); + m_current_time = m_params.m_is_time_set ? run_limit : replay_end; + } else { + // Batch mode: Submit all jobs upfront, then advance to infinity + // This uses the streaming API internally + for (num_jobs_t i = 0; i < m_trace.data().size(); ++i) { + const auto &job = m_trace.job_at(i); + sim_time_t submit_time = + convert_epoch(job.get_submit_time()); + submit_job(i, submit_time); + } + + // A configured maximum is an inclusive event-time boundary. Without + // one, use the internal drain sentinel and stop at the last real event. + const sim_time_t run_limit = + m_params.m_is_time_set + ? m_params.m_max_time + : std::numeric_limits::max(); + advance_to(run_limit); + } } // m_jobs_completed is tracked incrementally during the run itself @@ -149,8 +213,13 @@ template void BasicSimulation::run() { template void BasicSimulation::print_stats(std::ostream &os) const { os << "=== Simulation Statistics ===" << std::endl; - os << "Total jobs: " << (m_trace.data().size() + m_trace.num_reclaimed()) - << std::endl; + os << "Total jobs: "; + if (m_params.m_sim_start_time != 0.0) { + os << m_jobs_submitted; + } else { + os << (m_trace.data().size() + m_trace.num_reclaimed()); + } + os << std::endl; os << "Jobs submitted: " << m_jobs_submitted << std::endl; // m_trace.completed_count() (populated via write_job_line(), called // both at reclaim time and by write_simulated_trace()'s final @@ -231,8 +300,12 @@ num_jobs_t BasicSimulation::initialize_trace(num_jobs_t max_jobs) { } m_current_time = 0.0; + reset_capacity_schedule(); m_jobs_submitted = 0; m_jobs_completed = 0; + m_pre_start_jobs = 0; + m_warm_resource_area = 0.0; + m_warm_resource_end = 0.0; if (m_custom_scheduler != nullptr) { m_custom_scheduler->reset_resource_accounting(); } @@ -240,6 +313,119 @@ num_jobs_t BasicSimulation::initialize_trace(num_jobs_t max_jobs) { return static_cast(m_trace.data().size()); } +template +void BasicSimulation::run_warm_start() { + const sim_time_t sim_start_time = m_params.m_sim_start_time; + const sim_time_t run_limit = + m_params.m_is_time_set + ? m_params.m_max_time + : std::numeric_limits::max(); + if (m_trace.dcols().get_trace_mode() != TraceMode::REPLAY) { + throw std::runtime_error( + "--sim_start_time requires replay-format input with begin_time and " + "end_time columns so jobs already running at the boundary can be " + "identified"); + } + + const job_no_t first_job = static_cast(m_trace.num_reclaimed()); + const job_no_t jobs_end = + static_cast(m_trace.num_reclaimed() + m_trace.data().size()); + m_pre_start_jobs = 0; + m_warm_resource_area = 0.0; + m_warm_resource_end = sim_start_time; + + for (job_no_t job_no = first_job; job_no < jobs_end; ++job_no) { + auto &job = m_trace.job_at(job_no); + const sim_time_t begin = convert_epoch(job.get_begin_time()); + const sim_time_t end = convert_epoch(job.get_end_time()); + const sim_time_t submit = convert_epoch(job.get_submit_time()); + + if (begin < sim_start_time) { + // Historical starts bypass the scheduler. Replay establishes both live + // allocation and policy-specific state (notably Pcon) exactly once. + m_trace.enqueue_replay_job(job_no); + ++m_pre_start_jobs; + if (end > sim_start_time) { + // Its departure is fixed historical state, so reservations should + // use that known end rather than a possibly stale original limit. + m_running_jobs[job_no] = {begin, job.get_actual_run_time(), + job.get_num_nodes()}; + m_warm_resource_area += + static_cast(job.get_num_nodes()) * (end - sim_start_time); + m_warm_resource_end = std::max(m_warm_resource_end, end); + } + } else if (submit >= sim_start_time) { + // Let the normal scheduler replace the historical begin/end times in + // the counterfactual run, and select this simulated job's duration by + // the same run-time policy used in an ordinary simulation. Warmup jobs + // bypass this branch and retain their recorded timing. + job.prepare_for_resimulation(); + determine_one_job_run_time(job); + } else { + // This job was waiting before the boundary. Reconstructing an inherited + // wait queue is a separate policy decision, so it is intentionally not + // admitted into either the seed state or the post-boundary workload. + job.suppress_output(); + } + } + + // Replay only the history required to establish state at t. Old departures + // are suppressed before Trace can run any output/reclamation hook. + while (!m_trace.pending_events().empty()) { + const auto event = *m_trace.pending_events().begin(); + const sim_time_t event_time = convert_epoch(event.get_time()); + if (event_time > sim_start_time) { + break; + } + if (event.is_departure()) { + m_trace.job_at(event.get_job_idx()).suppress_output(); + --m_pre_start_jobs; + } + m_trace.process_single_event(); + } + + m_current_time = sim_start_time; + apply_capacity_changes(sim_start_time); + reset_capacity_accounting(sim_start_time); + + // This is the accounting/output boundary: retain occupancy and Pcon state, + // discard every pre-t sample, and establish a baseline at t. + m_trace.reset_resource_recording(sim_start_time); + m_trace.start_resource_trace(m_params.get_resource_trace(), + m_params.m_total_nodes, m_params.m_msec_output); + m_trace.start_simulated_trace(m_params.get_outfile(), m_params.m_msec_output); + if (m_custom_scheduler != nullptr) { + m_custom_scheduler->reset_resource_accounting(sim_start_time, + get_available_nodes()); + } + + // No auxiliary job list is retained: prepared post-boundary records keep + // their real submission time, while every excluded/finished historical + // record carries the existing unscheduled sentinel. + for (job_no_t job_no = first_job; job_no < jobs_end; ++job_no) { + const auto submit_epoch = m_trace.job_at(job_no).get_submit_time(); + if (submit_epoch != Job_Record::unscheduled_sentinel()) { + const sim_time_t submit = convert_epoch(submit_epoch); + if (submit >= sim_start_time) { + submit_job(job_no, submit); + } + } + } + + if (m_pre_start_jobs != 0) { + if (m_custom_scheduler != nullptr) { + advance_to_impl(run_limit, m_custom_scheduler); + } else { + advance_to_impl(run_limit, nullptr); + } + } + + // The ordinary stage has no historical-job test in its compiled loop. If + // the configured horizon ended during warm-up, the current time is already + // at run_limit and this call is an inexpensive no-op. + advance_to(run_limit); +} + template void BasicSimulation::determine_one_job_run_time(Job_Record &job) { // Scheduler uses time_limit as the best estimator for planning (realistic @@ -292,8 +478,12 @@ void BasicSimulation::run_progressive() { // upfront; each file gets loaded as the driving loop below reaches it. m_trace.data().clear(); m_current_time = 0.0; + reset_capacity_schedule(); m_jobs_submitted = 0; m_jobs_completed = 0; + m_pre_start_jobs = 0; + m_warm_resource_area = 0.0; + m_warm_resource_end = 0.0; if (m_custom_scheduler != nullptr) { m_custom_scheduler->reset_resource_accounting(); } @@ -352,13 +542,23 @@ void BasicSimulation::run_progressive() { // last submit_time as the target instead of infinity. sim_time_t last_submit_time = convert_epoch( m_trace.job_at(job_nos.back()).get_submit_time()); - advance_to(last_submit_time); + const sim_time_t run_limit = + m_params.m_is_time_set ? m_params.m_max_time : last_submit_time; + advance_to(std::min(last_submit_time, run_limit)); + if (m_params.m_is_time_set && last_submit_time >= m_params.m_max_time) { + break; + } } - // Drain whatever's still running after the last file - same - // "advance to infinity, loop exits once wait_queue and event_queue - // are both empty" postcondition single-file batch mode relies on. - advance_to(std::numeric_limits::max()); + // Drain whatever is still running, or stop at the configured inclusive + // time boundary. + const sim_time_t run_limit = + m_params.m_is_time_set + ? m_params.m_max_time + : std::numeric_limits::max(); + if (m_current_time < run_limit) { + advance_to(run_limit); + } } template @@ -412,7 +612,12 @@ tdiff_t BasicSimulation::sample_run_time(tdiff_t time_limit, template void BasicSimulation::write_simulated_trace() { - m_trace.write_simulated_trace(m_params.get_outfile(), m_params.m_msec_output); + const sim_time_t completed_through = + m_params.m_is_time_set + ? m_params.m_max_time + : std::numeric_limits::max(); + m_trace.write_simulated_trace(m_params.get_outfile(), m_params.m_msec_output, + completed_through); if (m_params.m_verbose && !m_params.get_outfile().empty()) { std::cout << "Simulated trace written to: " << m_params.get_outfile() << std::endl; @@ -556,7 +761,7 @@ void BasicSimulation::submit_job(job_no_t job_idx, tdiff_t run_time_estimate = job.get_limit_time(); num_nodes_t nodes = job.get_num_nodes(); - if (nodes > m_params.m_total_nodes) { + if (nodes > m_job_rejection_capacity) { // This job can never be scheduled, regardless of how long the // simulation runs - total_nodes is fixed for the whole run, so // no future state ever frees up enough capacity. Reject it here, @@ -572,7 +777,7 @@ void BasicSimulation::submit_job(job_no_t job_idx, // forever, since end_time never resolves for it either. job.set_submit_time(Job_Record::unscheduled_sentinel()); std::cerr << "Job " << job_idx << " rejected: requests " << nodes - << " nodes, exceeds total_nodes (" << m_params.m_total_nodes + << " nodes, exceeds total_nodes (" << m_job_rejection_capacity << "); this job can never be scheduled." << std::endl; return; } @@ -618,17 +823,75 @@ void BasicSimulation::record_queue_arrivals( m_pending_queue_arrivals.erase(arrivals); } +template +void BasicSimulation::reset_capacity_schedule() { + m_next_capacity_change = 0; + m_current_capacity = m_params.m_total_nodes; + m_trace.reset_resource_capacity(m_params.m_total_nodes); + reset_capacity_accounting(m_current_time); +} + +template +void BasicSimulation::reset_capacity_accounting( + sim_time_t start_time) { + m_capacity_area = 0.0; + m_capacity_area_time = start_time; +} + +template +void BasicSimulation::advance_capacity_accounting_to( + sim_time_t current_time) { + if (!std::isfinite(current_time)) { + return; + } + if (current_time < m_capacity_area_time) { + throw std::logic_error("capacity accounting cannot move backward in time"); + } + const num_nodes_t effective_capacity = + std::max(m_current_capacity, get_nodes_in_use()); + m_capacity_area += static_cast(effective_capacity) * + (current_time - m_capacity_area_time); + m_capacity_area_time = current_time; +} + +template +tdiff_t BasicSimulation::capacity_area_through( + sim_time_t through_time) const { + tdiff_t area = m_capacity_area; + if (std::isfinite(through_time) && through_time > m_capacity_area_time) { + const num_nodes_t effective_capacity = + std::max(m_current_capacity, get_nodes_in_use()); + area += static_cast(effective_capacity) * + (through_time - m_capacity_area_time); + } + return area; +} + +template +bool BasicSimulation::apply_capacity_changes( + sim_time_t current_time) { + bool changed = false; + while (m_next_capacity_change < m_capacity_changes.size() && + m_capacity_changes[m_next_capacity_change].time <= current_time) { + const auto &change = m_capacity_changes[m_next_capacity_change++]; + m_current_capacity = change.total_nodes; + m_trace.set_resource_capacity(change.time, change.total_nodes); + changed = true; + } + return changed; +} + template void BasicSimulation::advance_to(sim_time_t target_time) { if (m_custom_scheduler != nullptr) { - advance_to_impl(target_time, m_custom_scheduler); + advance_to_impl(target_time, m_custom_scheduler); } else { - advance_to_impl(target_time, nullptr); + advance_to_impl(target_time, nullptr); } } template -template +template void BasicSimulation::advance_to_impl( sim_time_t target_time, CustomFCFSScheduler *custom_scheduler) { if constexpr (!AccountResources) { @@ -653,11 +916,11 @@ void BasicSimulation::advance_to_impl( // jobs appended at the current time by a streaming caller. m_scheduler->sync_to(m_current_time); record_queue_arrivals(m_current_time); + apply_capacity_changes(m_current_time); if (m_scheduler->has_eligible_jobs()) { // Call scheduler to evaluate newly arriving jobs while (true) { - num_nodes_t free_nodes = - m_params.m_total_nodes - m_trace.get_nodes_in_use(); + num_nodes_t free_nodes = get_available_nodes(); auto jobs_to_run = m_scheduler->schedule(free_nodes, m_running_jobs, m_current_time); @@ -682,21 +945,29 @@ void BasicSimulation::advance_to_impl( } } if constexpr (AccountResources) { - custom_scheduler->commit_available_nodes(m_params.m_total_nodes - - m_trace.get_nodes_in_use()); + custom_scheduler->commit_available_nodes(get_available_nodes()); } // Main event loop - process events and make scheduling decisions until // complete Compute loop state variables once before entering loop size_t active_count = m_scheduler->active_job_count(); sim_time_t next_arrival = m_scheduler->get_next_arrival_time(); + sim_time_t next_capacity = + m_next_capacity_change < m_capacity_changes.size() + ? m_capacity_changes[m_next_capacity_change].time + : std::numeric_limits::max(); m_queue_length_peak = std::max(m_queue_length_peak, active_count); - // Continue while: (1) jobs waiting to be scheduled, OR (2) events pending - // (jobs running), OR (3) future job arrivals + // Continue while: (1) jobs are waiting, (2) jobs are running, or (3) jobs + // will arrive. A finite streaming advance also consumes capacity changes + // through its requested boundary even while idle. By contrast, batch mode + // drains to max() and must not let unused schedule entries keep a completed + // simulation alive or emit irrelevant resource rows. while (active_count > 0 || !m_trace.pending_events().empty() || - next_arrival < std::numeric_limits::max()) { + next_arrival < std::numeric_limits::max() || + (target_time < std::numeric_limits::max() && + next_capacity <= target_time)) { if (m_params.m_verbose) { std::cout << "Loop iter: active=" << active_count << " events=" << m_trace.pending_events().size() @@ -719,10 +990,11 @@ void BasicSimulation::advance_to_impl( bool should_schedule = false; if (has_replay_event && next_replay_time <= next_arrival && - next_replay_time <= target_time) { + next_replay_time <= next_capacity && next_replay_time <= target_time) { // Process replay events at this time // Advance time FIRST m_current_time = next_replay_time; + advance_capacity_accounting_to(m_current_time); if constexpr (AccountResources) { custom_scheduler->advance_resource_accounting_to(m_current_time); } @@ -751,6 +1023,28 @@ void BasicSimulation::advance_to_impl( bool is_end = !event.is_arrival(); job_no_t event_job_idx = event.get_job_idx(); + if (is_end) { + if constexpr (WarmStage) { + auto &job = m_trace.job_at(event_job_idx); + const sim_time_t begin = + convert_epoch(job.get_begin_time()); + if (begin < m_params.m_sim_start_time) { + // Mutate before Trace processes the departure: a periodic flush + // at this event can then neither write nor account the seed. + job.suppress_output(); + if (m_pre_start_jobs == 0) { + throw std::logic_error( + "warm-start historical-job count underflow"); + } + --m_pre_start_jobs; + } else { + ++m_jobs_completed; + } + } else { + ++m_jobs_completed; + } + } + // Process this event (END or START) - records a // resource-history sample internally (Trace's own Context). m_trace.process_single_event(); @@ -759,19 +1053,22 @@ void BasicSimulation::advance_to_impl( if (is_end) { processed_end_event = true; m_running_jobs.erase(event_job_idx); - m_jobs_completed++; } } + const bool capacity_changed = apply_capacity_changes(m_current_time); + // Only call scheduler if we processed END events (resources freed) - should_schedule = processed_end_event; + should_schedule = processed_end_event || capacity_changed || + next_arrival == m_current_time; } else if (next_arrival < std::numeric_limits::max() && - next_arrival <= target_time) { + next_arrival <= next_capacity && next_arrival <= target_time) { // Job arrival - advance time FIRST // Note: Check next_arrival < infinity to avoid infinite loop // If no jobs arriving, scheduler should pick from waiting queue instead m_current_time = next_arrival; + advance_capacity_accounting_to(m_current_time); if constexpr (AccountResources) { custom_scheduler->advance_resource_accounting_to(m_current_time); } @@ -782,11 +1079,23 @@ void BasicSimulation::advance_to_impl( // this doesn't rely on that. m_scheduler->sync_to(m_current_time); record_queue_arrivals(m_current_time); + apply_capacity_changes(m_current_time); // jobs_at_next_arrival already collected during wait_queue scan // TODO: Pass jobs_at_next_arrival to scheduler for efficient evaluation // For now, just set flag to schedule should_schedule = true; + } else if (next_capacity < std::numeric_limits::max() && + next_capacity <= target_time) { + m_current_time = next_capacity; + advance_capacity_accounting_to(m_current_time); + if constexpr (AccountResources) { + custom_scheduler->advance_resource_accounting_to(m_current_time); + } + m_scheduler->sync_to(m_current_time); + record_queue_arrivals(m_current_time); + apply_capacity_changes(m_current_time); + should_schedule = true; } else { // No arrivals and no replay events before target_time if (m_params.m_verbose) { @@ -803,8 +1112,7 @@ void BasicSimulation::advance_to_impl( if (should_schedule) { // Keep calling scheduler until it can't start any more jobs while (true) { - num_nodes_t free_nodes = - m_params.m_total_nodes - m_trace.get_nodes_in_use(); + num_nodes_t free_nodes = get_available_nodes(); auto jobs_to_run = m_scheduler->schedule(free_nodes, m_running_jobs, m_current_time); @@ -842,16 +1150,26 @@ void BasicSimulation::advance_to_impl( } } if constexpr (AccountResources) { - custom_scheduler->commit_available_nodes(m_params.m_total_nodes - - m_trace.get_nodes_in_use()); + custom_scheduler->commit_available_nodes(get_available_nodes()); } // Update loop state variables at end of iteration active_count = m_scheduler->active_job_count(); next_arrival = m_scheduler->get_next_arrival_time(); + next_capacity = m_next_capacity_change < m_capacity_changes.size() + ? m_capacity_changes[m_next_capacity_change].time + : std::numeric_limits::max(); // Peak queue length after all events and scheduling at this timestamp. m_queue_length_peak = std::max(m_queue_length_peak, active_count); + + if constexpr (WarmStage) { + // Transition only after every peer event and scheduling decision at the + // last historical departure's timestamp has settled. + if (m_pre_start_jobs == 0) { + return; + } + } } // Loop exited - log final state for debugging @@ -875,12 +1193,20 @@ void BasicSimulation::advance_to_impl( // is pure bookkeeping - every scheduling decision above was already // made using real event times, never target_time, so this can't // change any of them. + const bool drain_to_completion = + target_time == std::numeric_limits::max(); if constexpr (AccountResources) { - if (m_trace.get_nodes_in_use() > 0) { + if (!drain_to_completion && m_trace.get_nodes_in_use() > 0) { custom_scheduler->advance_resource_accounting_to(target_time); + advance_capacity_accounting_to(target_time); } } - m_current_time = target_time; + // max() is the internal/public drain sentinel, not a meaningful simulated + // timestamp. After a drain, preserve the last real event time rather than + // exposing max() (which also overflows integer-formatted CLI output). + if (!drain_to_completion) { + m_current_time = target_time; + } } template @@ -953,12 +1279,13 @@ BasicSimulation::get_statistics() const { // Resource utilization stats.total_nodes = m_params.m_total_nodes; stats.nodes_in_use = get_nodes_in_use(); - stats.nodes_available = stats.total_nodes - stats.nodes_in_use; + stats.nodes_available = get_available_nodes(); // Calculate wait times and turnaround times tdiff_t total_wait = 0.0; tdiff_t total_turnaround = 0.0; sim_time_t max_completion = 0.0; + sim_time_t max_scheduled_completion = 0.0; num_jobs_t completed_count = 0; tdiff_t total_node_seconds = 0.0; @@ -971,17 +1298,25 @@ BasicSimulation::get_statistics() const { // started," silently excluding it from these averages. This // matches the same convention now used for m_jobs_completed // above (see the end-of-run() completion count). - if (job.is_scheduled()) { + if (!job.is_scheduled()) { + continue; + } + + const sim_time_t completion = + convert_epoch(job.get_end_time()); + max_scheduled_completion = std::max(max_scheduled_completion, completion); + total_node_seconds += + static_cast(job.get_num_nodes()) * job.get_actual_run_time(); + if (completion <= stats.current_time) { tdiff_t wait = job.get_wait_time(); tdiff_t exec = job.get_actual_run_time(); total_wait += wait; total_turnaround += (wait + exec); - total_node_seconds += static_cast(job.get_num_nodes()) * exec; - sim_time_t completion = convert_epoch(job.get_end_time()); max_completion = std::max(max_completion, completion); completed_count++; } + } stats.avg_wait_time = @@ -1000,16 +1335,17 @@ BasicSimulation::get_statistics() const { std::isfinite(stats.current_time) && stats.nodes_in_use > 0 ? stats.current_time : m_custom_scheduler->m_resource_area_time; + const tdiff_t capacity_area = capacity_area_through(accounting_horizon); stats.utilization = - m_custom_scheduler->utilization_through(accounting_horizon); + capacity_area > 0.0 ? stats.resource_area / capacity_area : 0.0; } else { // Preserve the original post-hoc statistic for standard schedulers. - stats.resource_area = total_node_seconds; + stats.resource_area = total_node_seconds + m_warm_resource_area; + const sim_time_t accounting_end = + std::max(max_scheduled_completion, m_warm_resource_end); + const tdiff_t capacity_area = capacity_area_through(accounting_end); stats.utilization = - (stats.total_nodes > 0 && stats.makespan > 0) - ? total_node_seconds / - (static_cast(stats.total_nodes) * stats.makespan) - : 0.0; + capacity_area > 0.0 ? stats.resource_area / capacity_area : 0.0; } return stats; diff --git a/src/sim/sim.hpp b/src/sim/sim.hpp index 5b3cc21..8f4584d 100644 --- a/src/sim/sim.hpp +++ b/src/sim/sim.hpp @@ -14,6 +14,7 @@ #error "no config" #endif +#include #include #include #include @@ -26,6 +27,7 @@ #include "common.hpp" #include "params/sim_params.hpp" +#include "sim/capacity_schedule.hpp" #include "sim/scheduler_base.hpp" #include "sim/scheduler_fcfs_custom.hpp" #include "trace/dr_event.hpp" @@ -65,9 +67,28 @@ template class BasicSimulation { /// Current simulation time sim_time_t m_current_time; + /// Fixed physical limit used only to reject jobs that can never fit. + num_nodes_t m_job_rejection_capacity; + + /// Ordered external machine-capacity changes and the active cursor. + std::vector m_capacity_changes; + size_t m_next_capacity_change; + num_nodes_t m_current_capacity; + /// Integral of effective capacity over the current accounting horizon. + /// Without a schedule, effective capacity remains m_params.m_total_nodes. + tdiff_t m_capacity_area; + /// Last timestamp incorporated into m_capacity_area. + sim_time_t m_capacity_area_time; + /// Counters num_jobs_t m_jobs_completed; num_jobs_t m_jobs_submitted; ///< Jobs submitted during the current run. + /// Historical jobs still running during the temporary warm-start stage. + size_t m_pre_start_jobs; + /// Post-boundary node-seconds contributed by suppressed historical jobs. + tdiff_t m_warm_resource_area; + /// Latest departure among historical jobs active after the boundary. + sim_time_t m_warm_resource_end; /// Serializable random-number engine used for duration sampling. RNGen<> m_rng; @@ -106,9 +127,10 @@ template class BasicSimulation { /** * @brief Run a complete batch simulation for the configured trace. * @details - * Loads and prepares the input trace when necessary, submits every job, - * and drains the event queue. For externally fed work, use append_job() - * or append_jobs() followed by advance_to() instead. + * Loads and prepares the input trace when necessary and submits its jobs. + * With max_time configured, processes events through that inclusive + * boundary; otherwise drains the event queue. For externally fed work, use + * append_job() or append_jobs() followed by advance_to() instead. */ void run(); @@ -237,8 +259,10 @@ template class BasicSimulation { * - All jobs have already been submitted, OR * - External tool knows the next job arrival is at >= target_time * - * POSTCONDITION: m_current_time == target_time, and all scheduling - * decisions have been made up to that time. + * POSTCONDITION: for a finite target, m_current_time == target_time and all + * scheduling decisions have been made up to that time. The maximum + * representable value is treated as a drain sentinel; after draining, + * m_current_time is the last real event time. * * Jobs selected by the scheduler are recorded through Trace::insert_job(). * @see submit_job() @@ -254,13 +278,18 @@ template class BasicSimulation { /** * @brief Return instantaneous node utilization. - * @return nodes currently used divided by total configured nodes, in [0,1]. + * @return Nodes currently used divided by effective current capacity. + * @details Effective capacity is the scheduled capacity, raised to current + * occupancy while a non-preemptive reduction is still draining. The result + * therefore remains in [0,1], including when scheduled capacity is zero. */ double get_current_utilization() const { - return m_params.m_total_nodes == 0 + const auto used = get_nodes_in_use(); + const auto effective_capacity = std::max(m_current_capacity, used); + return effective_capacity == 0 ? 0.0 - : static_cast(get_nodes_in_use()) / - static_cast(m_params.m_total_nodes); + : static_cast(used) / + static_cast(effective_capacity); } /** @@ -298,9 +327,13 @@ template class BasicSimulation { * @return Unallocated-node count as num_nodes_t. */ num_nodes_t get_available_nodes() const { - return m_params.m_total_nodes - get_nodes_in_use(); + const auto used = get_nodes_in_use(); + return used < m_current_capacity ? m_current_capacity - used : 0; } + /** @brief Capacity currently available to this simulated workload. */ + num_nodes_t get_current_capacity() const { return m_current_capacity; } + /** * @brief Get the count of jobs that have arrived but remain unscheduled. * @@ -437,8 +470,9 @@ template class BasicSimulation { tdiff_t resource_area; /** * @brief Resource-area utilization. - * @details Uses the live accounting horizon for Custom FCFS and makespan - * for standard schedulers. + * @details Divides allocated-node area by effective-capacity area. Uses + * the live accounting horizon for Custom FCFS and makespan for standard + * schedulers. */ double utilization; tdiff_t avg_wait_time; ///< Mean completed-job wait duration. @@ -504,11 +538,19 @@ template class BasicSimulation { num_jobs_t initialize_trace(num_jobs_t max_jobs = 0); protected: - /** Advance using a compile-time-selected Custom-FCFS accounting path. */ - template + /** + * Advance using compile-time-selected accounting and warm-start paths. + * WarmStage is instantiated only while historical jobs remain; the normal + * stage contains no pre-start-time condition. + */ + template void advance_to_impl(sim_time_t target_time, CustomFCFSScheduler *custom_scheduler); + /** Seed replay jobs active before m_params.m_sim_start_time, then reschedule + * post-boundary work through the ordinary scheduler. */ + void run_warm_start(); + /** * @brief Process an arrival event for an existing trace job. * @param[in] job_idx Identifier of the arriving job. @@ -586,6 +628,21 @@ template class BasicSimulation { */ void record_queue_arrivals(sim_time_t current_time); + /** Reset the capacity cursor to the configured maximum. */ + void reset_capacity_schedule(); + + /** Reset time-integrated effective-capacity accounting at start_time. */ + void reset_capacity_accounting(sim_time_t start_time); + + /** Accumulate effective capacity through current_time. */ + void advance_capacity_accounting_to(sim_time_t current_time); + + /** Return effective-capacity area through a finite snapshot time. */ + tdiff_t capacity_area_through(sim_time_t through_time) const; + + /** Apply every capacity change effective at the supplied timestamp. */ + bool apply_capacity_changes(sim_time_t current_time); + /** * @brief Sample a job duration from the configured distribution. * @param[in] time_limit User-provided time limit. diff --git a/src/trace/data_columns.cpp b/src/trace/data_columns.cpp index 70f11df..9093569 100644 --- a/src/trace/data_columns.cpp +++ b/src/trace/data_columns.cpp @@ -34,7 +34,7 @@ Data_Columns::Data_Columns() m_has_q_id_column(false), #endif m_col_to_avoid_idx(std::numeric_limits::max()), - m_trace_format("simple"), m_timestamp_format("iso"), + m_trace_format("simple"), m_timestamp_format("epoch"), m_timezone_str("America/Los_Angeles"), m_trace_mode(TraceMode::REPLAY) // Default to replay { @@ -58,7 +58,7 @@ Data_Columns::Data_Columns(const std::string &format) m_has_q_id_column(false), #endif m_col_to_avoid_idx(std::numeric_limits::max()), - m_trace_format(format), m_timestamp_format("iso"), + m_trace_format(format), m_timestamp_format("epoch"), m_timezone_str("America/Los_Angeles"), m_trace_mode(TraceMode::REPLAY) // Will be detected in check_header { @@ -131,7 +131,7 @@ Data_Columns::~Data_Columns() { } tzset(); if (m_cur_tz != nullptr) { - delete m_cur_tz; + std::free(m_cur_tz); m_cur_tz = nullptr; } } @@ -169,7 +169,7 @@ void Data_Columns::init() { // the daylight saving condition. if (m_cur_tz != nullptr) { - delete m_cur_tz; + std::free(m_cur_tz); m_cur_tz = nullptr; } @@ -180,7 +180,7 @@ void Data_Columns::init() { memcpy((void *)m_cur_tz, (void *)tz, tz_str_len * sizeof(char)); } - setenv("TZ", DATA_TIMEZONE, 1); + setenv("TZ", m_timezone_str.c_str(), 1); tzset(); } @@ -307,6 +307,15 @@ bool Data_Columns::check_header(const std::string &fname) { {find_column({"q_id"}), "q_id"}); } #endif + + // A replay trace does not require actual_run_time because it can be + // derived from begin_time/end_time. When supplied, retain it so the row + // loader can verify that all three observed execution fields agree. + auto [found, actual_run_time_idx] = + find_column_optional(actual_run_time_aliases); + if (found) { + m_cols_to_read.push_back({actual_run_time_idx, "actual_run_time"}); + } } else { // Simulation mode: no begin_time or end_time col_no_t num_nodes_idx = find_column({"num_nodes"}); diff --git a/src/trace/data_columns.hpp b/src/trace/data_columns.hpp index ca9c42a..144b00d 100644 --- a/src/trace/data_columns.hpp +++ b/src/trace/data_columns.hpp @@ -45,7 +45,7 @@ class Data_Columns { col_by_name_t m_col_by_name; /// Saved process timezone, restored when this mapping is destroyed. - const char *m_cur_tz; + char *m_cur_tz; /// Number of physical columns declared by the validated header. num_cols_t m_total_columns; @@ -72,16 +72,18 @@ class Data_Columns { /** @brief Construct a mapping for a named trace format. * @param[in] format Supported format name, such as "simple" or "lassen". */ Data_Columns(const std::string &format); - /** @brief Construct a mapping with timestamp and timezone controls. + /** @brief Construct a mapping with timestamp compatibility metadata and a + * timezone used for calendar-time parsing. * @param[in] format Supported trace format name. - * @param[in] timestamp_format Timestamp encoding name. + * @param[in] timestamp_format Retained epoch/iso compatibility setting; + * input timestamp encoding is auto-detected. * @param[in] timezone Timezone used for timestamps without offsets. */ Data_Columns(const std::string &format, const std::string ×tamp_format, const std::string &timezone); /// Restore the process timezone saved during construction. virtual ~Data_Columns(); - /// Return the configured timestamp encoding name. + /// Return the retained timestamp-format compatibility value. std::string get_timestamp_format() const { return m_timestamp_format; } /// Return the timezone used for timestamps without explicit offsets. std::string get_timezone() const { return m_timezone_str; } @@ -143,7 +145,7 @@ class Data_Columns { /// Requested input layout name, such as `simple` or `lassen`. std::string m_trace_format; - /// Requested timestamp encoding, such as `epoch` or `iso`. + /// Retained timestamp-format compatibility value (`epoch` or `iso`). std::string m_timestamp_format; /// Timezone for timestamps that do not carry their own offset. std::string m_timezone_str; diff --git a/src/trace/epoch.cpp b/src/trace/epoch.cpp index 171cf2d..556c649 100644 --- a/src/trace/epoch.cpp +++ b/src/trace/epoch.cpp @@ -61,7 +61,11 @@ std::ostream &operator<<(std::ostream &os, const epoch_t &t) { * Check if the give string is timestamp */ bool is_timestamp(const std::string &time_str) { - std::istringstream iss{time_str}; + std::string normalized = time_str; + if (normalized.size() > 10 && normalized[10] == 'T') { + normalized[10] = ' '; + } + std::istringstream iss{normalized}; std::tm t{}; t.tm_isdst = -1; @@ -74,7 +78,14 @@ bool is_timestamp(const std::string &time_str) { * fractional second. */ epoch_t convert_time(const std::string &time_str) { - std::istringstream iss{time_str}; + // Accept both the historical "YYYY-MM-DD HH:MM:SS" spelling and the ISO + // 8601 date/time separator. Offset-bearing values are handled separately + // by parse_time_with_timezone(). + std::string normalized = time_str; + if (normalized.size() > 10 && normalized[10] == 'T') { + normalized[10] = ' '; + } + std::istringstream iss{normalized}; std::tm t{}; t.tm_isdst = -1; diff --git a/src/trace/job_io.cpp b/src/trace/job_io.cpp index 5a93551..5428bf4 100644 --- a/src/trace/job_io.cpp +++ b/src/trace/job_io.cpp @@ -57,6 +57,8 @@ int load(const string &fname, const Data_Columns &dcols, max_cnt = std::numeric_limits::max(); } num_jobs_t cnt = static_cast(0u); + TimestampEncoding timestamp_encoding = TimestampEncoding::EPOCH; + bool timestamp_encoding_detected = false; while (std::getline(ifs, line)) { // Read a line if (cnt++ >= max_cnt) { @@ -121,15 +123,20 @@ int load(const string &fname, const Data_Columns &dcols, rec_str.insert(rec_str.begin() + queue_pos, "pbatch"); } + if (!timestamp_encoding_detected) { + timestamp_encoding = detect_timestamp_encoding(rec_str.at(1)); + timestamp_encoding_detected = true; + } + try { // Constructor may raise an exception based on filtering. // In that case, it can be handled as below to ignore this // particular sample that is not compliant. #if SHOW_ORG_NO // line number starts from 1 - data.push_back(Job_Record(cnt, rec_str)); + data.push_back(Job_Record(cnt, rec_str, timestamp_encoding)); #else - data.push_back(Job_Record(rec_str)); + data.push_back(Job_Record(rec_str, timestamp_encoding)); #endif } catch (std::domain_error &e) { // Ignore this case @@ -174,6 +181,8 @@ int load(const string &fname, const Data_Columns &dcols, if (max_cnt == static_cast(0u)) { max_cnt = std::numeric_limits::max(); } + TimestampEncoding timestamp_encoding = TimestampEncoding::EPOCH; + bool timestamp_encoding_detected = false; if (dcols.has_q_id_column()) { const auto q_idx = dcols.get_queue_idx(); @@ -202,8 +211,13 @@ int load(const string &fname, const Data_Columns &dcols, } fields.emplace_back(std::move(value)); } + if (!timestamp_encoding_detected) { + timestamp_encoding = detect_timestamp_encoding(fields.at(1)); + timestamp_encoding_detected = true; + } data.emplace_back(fields, queue, - dcols.get_trace_mode() == TraceMode::REPLAY); + dcols.get_trace_mode() == TraceMode::REPLAY, + timestamp_encoding); #if SHOW_ORG_NO data.back().set_org_line_no(cnt); #endif @@ -239,8 +253,13 @@ int load(const string &fname, const Data_Columns &dcols, const auto &pos = val_pos[col_idx]; fields.emplace_back(trim(line.substr(pos.first, pos.second))); } + if (!timestamp_encoding_detected) { + timestamp_encoding = detect_timestamp_encoding(fields.at(1)); + timestamp_encoding_detected = true; + } data.emplace_back(fields, Queue1, - dcols.get_trace_mode() == TraceMode::REPLAY); + dcols.get_trace_mode() == TraceMode::REPLAY, + timestamp_encoding); #if SHOW_ORG_NO data.back().set_org_line_no(cnt); #endif diff --git a/src/trace/job_record.cpp b/src/trace/job_record.cpp index 61d30b9..fd74b22 100644 --- a/src/trace/job_record.cpp +++ b/src/trace/job_record.cpp @@ -11,11 +11,39 @@ #include "trace/job_record.hpp" #include "trace/parse_utils.hpp" +#include #include #include namespace dr_evt { +namespace { + +// epoch_t stores fractional seconds as float, so a duration reconstructed +// from two timestamps can differ slightly from the input double. +constexpr tdiff_t runtime_timestamp_tolerance = 1.0e-6; + +void validate_replay_run_time(tdiff_t actual_run_time, + tdiff_t recorded_run_time) { + if (!std::isfinite(actual_run_time) || + std::fabs(actual_run_time - recorded_run_time) > + runtime_timestamp_tolerance) { + throw std::domain_error{ + "Replay actual_run_time must equal end_time - begin_time"}; + } +} + +void validate_simulation_run_time(tdiff_t actual_run_time, + timeout_t time_limit) { + if (!std::isfinite(actual_run_time) || + actual_run_time > static_cast(time_limit)) { + throw std::domain_error{ + "Simulation actual_run_time must not exceed time_limit"}; + } +} + +} // namespace + unsigned int Job_Record::num_inputs = 0u; Job_Record::Job_Record(const Job_Record &o) @@ -112,10 +140,12 @@ Job_Record::Job_Record(const epoch_t &submit_time, num_nodes_t num_nodes, } #if SHOW_ORG_NO -Job_Record::Job_Record(job_no_t no, const std::vector &str_vec) +Job_Record::Job_Record(job_no_t no, const std::vector &str_vec, + TimestampEncoding timestamp_encoding) : m_org_no(no), #else -Job_Record::Job_Record(const std::vector &str_vec) +Job_Record::Job_Record(const std::vector &str_vec, + TimestampEncoding timestamp_encoding) : #endif #if MARK_DAT_PERIOD @@ -145,17 +175,17 @@ Job_Record::Job_Record(const std::vector &str_vec) #endif // Check mode based on number of fields: - // Replay mode (6): num_nodes, begin_time, end_time, submit_time, queue, - // time_limit Simulation mode (4 or 5): num_nodes, submit_time, queue, - // time_limit[, actual_run_time] - bool is_replay_mode = (num_inputs == 6); + // Replay mode (6 or 7): num_nodes, begin_time, end_time, submit_time, queue, + // time_limit[, actual_run_time]. Simulation mode (4 or 5): num_nodes, + // submit_time, queue, time_limit[, actual_run_time]. + bool is_replay_mode = (num_inputs >= 6); bool has_actual_run_time = (num_inputs == 5 || num_inputs == 7); if (is_replay_mode) { // Replay mode: has begin_time and end_time - set_by(m_t_begin, *it++); - set_by(m_t_end, *it++); - set_by(m_t_submit, *it++); + set_by(m_t_begin, *it++, timestamp_encoding); + set_by(m_t_end, *it++, timestamp_encoding); + set_by(m_t_submit, *it++, timestamp_encoding); #if EVENT_TIME_ORDER if ((m_t_begin > m_t_end) || (m_t_submit > m_t_begin)) { @@ -189,8 +219,13 @@ Job_Record::Job_Record(const std::vector &str_vec) #endif set_by(m_t_limit, *it++); - // Compute actual_run_time from recorded times - m_actual_run_time = static_cast(m_t_end - m_t_begin); + const tdiff_t recorded_run_time = static_cast(m_t_end - m_t_begin); + if (has_actual_run_time) { + set_by(m_actual_run_time, *it++); + validate_replay_run_time(m_actual_run_time, recorded_run_time); + } else { + m_actual_run_time = recorded_run_time; + } m_is_simulated = false; } else { // Simulation mode: no begin_time or end_time in input. @@ -208,7 +243,7 @@ Job_Record::Job_Record(const std::vector &str_vec) // scheduler," consistent with the replay-mode branch's intent. m_is_simulated = false; - set_by(m_t_submit, *it++); + set_by(m_t_submit, *it++, timestamp_encoding); #if DR_EVT_LEGACY_QUEUE_INPUT set_by(m_q, *it++); #else @@ -220,6 +255,7 @@ Job_Record::Job_Record(const std::vector &str_vec) // determine_job_run_time() if (has_actual_run_time) { set_by(m_actual_run_time, *it++); + validate_simulation_run_time(m_actual_run_time, m_t_limit); } else { m_actual_run_time = 0.0; } @@ -234,7 +270,8 @@ Job_Record::Job_Record(const std::vector &str_vec) } Job_Record::Job_Record(const std::vector &fields, - job_queue_t queue, bool is_replay_mode) + job_queue_t queue, bool is_replay_mode, + TimestampEncoding timestamp_encoding) : m_q(queue) #if MARK_DAT_PERIOD , @@ -247,9 +284,10 @@ Job_Record::Job_Record(const std::vector &fields, const auto expected_fields = is_replay_mode ? 5u : 3u; if (fields.size() != expected_fields && - !(!is_replay_mode && fields.size() == expected_fields + 1u)) { + fields.size() != expected_fields + 1u) { throw std::invalid_argument{"Queue-free record format does not match"}; } + const bool has_actual_run_time = fields.size() == expected_fields + 1u; auto it = fields.cbegin(); set_by(m_num_nodes, *it++); @@ -263,9 +301,9 @@ Job_Record::Job_Record(const std::vector &fields, #endif if (is_replay_mode) { - set_by(m_t_begin, *it++); - set_by(m_t_end, *it++); - set_by(m_t_submit, *it++); + set_by(m_t_begin, *it++, timestamp_encoding); + set_by(m_t_end, *it++, timestamp_encoding); + set_by(m_t_submit, *it++, timestamp_encoding); #if EVENT_TIME_ORDER if ((m_t_begin > m_t_end) || (m_t_submit > m_t_begin)) { @@ -284,16 +322,23 @@ Job_Record::Job_Record(const std::vector &fields, #endif set_by(m_t_limit, *it++); - m_actual_run_time = static_cast(m_t_end - m_t_begin); + const tdiff_t recorded_run_time = static_cast(m_t_end - m_t_begin); + if (has_actual_run_time) { + set_by(m_actual_run_time, *it++); + validate_replay_run_time(m_actual_run_time, recorded_run_time); + } else { + m_actual_run_time = recorded_run_time; + } m_is_simulated = false; } else { m_t_begin = unscheduled_sentinel(); m_t_end = unscheduled_sentinel(); m_is_simulated = false; - set_by(m_t_submit, *it++); + set_by(m_t_submit, *it++, timestamp_encoding); set_by(m_t_limit, *it++); - if (fields.size() == expected_fields + 1u) { + if (has_actual_run_time) { set_by(m_actual_run_time, *it++); + validate_simulation_run_time(m_actual_run_time, m_t_limit); } else { m_actual_run_time = 0.0; } diff --git a/src/trace/job_record.hpp b/src/trace/job_record.hpp index d2c0ed7..0a67f1b 100644 --- a/src/trace/job_record.hpp +++ b/src/trace/job_record.hpp @@ -14,6 +14,7 @@ #include "common.hpp" #include "trace/epoch.hpp" +#include "trace/parse_utils.hpp" #include #include #include @@ -70,11 +71,14 @@ class Job_Record { public: #if SHOW_ORG_NO - Job_Record(job_no_t n, const std::vector &svec) noexcept(false); + Job_Record(job_no_t n, const std::vector &svec, + TimestampEncoding timestamp_encoding) noexcept(false); #else /** @brief Parse a job record from the fields of one trace row. - * @param[in] str_vec Parsed input fields in configured column order. */ - Job_Record(const std::vector &str_vec) noexcept(false); + * @param[in] str_vec Parsed input fields in configured column order. + * @param[in] timestamp_encoding Encoding detected for the input trace. */ + Job_Record(const std::vector &str_vec, + TimestampEncoding timestamp_encoding) noexcept(false); #endif /** @@ -85,9 +89,11 @@ class Job_Record { * @param[in] queue Typed queue identifier for the job. * @param[in] is_replay_mode Whether input timestamps describe replayed * execution rather than a new simulation. + * @param[in] timestamp_encoding Encoding detected for the input trace. */ Job_Record(const std::vector &fields, job_queue_t queue, - bool is_replay_mode) noexcept(false); + bool is_replay_mode, + TimestampEncoding timestamp_encoding) noexcept(false); /** @brief Copy a job record. * @param[in] other Record to copy. */ @@ -135,6 +141,18 @@ class Job_Record { /// will never resolve" rather than "still waiting." /** @param[in] t Replacement submission time. */ void set_submit_time(const epoch_t &t) { m_t_submit = t; } + /** Exclude a completed warm-start seed from output, stats, and retention. */ + void suppress_output() { + m_is_simulated = false; + m_t_end = unscheduled_sentinel(); + m_t_submit = unscheduled_sentinel(); + } + /** Retain replay duration and attributes but clear its recorded schedule. */ + void prepare_for_resimulation() { + m_is_simulated = false; + m_t_begin = unscheduled_sentinel(); + m_t_end = unscheduled_sentinel(); + } /** @brief Return time spent waiting before execution. */ tdiff_t get_wait_time() const { return (m_t_begin - m_t_submit); } /** @brief Return the requested run-time limit. */ diff --git a/src/trace/parse_utils.cpp b/src/trace/parse_utils.cpp index 47b033e..de10c1f 100644 --- a/src/trace/parse_utils.cpp +++ b/src/trace/parse_utils.cpp @@ -58,36 +58,36 @@ std::map jobq2str{ {QueueUnknown, ""}}; #endif -void set_by(epoch_t &t, const std::string &str) { - // Auto-detect format: if string contains only digits (and optional minus - // sign), treat as Unix epoch seconds; otherwise parse as ISO timestamp +TimestampEncoding detect_timestamp_encoding(const std::string &str) { + size_t pos = 0; + try { + std::stod(str, &pos); + if (pos == str.size()) { + return TimestampEncoding::EPOCH; + } + } catch (const std::exception &) { + } + return TimestampEncoding::CALENDAR; +} + +void set_by(epoch_t &t, const std::string &str, TimestampEncoding encoding) { if (str.empty()) { t = {0, 0.0f}; return; } - bool is_epoch = true; - for (char c : str) { - if (!std::isdigit(c) && c != '-' && c != '.') { - is_epoch = false; - break; - } - } - - if (is_epoch) { + if (encoding == TimestampEncoding::EPOCH) { // Parse as Unix epoch seconds - try { - size_t pos; - double seconds = std::stod(str, &pos); - time_t sec_int = static_cast(seconds); - float sec_frac = static_cast(seconds - sec_int); - t = {sec_int, sec_frac}; - } catch (...) { - // Fallback to ISO parsing if epoch parsing fails - t = convert_time(str); + size_t pos = 0; + const double seconds = std::stod(str, &pos); + if (pos != str.size()) { + throw std::invalid_argument{"Failed to parse epoch timestamp: " + str}; } + time_t sec_int = static_cast(seconds); + float sec_frac = static_cast(seconds - sec_int); + t = {sec_int, sec_frac}; } else { - // Parse as ISO/human-readable timestamp + // Parse as an ISO/human-readable calendar timestamp. // Check if it has timezone offset (±HH:MM or Z) bool has_timezone = (str.find_last_of("+-Z") != std::string::npos && str.find_last_of("+-Z") > 10); @@ -105,6 +105,10 @@ void set_by(epoch_t &t, const std::string &str) { } } +void set_by(epoch_t &t, const std::string &str) { + set_by(t, str, detect_timestamp_encoding(str)); +} + void set_by(unsigned &v, const std::string &str) { size_t pos; v = static_cast(stoi(str, &pos)); diff --git a/src/trace/parse_utils.hpp b/src/trace/parse_utils.hpp index 6fd8c02..2e59029 100644 --- a/src/trace/parse_utils.hpp +++ b/src/trace/parse_utils.hpp @@ -21,9 +21,21 @@ namespace dr_evt { /** \addtogroup dr_evt_trace * @{ */ +/** Timestamp encoding selected once for an input trace or schedule. */ +enum class TimestampEncoding { EPOCH, CALENDAR }; + +/** @brief Detect whether one representative value is numeric epoch time or a + * calendar timestamp. */ +TimestampEncoding detect_timestamp_encoding(const std::string &str); + /** @brief Parse an epoch timestamp. @param[out] t Parsed timestamp. @param[in] * str Input text. */ void set_by(epoch_t &t, const std::string &str); +/** @brief Parse a timestamp using an encoding already selected for its input. + * @param[out] t Parsed timestamp. + * @param[in] str Input text. + * @param[in] encoding Encoding detected once for the containing input. */ +void set_by(epoch_t &t, const std::string &str, TimestampEncoding encoding); /** @brief Parse an unsigned integer. @param[out] v Parsed value. @param[in] str * Input text. */ void set_by(unsigned &v, const std::string &str); diff --git a/src/trace/trace.cpp b/src/trace/trace.cpp index 2a29991..f6fdab1 100644 --- a/src/trace/trace.cpp +++ b/src/trace/trace.cpp @@ -34,6 +34,9 @@ BasicTrace::BasicTrace(const std::string &fname) m_resource_history_capacity(0), m_resource_history_capacity_resolved(false), m_resource_trace_total_nodes(static_cast(0u)), + m_resource_trace_current_capacity(static_cast(0u)), + m_resource_capacity_initialized(false), m_resource_recording_start(0.0), + m_resource_recording_baseline{}, m_has_resource_recording_baseline(false), m_resource_trace_msec(false), m_simulated_trace_msec(false), m_next_job_to_write(0) { if (!m_dcols.check_header(fname)) { @@ -54,6 +57,9 @@ BasicTrace::BasicTrace(const std::string &fname, m_resource_history_capacity(0), m_resource_history_capacity_resolved(false), m_resource_trace_total_nodes(static_cast(0u)), + m_resource_trace_current_capacity(static_cast(0u)), + m_resource_capacity_initialized(false), m_resource_recording_start(0.0), + m_resource_recording_baseline{}, m_has_resource_recording_baseline(false), m_resource_trace_msec(false), m_simulated_trace_msec(false), m_next_job_to_write(0) { if (!m_dcols.check_header(fname)) { @@ -78,6 +84,9 @@ BasicTrace::BasicTrace(const std::string &fname, m_resource_history_capacity(0), m_resource_history_capacity_resolved(false), m_resource_trace_total_nodes(static_cast(0u)), + m_resource_trace_current_capacity(static_cast(0u)), + m_resource_capacity_initialized(false), m_resource_recording_start(0.0), + m_resource_recording_baseline{}, m_has_resource_recording_baseline(false), m_resource_trace_msec(false), m_simulated_trace_msec(false), m_next_job_to_write(0) { if (!m_dcols.check_header(fname)) { @@ -251,7 +260,8 @@ void BasicTrace::process_events_until(const epoch_t &t_sub) { template void BasicTrace::run_job_trace(const std::string &resource_trace_file, - num_nodes_t total_nodes) { + num_nodes_t total_nodes, + sim_time_t max_time) { if (m_data.empty()) { return; } @@ -287,6 +297,9 @@ void BasicTrace::run_job_trace(const std::string &resource_trace_file, static_cast(m_num_reclaimed + m_data.size()); for (job_no_t job_no = first_job; job_no < jobs_end; ++job_no) { const auto t_sub = job_at(job_no).get_submit_time(); + if (convert_epoch(t_sub) > max_time) { + break; + } process_events_until(t_sub); auto &job = job_at(job_no); @@ -301,9 +314,13 @@ void BasicTrace::run_job_trace(const std::string &resource_trace_file, m_ctx.m_evtq.emplace(job_no, job.get_end_time(), departure); m_replay_jobs_enqueued = static_cast(job_no) + 1; } - // Process all the remaiing events. Use any time later than any timestamp - // in the trace for flushing. - process_events_until(convert_time(max_tstamp)); + if (max_time == std::numeric_limits::max()) { + // Process all remaining events. Use any time later than any timestamp in + // the trace for flushing. + process_events_until(convert_time(max_tstamp)); + } else { + run_until_inclusive(max_time); + } write_resource_trace(resource_trace_file, total_nodes); } @@ -577,6 +594,24 @@ void BasicTrace::insert_job(job_no_t job_idx, sim_time_t start_time) { // Update job record with computed times (for output) job_at(job_idx).set_begin_time(start_epoch); job_at(job_idx).compute_end_time(); + + // A replay-format record rescheduled by Simulation's warm-start path now + // owns ordinary scheduler-created events. Mark its permanent position as + // enqueued so replay-mode front reclamation can advance past it. + m_replay_jobs_enqueued = + std::max(m_replay_jobs_enqueued, static_cast(job_idx) + 1); +} + +template +void BasicTrace::enqueue_replay_job(job_no_t job_idx) { + if (m_dcols.get_trace_mode() != TraceMode::REPLAY) { + throw std::logic_error("enqueue_replay_job() requires replay-format input"); + } + const auto &job = job_at(job_idx); + m_ctx.m_evtq.emplace(job_idx, job.get_begin_time(), arrival); + m_ctx.m_evtq.emplace(job_idx, job.get_end_time(), departure); + m_replay_jobs_enqueued = + std::max(m_replay_jobs_enqueued, static_cast(job_idx) + 1); } template @@ -596,8 +631,8 @@ void BasicTrace::run_until_inclusive(sim_time_t target_time) { while (!m_ctx.m_evtq.empty()) { const auto &event = *m_ctx.m_evtq.begin(); const epoch_t event_timestamp = event.get_time(); - sim_time_t event_time = static_cast(event_timestamp.first) + - event_timestamp.second; + sim_time_t event_time = + static_cast(event_timestamp.first) + event_timestamp.second; if (event_time > target_time) { break; // Stop after processing all events <= target_time @@ -659,6 +694,14 @@ template void BasicTrace::start_resource_trace(const std::string &filename, num_nodes_t total_nodes, bool msec) { + // Capacity is also needed by in-memory samples when output is opened only + // later by write_resource_trace(). + m_resource_trace_total_nodes = total_nodes; + m_resource_trace_msec = msec; + if (!m_resource_capacity_initialized) { + m_resource_trace_current_capacity = total_nodes; + m_resource_capacity_initialized = true; + } if (filename.empty()) { return; } @@ -675,21 +718,34 @@ void BasicTrace::start_resource_trace(const std::string &filename, << std::endl; return; } - m_resource_trace_total_nodes = total_nodes; - m_resource_trace_msec = msec; - std::string header = "time,free_nodes,allocated_nodes"; header += Policy::resource_columns(); header += "\n"; m_resource_trace_ofs << header; - // Baseline row: all nodes free at time 0, matching the convention - // used elsewhere for this file format. - const auto baseline_sample = this->sample(epoch_t{}, 0); - std::string baseline = format_sim_time(0.0, msec) + "," + - std::to_string(total_nodes) + ",0" + - Policy::resource_values(baseline_sample) + "\n"; + // Baseline row. For a traditional run this remains time 0 with no + // allocation. A warm run resets m_resource_recording_start at its boundary + // while preserving the occupancy and Pcon state established by replay. + const time_t baseline_sec = static_cast(m_resource_recording_start); + const epoch_t baseline_time = { + baseline_sec, + static_cast(m_resource_recording_start - baseline_sec)}; + const auto baseline_sample = + m_has_resource_recording_baseline + ? m_resource_recording_baseline + : this->sample(baseline_time, m_ctx.m_n_nodes_in_use, + m_resource_trace_current_capacity); + const auto allocated = baseline_sample.allocated; + const auto capacity = baseline_sample.capacity; + const sim_time_t baseline_output_time = + convert_epoch(baseline_sample.time); + std::string baseline = + format_sim_time(baseline_output_time, msec) + "," + + std::to_string(allocated < capacity ? capacity - allocated : 0) + "," + + std::to_string(allocated) + Policy::resource_values(baseline_sample) + + "\n"; m_resource_trace_ofs << baseline; + m_has_resource_recording_baseline = false; } template @@ -732,7 +788,9 @@ template void BasicTrace::flush_resource_history() { buf += format_sim_time(convert_epoch(sample.time), m_resource_trace_msec) + "," + - std::to_string(m_resource_trace_total_nodes - sample.allocated) + + std::to_string(sample.allocated < sample.capacity + ? sample.capacity - sample.allocated + : 0) + "," + std::to_string(sample.allocated) + Policy::resource_values(sample) + "\n"; if (buf.size() >= blk_sz) { @@ -756,7 +814,17 @@ void BasicTrace::record_resource_sample(const epoch_t &time, // in one batch, rather than reclaiming one at a time. flush_resource_history(); } - m_ctx.m_resource_history.push_back(this->sample(time, allocated)); + m_ctx.m_resource_history.push_back( + this->sample(time, allocated, m_resource_trace_current_capacity)); +} + +template +void BasicTrace::set_resource_capacity(sim_time_t time, + num_nodes_t capacity) { + m_resource_trace_current_capacity = capacity; + const time_t sec = static_cast(time); + record_resource_sample({sec, static_cast(time - sec)}, + m_ctx.m_n_nodes_in_use); } template @@ -990,7 +1058,8 @@ void BasicTrace::start_simulated_trace(const std::string &filename, template void BasicTrace::write_simulated_trace(const std::string &filename, - bool msec) { + bool msec, + sim_time_t completed_through) { if (filename.empty()) { return; } @@ -1004,7 +1073,10 @@ void BasicTrace::write_simulated_trace(const std::string &filename, size_t job_no = m_num_reclaimed; for (const auto &job : m_data) { if (job_no == m_next_job_to_write) { - write_job_line(job); + if (job.is_scheduled() && + convert_epoch(job.get_end_time()) <= completed_through) { + write_job_line(job); + } ++m_next_job_to_write; } else if (job_no > m_next_job_to_write) { throw std::logic_error( diff --git a/src/trace/trace.hpp b/src/trace/trace.hpp index 34834e3..e4e97f9 100644 --- a/src/trace/trace.hpp +++ b/src/trace/trace.hpp @@ -16,6 +16,7 @@ #include #include #include +#include #include #include #include @@ -201,6 +202,16 @@ template class BasicTrace : private Policy { std::ofstream m_resource_trace_ofs; /// Cluster size used to derive free nodes in resource-trace output. num_nodes_t m_resource_trace_total_nodes; + /// Capacity effective at the current resource-history timestamp. + num_nodes_t m_resource_trace_current_capacity; + /// Whether the initial resource capacity has been installed. + bool m_resource_capacity_initialized; + /// Timestamp used for the first row when resource recording begins. + sim_time_t m_resource_recording_start; + /// Policy-aware snapshot retained when warm output is opened after the run. + resource_sample_t m_resource_recording_baseline; + /// Whether m_resource_recording_baseline should replace the default row. + bool m_has_resource_recording_baseline; /// Whether resource-trace timestamps are rendered in milliseconds. bool m_resource_trace_msec; @@ -230,7 +241,8 @@ template class BasicTrace : private Policy { /** @brief Construct a trace with explicit input and timestamp formats. * @param[in] fname Input path. * @param[in] format Trace format name. - * @param[in] timestamp_format Timestamp encoding name. + * @param[in] timestamp_format Retained epoch/iso compatibility setting; + * input timestamp encoding is auto-detected. * @param[in] timezone Default timezone for timestamps without offsets. */ BasicTrace(const std::string &fname, const std::string &format, const std::string ×tamp_format, const std::string &timezone); @@ -353,9 +365,13 @@ template class BasicTrace : private Policy { * time,free_nodes,allocated_nodes resource-occupancy trace; no * such file is written if left empty. * @param[in] total_nodes Pool size used only to derive free_nodes above. + * @param[in] max_time Inclusive event-time boundary. The default drains + * the trace completely. */ - void run_job_trace(const std::string &resource_trace_file = std::string(), - num_nodes_t total_nodes = static_cast(0u)); + void run_job_trace( + const std::string &resource_trace_file = std::string(), + num_nodes_t total_nodes = static_cast(0u), + sim_time_t max_time = std::numeric_limits::max()); /** * @brief Record that the scheduler has started an existing job. @@ -371,6 +387,9 @@ template class BasicTrace : private Policy { */ void insert_job(job_no_t job_idx, sim_time_t start_time); + /** Enqueue the recorded begin/end events of one replay-format job. */ + void enqueue_replay_job(job_no_t job_idx); + /** * @brief Append a genuinely new job to m_data - the real streaming * insertion point (unlike insert_job()/submit_job(), which both @@ -580,6 +599,21 @@ template class BasicTrace : private Policy { void start_resource_trace(const std::string &filename, num_nodes_t total_nodes, bool msec = false); + /** + * Discard resource samples before a warm-start boundary while preserving + * live occupancy and policy state. The next resource trace begins with a + * baseline at start_time. + */ + void reset_resource_recording(sim_time_t start_time) { + m_ctx.m_resource_history.clear(); + m_resource_recording_start = start_time; + const time_t sec = static_cast(start_time); + m_resource_recording_baseline = + this->sample({sec, static_cast(start_time - sec)}, + m_ctx.m_n_nodes_in_use, m_resource_trace_current_capacity); + m_has_resource_recording_baseline = true; + } + /** * @brief Write this Trace's recorded resource-occupancy history to a * CSV file (same "time,free_nodes,allocated_nodes" format used by the @@ -603,6 +637,23 @@ template class BasicTrace : private Policy { void write_resource_trace(const std::string &filename, num_nodes_t total_nodes, bool msec = false); + /** + * Record a time-varying capacity transition in the resource history. + * Running allocation is unchanged; free nodes are clamped to zero when a + * non-preemptive reduction temporarily leaves the system overcommitted. + */ + void set_resource_capacity(sim_time_t time, num_nodes_t capacity); + + /** + * Reset the effective resource capacity without recording a transition. + * Simulation uses this at the start of each run so a reused Trace does not + * retain the previous run's final scheduled capacity. + */ + void reset_resource_capacity(num_nodes_t capacity) { + m_resource_trace_current_capacity = capacity; + m_resource_capacity_initialized = true; + } + /** * @brief Open filename early so a job's line gets written the moment * it's reclaimed from m_data, rather than only at the very end - same @@ -630,8 +681,13 @@ template class BasicTrace : private Policy { * @param[in] filename Output path; no-op if empty * @param[in] msec Format timestamps with millisecond precision instead of * truncating to whole seconds + * @param[in] completed_through Include only jobs completed at or before + * this time. The default includes every scheduled job. */ - void write_simulated_trace(const std::string &filename, bool msec = false); + void write_simulated_trace( + const std::string &filename, bool msec = false, + sim_time_t completed_through = + std::numeric_limits::max()); /** * @brief Explicitly write and reclaim the completed front prefix. diff --git a/src/trace/trace_policy.hpp b/src/trace/trace_policy.hpp index 9048ea5..973e86a 100644 --- a/src/trace/trace_policy.hpp +++ b/src/trace/trace_policy.hpp @@ -51,12 +51,14 @@ class Pcon_Job_Record : public Job_Record { struct Standard_Resource_Sample { epoch_t time; num_nodes_t allocated; + num_nodes_t capacity; }; /** Resource-history entry emitted by the Pcon experiment. */ struct Pcon_Resource_Sample { epoch_t time; num_nodes_t allocated; + num_nodes_t capacity; Pcon_Values pcon; }; @@ -74,9 +76,9 @@ struct Standard_Trace_Policy { return record_type(submit_time, num_nodes, queue, limit_time); } - static resource_sample_type sample(const epoch_t &time, - num_nodes_t allocated) { - return {time, allocated}; + static resource_sample_type sample(const epoch_t &time, num_nodes_t allocated, + num_nodes_t capacity) { + return {time, allocated, capacity}; } static const char *resource_columns() { return ""; } static std::string resource_values(const resource_sample_type &) { @@ -100,9 +102,9 @@ struct Pcon_Trace_Policy { return record_type(Job_Record(submit_time, num_nodes, queue, limit_time)); } - resource_sample_type sample(const epoch_t &time, - num_nodes_t allocated) const { - return {time, allocated, m_current}; + resource_sample_type sample(const epoch_t &time, num_nodes_t allocated, + num_nodes_t capacity) const { + return {time, allocated, capacity, m_current}; } static const char *resource_columns() { return ",avgpcon,minpcon,maxpcon"; } static std::string resource_values(const resource_sample_type &sample) { diff --git a/tests/README.md b/tests/README.md index a006e88..3006131 100644 --- a/tests/README.md +++ b/tests/README.md @@ -44,6 +44,9 @@ all registered with CTest: ./tests/run_progressive_load_tests.sh ./tests/run_configs_tests.sh ./tests/run_python_tests.sh +./tests/run_warm_start_validation_tests.sh \ + "${CMAKE_INSTALL_PREFIX}/bin/simulator" +./tests/run_max_time_tests.sh "${CMAKE_INSTALL_PREFIX}/bin/simulator" ./tests/run_grpc_tests.sh ./tests/run_backfill_window_grpc_test.sh python3 tests/test_grpc_single_coordinator.py \ @@ -59,31 +62,34 @@ or are reported as skipped. | Category | Count | Runner or registration | Coverage | |---|---:|---|---| | Scheduler correctness | 34 | `run_scheduler_correctness_tests.sh` | C++/Python schedule and resource-trace consistency | -| Custom FCFS | 7 | `run_custom_scheduler_tests.sh` | Five focused API checks plus two golden schedules, including 2,000 jobs | +| Custom FCFS | 8 | `run_custom_scheduler_tests.sh` | Six focused API checks plus two golden schedules, including warm-start accounting and 2,000 jobs | | Queue implementation differential | 34 × 4 | `run_fcfs_queue_implementation_tests.sh --correctness` | Equivalent schedules across deque, multimap, block, and circular queues | | Column aliases | 8 | `run_column_alias_tests.sh` | Accepted runtime-column aliases and missing-column rejection | | Run-time mode | 7 | `run_time_mode_tests.sh` | Actual, limit, distribution, capping, and planning behavior | | Unit | 7 | `run_unit_tests.sh` | Basic parsing, formats, and execution | -| Feature | 6 | `run_feature_tests.sh` | Policies, modes, rejection, and output formats | +| Feature | 8 | `run_feature_tests.sh` | Policies, modes, rejection, output formats, time-varying capacity, and replay-based warm start | | Scale | 7 | `run_scale_tests.sh` | Workloads from 10 to 10,000 jobs | | Conservative backfilling | 2 | two conservative runners above | Behavioral and C++/Python comparisons | | Replay | 5 | `run_replay_tests.sh` | Resource equivalence and reclamation safety | | Resource history | 5 | `run_resource_history_tests.sh` | Circular-buffer output and capacity handling | | Job store | 6 | `run_job_store_tests.sh` | Capacity, growth/abort, reclamation, and statistics | -| Append-job | 22 | `run_append_job_tests.sh` | 19 in-process C++ checks plus 3 optional gRPC checks | +| Append-job | 25 | `run_append_job_tests.sh` | 20 in-process C++ checks plus 5 optional gRPC checks, including capacity-aware instantaneous/aggregate utilization, warm start, and validation | | Progressive loading | 15 | `run_progressive_load_tests.sh` | 11 C++ checks plus 4 CLI checks for multi-file loading, bounded storage, and memory checks | -| Protobuf configuration | 9 | `run_configs_tests.sh` | Configuration/CLI parity and documented examples | -| Python API | 17 | `run_python_tests.sh` | Bindings, callbacks, streaming, monitoring, and policy APIs | +| Protobuf configuration | 12 | `run_configs_tests.sh` | Configuration/CLI parity, capacity/simulation-start-time validation, and documented examples | +| Python API | 18 | `run_python_tests.sh` | Bindings, callbacks, streaming, monitoring, policy APIs, and warm-start execution | | gRPC client/server | 2 | `run_grpc_tests.sh` | Single-pair and optional MPI multi-server behavior | -| Backfill-window gRPC | 3 repeated checks | `run_backfill_window_grpc_test.sh` | Focused rerun of the gRPC streaming binary; one check targets the backfill window | +| Backfill-window gRPC | 5 repeated checks | `run_backfill_window_grpc_test.sh` | Focused rerun of the gRPC streaming binary; one check targets the backfill window | | Single-coordinator gRPC | 1 | `test_grpc_single_coordinator.py` | Synchronized independent simulation servers | -| Queue input schema | 1 binary | CTest or installed `test_queue_input` | Legacy queue names or numeric queue IDs | +| Queue input schema | 1 binary | CTest or installed `test_queue_input` | Legacy queue names or numeric queue IDs, plus accepted and rejected replay/simulation runtime invariants | | Ser20-disabled serialization | 2 binaries | `t_state_rngen` and `t_state` | Native state serialization without Ser20 | -| Native CTest | 12, plus 1 with MPI | CTest | RNG and binary serialization, trace policies, replay reclamation, custom scheduling, append/streaming APIs, queue implementations, and CLI dispatch | +| Trace tools | 3 | `test_trace_tools.py` via CTest | Capacity inference, simulator-format conversion, direct schedule loading, and warm-start boundary/output behavior | +| Maximum time | 3 CLI cases | `run_max_time_tests.sh` | Inclusive cutoff behavior in simulation, replay, and warm-start execution | +| Warm start | 1 native binary + 9 CLI cases | CTest (`test_warm_start`, `test_warm_start_validation`) | Boundary classification, two-stage execution, runtime modes, capacity transitions, policies/queues, accounting/output, numeric/ISO simulation-start times, zero-start replay, inclusive maximum time, per-file timestamp encoding, and invalid configurations | +| Native CTest | 16, plus 1 with MPI | CTest | RNG and binary serialization, trace policies, replay reclamation, custom scheduling, append/streaming APIs, maximum-time and warm-start coverage, capacity parsing, queue implementations, and CLI dispatch; CTest also registers the Python trace-tools test | The gRPC portion of the append-job runner is skipped when gRPC support was not -built. The backfill-window runner executes the same three-case gRPC test binary -as the append-job runner, so it is a focused rerun rather than three additional +built. The backfill-window runner executes the same five-check gRPC test binary +as the append-job runner, so it is a focused rerun rather than five additional unique checks. Counts describe the checks performed by each runner; the native CTest and focused-runner rows intentionally overlap. @@ -92,6 +98,7 @@ CTest and focused-runner rows intentionally overlap. - `test_traces/scheduler_correctness/`: small FCFS/EASY comparison fixtures; - `test_traces/unit/`: parsing and basic execution fixtures; - `test_traces/feature/`: policy, replay, and buffer fixtures; +- `test_traces/tools/`: capacity-analysis and warm-start fixtures; - `test_traces/scale/`: larger workloads; and - `test_configs/`: Protobuf configuration fixtures. @@ -161,16 +168,43 @@ ${CMAKE_INSTALL_PREFIX}/bin/tests/test_batch_vs_streaming The [append-job runner](run_append_job_tests.sh) provides direct C++ and gRPC `append_job()` coverage. Its C++ checks also validate Custom-FCFS -time-accounted resource area and prediction-horizon estimation, including -multiple same-time allocations and releases, successive completion-event -boundaries, the post-replay full-capacity tail, the `U=0` fallback, future-job -exclusion, and invalid inputs. +time-accounted resource area, capacity-aware instantaneous and aggregate +utilization (including a non-preemptive overcommit drain), and +prediction-horizon estimation. The cases include multiple same-time +allocations and releases, successive completion-event boundaries, the +post-replay full-capacity tail, the `U=0` fallback, future-job exclusion, and +invalid inputs. The warm-start binary independently checks integrated +effective capacity across capacity changes at and after the start boundary. [`test_batch_vs_streaming.cpp`](test_batch_vs_streaming.cpp) compares batch and incremental execution, while [`run_progressive_load_tests.sh`](run_progressive_load_tests.sh) covers the separate progressive-file input path. The runnable Python streaming example is [`python/example_streaming.py`](../python/example_streaming.py). +### Warm-start tests + +```bash +ctest --test-dir build --output-on-failure \ + -R '^(test_warm_start|test_warm_start_validation)$' +``` + +[`test_warm_start.cpp`](test_warm_start.cpp) verifies the replay-based two-stage +path: historical jobs seed occupancy without entering scheduled-job accounting, +ordinary jobs follow the selected runtime policy, and execution switches to +the ordinary event loop after the last warmup job completes. Its cases cover +all supported priority/backfill/queue combinations, randomized differential +workloads, fractional and simultaneous events, empty histories and tails, +capacity overcommit, a capacity change at the final warmup departure, the +zero-start full-replay path, and inclusive `max_time` cutoffs in simulation, +replay, and warm-start modes. The `actual`, `limit`, and deterministic +`distribution` cases confirm that runtime selection remains independent of +warm-start classification. + +[`run_warm_start_validation_tests.sh`](run_warm_start_validation_tests.sh) +checks CLI rejection of negative and non-finite start times, invalid maximum +times, non-replay input, and progressive file lists. It requires the simulator +path when invoked directly; CTest supplies that argument automatically. + ### Python API tests After building with `DR_EVT_BUILD_PYTHON=ON`, run: @@ -181,7 +215,7 @@ After building with `DR_EVT_BUILD_PYTHON=ON`, run: [`test_python_api.py`](test_python_api.py) exercises configuration, single and batched job append, -time advancement, statistics, output, and error handling. +time advancement, statistics, warm-start execution, output, and error handling. ### Distributed client/server tests diff --git a/tests/run_append_job_tests.sh b/tests/run_append_job_tests.sh index b238742..5f1b66f 100755 --- a/tests/run_append_job_tests.sh +++ b/tests/run_append_job_tests.sh @@ -8,10 +8,10 @@ # sitting in a preloaded m_data. See # docs/dev/OUTPUT_TRACE_BUFFERS.md for the design. # -# The in-process binary contains 19 focused append, batch, capacity, +# The in-process binary contains 20 focused append, batch, capacity, # advancement, accounting, and memory-pressure checks. When gRPC is built, -# the runner also executes three wire-level checks covering AppendJobRequest, -# AppendJobsRequest, and GetBackfillWindowRequest against a real server. +# the runner also executes wire-level append, monitoring, and warm-start batch +# checks against a real server. set -e @@ -46,7 +46,7 @@ echo "" PASS=0 FAIL=0 -# --- Test: C++ API (test_append_job_api.cpp's 19 focused checks) --- +# --- Test: C++ API (test_append_job_api.cpp's 20 focused checks) --- echo "Testing: append_job_api (C++ level)" # Test binaries are installed under bin/tests/ (see CMakeLists.txt's diff --git a/tests/run_configs_tests.sh b/tests/run_configs_tests.sh index 697b8ca..e0e81bd 100755 --- a/tests/run_configs_tests.sh +++ b/tests/run_configs_tests.sh @@ -256,6 +256,114 @@ else FAIL=$((FAIL + 1)) fi fi + +# Test 10: capacity_schedule is available through protobuf field 30, and a +# later CLI option overrides the config value just like the other options. +echo "Test 10: capacity schedule config and CLI precedence" +CAPACITY_TRACE="tests/test_traces/feature/capacity_schedule.csv" +CAPACITY_CONFIG="$TEST_WORK_DIR/capacity_config.pb" +CAPACITY_OVERRIDE="$TEST_WORK_DIR/capacity_override.csv" +cat > "$CAPACITY_CONFIG" <<'EOF' +total_nodes: 100 +trace_format: "simple" +timestamp_format: "epoch" +run_time_mode: "limit" +capacity_schedule: "tests/test_traces/feature/capacity_schedule.capacity.csv" +EOF +cat > "$CAPACITY_OVERRIDE" <<'EOF' +time,total_nodes +1000,25 +EOF + +$SIMULATOR "$CAPACITY_TRACE" \ + --total_nodes 100 \ + --trace_format simple \ + --timestamp_format epoch \ + --run_time_mode limit \ + --capacity_schedule tests/test_traces/feature/capacity_schedule.capacity.csv \ + --outfile "$TEST_WORK_DIR/cli_capacity.csv" + +$SIMULATOR "$CAPACITY_TRACE" \ + --config "$CAPACITY_CONFIG" \ + --outfile "$TEST_WORK_DIR/pb_capacity.csv" + +$SIMULATOR "$CAPACITY_TRACE" \ + --total_nodes 100 \ + --trace_format simple \ + --timestamp_format epoch \ + --run_time_mode limit \ + --capacity_schedule "$CAPACITY_OVERRIDE" \ + --outfile "$TEST_WORK_DIR/cli_capacity_override.csv" + +$SIMULATOR "$CAPACITY_TRACE" \ + --config "$CAPACITY_CONFIG" \ + --capacity_schedule "$CAPACITY_OVERRIDE" \ + --outfile "$TEST_WORK_DIR/pb_capacity_override.csv" + +if diff -q "$TEST_WORK_DIR/cli_capacity.csv" "$TEST_WORK_DIR/pb_capacity.csv" > /dev/null && \ + diff -q "$TEST_WORK_DIR/cli_capacity_override.csv" "$TEST_WORK_DIR/pb_capacity_override.csv" > /dev/null; then + echo " ✓ capacity schedule config and later CLI override both match" + PASS=$((PASS + 1)) +else + echo " ✗ capacity schedule config or CLI precedence differs" + FAIL=$((FAIL + 1)) +fi + +# Test 11: sim_start_time is available through protobuf field 31 and produces the +# same replay-based warm-start schedule as the CLI. +echo "Test 11: warm-start time config" +WARM_TRACE="tests/test_traces/feature/warm_start_native.csv" +WARM_CONFIG="$TEST_WORK_DIR/warm_start_config.pb" +cat > "$WARM_CONFIG" <<'EOF' +total_nodes: 100 +trace_format: "simple" +timestamp_format: "epoch" +run_time_mode: "actual" +sim_start_time: 50 +EOF + +$SIMULATOR "$WARM_TRACE" \ + --total_nodes 100 \ + --trace_format simple \ + --timestamp_format epoch \ + --run_time_mode actual \ + --sim_start_time 50 \ + --outfile "$TEST_WORK_DIR/cli_warm_start.csv" + +$SIMULATOR "$WARM_TRACE" \ + --config "$WARM_CONFIG" \ + --outfile "$TEST_WORK_DIR/pb_warm_start.csv" + +if diff -q "$TEST_WORK_DIR/cli_warm_start.csv" \ + "$TEST_WORK_DIR/pb_warm_start.csv" > /dev/null; then + echo " ✓ sim_start_time config matches CLI" + PASS=$((PASS + 1)) +else + echo " ✗ sim_start_time config differs from CLI" + FAIL=$((FAIL + 1)) +fi + +# Test 12: protobuf sim_start_time validation rejects every invalid numeric class. +echo "Test 12: invalid simulation-start-time config" +INVALID_START_OK=1 +for VALUE in -1 nan inf; do + INVALID_CONFIG="$TEST_WORK_DIR/invalid_start_${VALUE}.pb" + INVALID_LOG="$TEST_WORK_DIR/invalid_start_${VALUE}.log" + printf 'sim_start_time: %s\n' "$VALUE" > "$INVALID_CONFIG" + if $SIMULATOR "$WARM_TRACE" --config "$INVALID_CONFIG" \ + > "$INVALID_LOG" 2>&1; then + INVALID_START_OK=0 + elif ! grep -q "sim_start_time" "$INVALID_LOG"; then + INVALID_START_OK=0 + fi +done +if [ "$INVALID_START_OK" -eq 1 ]; then + echo " ✓ negative and non-finite sim_start_time values are rejected" + PASS=$((PASS + 1)) +else + echo " ✗ an invalid sim_start_time was accepted or misdiagnosed" + FAIL=$((FAIL + 1)) +fi echo "" echo "==========================================" echo "Results: $PASS passed, $FAIL failed" diff --git a/tests/run_max_time_tests.sh b/tests/run_max_time_tests.sh new file mode 100755 index 0000000..8c87228 --- /dev/null +++ b/tests/run_max_time_tests.sh @@ -0,0 +1,82 @@ +#!/bin/bash +# End-to-end checks for the simulator's inclusive --max_time boundary. + +set -u + +if [ "$#" -ne 1 ]; then + echo "Usage: $0 " >&2 + exit 2 +fi + +SIMULATOR="$1" +WORK_DIR="$(mktemp -d "${TMPDIR:-/tmp}/dr-evt-max-time.XXXXXXXX")" +trap 'rm -rf -- "$WORK_DIR"' EXIT INT TERM + +fail() { + echo "max_time test failed: $1" >&2 + exit 1 +} + +expect_line() { + grep -Fqx -- "$1" "$2" || fail "missing '$1' in $2" +} + +# Ordinary simulation: the t=5 completion and arrival are both processed. +# The second job remains running, and the t=20 arrival remains pending. +cat >"$WORK_DIR/simulation.csv" <<'EOF' +job_submit_time,num_nodes,time_limit,actual_run_time +0,2,5,5 +5,2,10,10 +20,1,1,1 +EOF +"$SIMULATOR" "$WORK_DIR/simulation.csv" --total_nodes 2 \ + --run_time_mode actual --max_time 5 \ + --outfile "$WORK_DIR/simulation.out.csv" \ + --resource_trace "$WORK_DIR/simulation.resources.csv" \ + >"$WORK_DIR/simulation.log" 2>&1 || fail "ordinary simulation exited nonzero" +expect_line "Current time: 5" "$WORK_DIR/simulation.log" +expect_line "Jobs completed: 1" "$WORK_DIR/simulation.log" +[ "$(wc -l <"$WORK_DIR/simulation.out.csv")" -eq 2 ] || \ + fail "ordinary output should contain exactly one completed job" +expect_line "0,0,5,2,0,5" "$WORK_DIR/simulation.out.csv" +[ "$(tail -n 1 "$WORK_DIR/simulation.resources.csv")" = "5,0,2" ] || \ + fail "ordinary simulation did not settle the inclusive t=5 events" + +# Full replay uses recorded begin/end times but observes the same boundary. +cat >"$WORK_DIR/replay.csv" <<'EOF' +job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit +0,0,5,2,0,5 +4,5,10,2,0,5 +11,11,12,1,0,1 +EOF +"$SIMULATOR" "$WORK_DIR/replay.csv" --total_nodes 4 --max_time 5 \ + --outfile "$WORK_DIR/replay.out.csv" \ + --resource_trace "$WORK_DIR/replay.resources.csv" \ + >"$WORK_DIR/replay.log" 2>&1 || fail "replay exited nonzero" +expect_line "Current time: 5" "$WORK_DIR/replay.log" +expect_line "Jobs completed: 1" "$WORK_DIR/replay.log" +[ "$(wc -l <"$WORK_DIR/replay.out.csv")" -eq 2 ] || \ + fail "replay output should contain exactly one completed job" +[ "$(tail -n 1 "$WORK_DIR/replay.resources.csv")" = "5,2,2" ] || \ + fail "replay did not settle the inclusive t=5 events" + +# A limit reached during warmup must stop there without entering an unbounded +# ordinary drain. Both the historical and newly scheduled job are still live. +cat >"$WORK_DIR/warm.csv" <<'EOF' +job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit +0,2,15,2,0,13 +10,20,24,2,0,4 +EOF +"$SIMULATOR" "$WORK_DIR/warm.csv" --total_nodes 4 \ + --sim_start_time 10 --max_time 12 --run_time_mode actual \ + --outfile "$WORK_DIR/warm.out.csv" \ + --resource_trace "$WORK_DIR/warm.resources.csv" \ + >"$WORK_DIR/warm.log" 2>&1 || fail "warm start exited nonzero" +expect_line "Current time: 12" "$WORK_DIR/warm.log" +expect_line "Jobs completed: 0" "$WORK_DIR/warm.log" +[ "$(wc -l <"$WORK_DIR/warm.out.csv")" -eq 1 ] || \ + fail "warm-start output should contain no completed jobs" +[ "$(tail -n 1 "$WORK_DIR/warm.resources.csv")" = "10,0,4" ] || \ + fail "warm-start live occupancy is incorrect at the boundary" + +echo "max_time CLI tests passed" diff --git a/tests/run_warm_start_validation_tests.sh b/tests/run_warm_start_validation_tests.sh new file mode 100755 index 0000000..6e8c974 --- /dev/null +++ b/tests/run_warm_start_validation_tests.sh @@ -0,0 +1,98 @@ +#!/bin/bash +# CLI/runtime rejection tests for invalid warm-start configurations. + +set -u + +if [ "$#" -ne 1 ]; then + echo "Usage: $0 " >&2 + exit 2 +fi + +SIMULATOR="$1" +SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" +REPO_ROOT="$(cd "$SCRIPT_DIR/.." && pwd)" +WORK_DIR="$(mktemp -d "${TMPDIR:-/tmp}/dr-evt-warm-validation.XXXXXXXX")" +trap 'rm -rf -- "$WORK_DIR"' EXIT INT TERM + +PASS=0 +FAIL=0 + +expect_failure() { + name="$1" + expected="$2" + shift 2 + log="$WORK_DIR/$name.log" + if "$@" >"$log" 2>&1; then + echo " ✗ $name unexpectedly succeeded" + FAIL=$((FAIL + 1)) + elif grep -q -- "$expected" "$log"; then + echo " ✓ $name" + PASS=$((PASS + 1)) + else + echo " ✗ $name failed without expected diagnostic: $expected" + sed 's/^/ /' "$log" + FAIL=$((FAIL + 1)) + fi +} + +expect_success() { + name="$1" + shift + log="$WORK_DIR/$name.log" + if "$@" >"$log" 2>&1; then + echo " ✓ $name" + PASS=$((PASS + 1)) + else + echo " ✗ $name failed" + sed 's/^/ /' "$log" + FAIL=$((FAIL + 1)) + fi +} + +cd "$REPO_ROOT" +REPLAY="tests/test_traces/feature/warm_start_native.csv" +SIMULATION="tests/test_traces/unit/simple_basic.csv" + +expect_failure negative_start "must be finite and nonnegative" \ + "$SIMULATOR" "$REPLAY" --sim_start_time -1 +expect_failure nan_start "must be finite and nonnegative" \ + "$SIMULATOR" "$REPLAY" --sim_start_time nan +expect_failure infinite_start "must be finite and nonnegative" \ + "$SIMULATOR" "$REPLAY" --sim_start_time inf +expect_failure negative_max_time "--max_time must be finite and nonnegative" \ + "$SIMULATOR" "$REPLAY" --max_time -1 +expect_failure max_before_start "--max_time must be greater than or equal" \ + "$SIMULATOR" "$REPLAY" --sim_start_time 50 --max_time 49 +expect_failure non_replay_input "requires replay-format input" \ + "$SIMULATOR" "$SIMULATION" --sim_start_time 1 --run_time_mode limit + +printf '%s\n' "$SIMULATION" >"$WORK_DIR/traces.list" +expect_failure infile_list "not supported with --infile_list" \ + "$SIMULATOR" --infile_list "$WORK_DIR/traces.list" --sim_start_time 1 \ + --run_time_mode limit + +ISO_REPLAY="$WORK_DIR/iso_replay.csv" +cat >"$ISO_REPLAY" <<'EOF' +job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit +2024-01-01T00:00:00,2024-01-01T00:00:01,2024-01-01T00:00:20,1,0,19 +2024-01-01T00:00:10,2024-01-01T00:00:12,2024-01-01T00:00:14,1,0,2 +EOF +expect_success iso_sim_start_time \ + "$SIMULATOR" "$ISO_REPLAY" \ + --sim_start_time "2024-01-01T00:00:10" \ + --timezone UTC --run_time_mode actual \ + --total_nodes 2 --outfile "$WORK_DIR/iso_jobs.csv" \ + --resource_trace "$WORK_DIR/iso_resources.csv" + +MIXED_TIMESTAMPS="$WORK_DIR/mixed_timestamps.csv" +cat >"$MIXED_TIMESTAMPS" <<'EOF' +job_submit_time,num_nodes,time_limit +2024-01-01T00:00:00,1,1 +1704067201,1,1 +EOF +expect_failure mixed_timestamp_encoding "Failed to parse time string" \ + "$SIMULATOR" "$MIXED_TIMESTAMPS" --run_time_mode limit \ + --timezone UTC --total_nodes 1 + +echo "Warm-start validation: $PASS passed, $FAIL failed" +test "$FAIL" -eq 0 diff --git a/tests/test_append_job_api.cpp b/tests/test_append_job_api.cpp index 68baacb..2c05dbc 100644 --- a/tests/test_append_job_api.cpp +++ b/tests/test_append_job_api.cpp @@ -61,10 +61,20 @@ bool approx_equal(double a, double b, double tol = 1e-6) { // (that's the whole point - each job only becomes known via // append_job()), so this file is always header-only, zero data rows. const char *EMPTY_TRACE_PATH = "/tmp/test_append_job_api_empty.csv"; +const char *CAPACITY_SCHEDULE_PATH = + "/tmp/test_append_job_api_capacity_schedule.csv"; +const char *CAPACITY_RESET_TRACE_PATH = + "/tmp/test_append_job_api_capacity_reset_resources.csv"; void write_empty_trace_fixture() { std::ofstream ofs(EMPTY_TRACE_PATH); ofs << "job_submit_time,num_nodes,time_limit\n"; + + std::ofstream capacity(CAPACITY_SCHEDULE_PATH); + capacity << "time,total_nodes\n" + << "5,0\n" + << "20,100\n" + << "1000,25\n"; } Sim_Params make_params() { @@ -568,11 +578,86 @@ void test_advance_to_idle_gap() { sim.advance_to(std::numeric_limits::max()); assert(sim.get_nodes_in_use() == 0); + assert(approx_equal(sim.get_current_time(), 550.0)); std::cout << " All jobs complete after draining to infinity" << std::endl; std::cout << " PASSED" << std::endl; } +// A finite streaming advance must consume capacity changes even if no jobs +// exist yet. Capacity zero then pauses a newly appended job until the increase +// at t=20. Once that job finishes, draining an otherwise empty simulation to +// infinity must not consume the irrelevant change at t=1000. +void test_capacity_schedule_across_streaming_advances() { + std::cout << "\n=== Capacity schedule across streaming advances ===" + << std::endl; + + Sim_Params params = make_params(); + params.m_capacity_schedule = CAPACITY_SCHEDULE_PATH; + Simulation sim(params); + sim.get_trace().load_data(0); + + sim.advance_to(10.0); + assert(approx_equal(sim.get_current_time(), 10.0)); + assert(sim.get_current_capacity() == 0); + assert(sim.get_available_nodes() == 0); + + const job_no_t job = sim.append_job(10.0, 30, kTestQueueInput, 10); + sim.advance_to(15.0); + assert(!sim.get_trace().job_at(job).is_scheduled()); + assert(sim.get_nodes_in_use() == 0); + + sim.advance_to(20.0); + assert(sim.get_current_capacity() == 100); + assert(sim.get_trace().job_at(job).is_scheduled()); + assert(sim.get_nodes_in_use() == 30); + + sim.advance_to(30.0); + assert(sim.get_nodes_in_use() == 0); + sim.advance_to(std::numeric_limits::max()); + assert(sim.get_current_capacity() == 100); + + // A finite advance on another idle simulation does consume the transition. + // Reinitializing that same Simulation must reset both Simulation's capacity + // and Trace's capacity used to derive free_nodes in resource samples. + Simulation reused(params); + reused.get_trace().load_data(0); + reused.advance_to(1000.0); + assert(reused.get_current_capacity() == 25); + reused.initialize_trace(); + assert(reused.get_current_capacity() == 100); + const job_no_t reset_job = reused.append_job(0.0, 30, kTestQueueInput, 10); + reused.advance_to(0.0); + assert(reused.get_trace().job_at(reset_job).is_scheduled()); + reused.write_resource_trace(CAPACITY_RESET_TRACE_PATH); + + std::ifstream resource_trace(CAPACITY_RESET_TRACE_PATH); + std::string line; + std::string last_line; + while (std::getline(resource_trace, line)) { + if (!line.empty()) { + last_line = line; + } + } + assert(last_line == "0,70,30"); + + // A non-preemptive reduction below live occupancy uses that occupancy as + // effective capacity until the running work drains. Down nodes therefore + // do not dilute either instantaneous or time-accounted utilization. + Simulation draining(params); + draining.get_trace().load_data(0); + draining.append_job(0.0, 30, kTestQueueInput, 10); + draining.advance_to(5.0); + assert(draining.get_current_capacity() == 0); + assert(draining.get_nodes_in_use() == 30); + assert(approx_equal(draining.get_current_utilization(), 1.0)); + const auto draining_stats = draining.get_statistics(); + assert(approx_equal(draining_stats.resource_area, 300.0)); + assert(approx_equal(draining_stats.utilization, 300.0 / 650.0)); + + std::cout << " PASSED" << std::endl; +} + // Test 15: append_jobs()'s own capacity/overflow handling must be the // only thing that decides grow-vs-abort for a batch - reclaim_front_jobs() // must not also weigh in with its own single-job-shaped fallback (see @@ -921,6 +1006,7 @@ int main() { test_online_scheduling(); test_no_resource_leaks(); test_advance_to_idle_gap(); + test_capacity_schedule_across_streaming_advances(); test_submit_job_records_busy_nodes(); test_append_jobs_memory_pressure(); test_resource_area_and_time_accounted_utilization(); diff --git a/tests/test_capacity_schedule.cpp b/tests/test_capacity_schedule.cpp new file mode 100644 index 0000000..5bd91a4 --- /dev/null +++ b/tests/test_capacity_schedule.cpp @@ -0,0 +1,79 @@ +/****************************************************************************** + * Copyright 2023 Lawrence Livermore National Security, LLC * + * See the top-level LICENSE file for details. * + * * + * SPDX-License-Identifier: MIT * + ******************************************************************************/ + +#define DR_EVT_HAS_CONFIG 1 +#include "sim/capacity_schedule.hpp" +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace dr_evt; + +namespace { + +void write_file(const std::filesystem::path &path, const std::string &text) { + std::ofstream output(path); + output << text; + assert(output.good()); +} + +void expect_invalid(const std::filesystem::path &path, + const std::string &contents) { + write_file(path, contents); + bool threw = false; + try { + (void)load_capacity_schedule(path.string(), 100); + } catch (const std::invalid_argument &) { + threw = true; + } + assert(threw); +} + +} // namespace + +int main() { + const auto suffix = + std::chrono::steady_clock::now().time_since_epoch().count(); + const auto work = std::filesystem::temp_directory_path() / + ("dr_evt_capacity_schedule_" + std::to_string(suffix)); + std::filesystem::create_directory(work); + + try { + assert(load_capacity_schedule("", 100).empty()); + + const auto valid = work / "valid.csv"; + write_file(valid, "ignored,total_nodes,time\nx,0,1\ny,100,2.5\n"); + const auto changes = load_capacity_schedule(valid.string(), 100); + assert(changes.size() == 2); + assert(changes[0].time == 1.0 && changes[0].total_nodes == 0); + assert(changes[1].time == 2.5 && changes[1].total_nodes == 100); + + expect_invalid(work / "empty.csv", ""); + expect_invalid(work / "header_only.csv", "time,total_nodes\n"); + expect_invalid(work / "missing_time.csv", "total_nodes\n10\n"); + expect_invalid(work / "missing_nodes.csv", "time\n1\n"); + expect_invalid(work / "short_row.csv", "time,x,total_nodes\n1,x\n"); + expect_invalid(work / "duplicate.csv", "time,total_nodes\n1,10\n1,20\n"); + expect_invalid(work / "decreasing.csv", "time,total_nodes\n2,10\n1,20\n"); + expect_invalid(work / "malformed.csv", "time,total_nodes\n1,nope\n"); + expect_invalid(work / "negative.csv", "time,total_nodes\n1,-1\n"); + expect_invalid(work / "above_max.csv", "time,total_nodes\n1,101\n"); + } catch (...) { + std::filesystem::remove_all(work); + throw; + } + + std::filesystem::remove_all(work); + std::cout << "Capacity schedule parser tests passed\n"; + return EXIT_SUCCESS; +} diff --git a/tests/test_custom_scheduler.cpp b/tests/test_custom_scheduler.cpp index 923ee88..f83506e 100644 --- a/tests/test_custom_scheduler.cpp +++ b/tests/test_custom_scheduler.cpp @@ -141,6 +141,36 @@ void test_batch_resource_area_accounting() { assert(std::abs(simulation.get_resource_area() - 4600.0) < 1e-12); } +void test_warm_start_resource_area_accounting() { + constexpr const char *trace_path = + "/tmp/dr_evt_custom_scheduler_warm_area.csv"; + { + std::ofstream trace(trace_path); + trace << "job_submit_time,begin_time,end_time,num_nodes,exit_status," + "time_limit\n" + << "0,1,5,2,0,4\n" + << "3,3,4,2,0,1\n"; + } + + Sim_Params params; + params.m_infile = trace_path; + params.m_total_nodes = 4; + params.m_sim_start_time = 3.0; + params.m_trace_format = "simple"; + params.m_timestamp_format = "epoch"; + params.m_run_time_mode = RunTimeMode::ACTUAL; + params.m_num_max_candidates = 2; + + Simulation simulation(params, cost_from_job_order, select_lowest_cost); + simulation.run(); + + const auto stats = simulation.get_statistics(); + assert(stats.jobs_completed == 1); + assert(std::abs(stats.resource_area - 6.0) < 1e-12); + assert(std::abs(stats.utilization - 0.75) < 1e-12); + assert(std::abs(simulation.get_resource_area() - 6.0) < 1e-12); +} + void test_matches_default_circular_easy() { constexpr const char *trace_path = "/tmp/dr_evt_custom_scheduler_equivalence.csv"; @@ -227,6 +257,7 @@ int main(int argc, char **argv) { test_selector_must_return_a_candidate(); test_current_utilization_api(); test_batch_resource_area_accounting(); + test_warm_start_resource_area_accounting(); test_matches_default_circular_easy(); std::cout << "Custom scheduler tests passed\n"; return 0; diff --git a/tests/test_grpc_streaming_api.cpp b/tests/test_grpc_streaming_api.cpp index 7950fac..3f92405 100644 --- a/tests/test_grpc_streaming_api.cpp +++ b/tests/test_grpc_streaming_api.cpp @@ -19,8 +19,10 @@ #include #include #include +#include #include #include +#include #include #include @@ -354,6 +356,107 @@ bool test_backfill_window(const std::string &server_address, return true; } +// Test 4: execute a replay-format warm start through the actual service. This +// uses RunRequest because InitializeTrace only loads records for streaming +// callers; it intentionally does not execute batch initialization semantics. +bool test_warm_start(const std::string &server_address, + const std::string &trace_file) { + std::cout << "=== Test: warm-start RunRequest ===\n"; + + // The service validates values independently of the CLI/config parsers. + for (const double invalid_value : + {-1.0, std::numeric_limits::quiet_NaN(), + std::numeric_limits::infinity()}) { + SimulationClient invalid(grpc::CreateChannel( + server_address, grpc::InsecureChannelCredentials())); + ClientMessage request; + auto *init = request.mutable_init(); + init->set_total_nodes(10); + init->set_trace_format("simple"); + init->set_timestamp_format("epoch"); + init->set_run_time_mode("actual"); + init->set_infile(trace_file); + init->set_session_name("invalid-warm-start"); + init->set_sim_start_time(invalid_value); + bool rejected = false; + try { + invalid.call(request); + } catch (const std::runtime_error &e) { + rejected = + std::string(e.what()).find("sim_start_time") != std::string::npos; + } + if (!rejected) { + std::cerr << " FAIL: invalid gRPC sim_start_time was not rejected\n"; + invalid.finish(); + return false; + } + invalid.finish(); + } + + SimulationClient client( + grpc::CreateChannel(server_address, grpc::InsecureChannelCredentials())); + try { + ClientMessage init_request; + auto *init = init_request.mutable_init(); + init->set_total_nodes(10); + init->set_trace_format("simple"); + init->set_timestamp_format("epoch"); + init->set_backfill_policy("easy"); + init->set_priority_policy("fcfs"); + init->set_run_time_mode("actual"); + init->set_infile(trace_file); + init->set_session_name("warm-start-test"); + init->set_sim_start_time(10.0); + client.call(init_request); + + ClientMessage run_request; + run_request.mutable_run(); + client.call(run_request); + + ClientMessage stats_request; + stats_request.mutable_get_statistics(); + const auto stats_response = client.call(stats_request); + const auto &stats = stats_response.get_statistics(); + const bool correct = + stats.jobs_submitted() == 2 && stats.jobs_completed() == 2 && + stats.jobs_running() == 0 && stats.jobs_waiting() == 0 && + stats.total_nodes() == 10 && stats.nodes_in_use() == 0 && + stats.nodes_available() == 10 && + std::fabs(stats.resource_area() - 51.0) < 1e-12 && + std::fabs(stats.utilization() - 0.85) < 1e-12 && + std::fabs(stats.avg_wait_time() - 1.5) < 1e-12 && + std::fabs(stats.avg_turnaround_time() - 4.5) < 1e-12 && + std::fabs(stats.makespan() - 16.0) < 1e-12; + if (!correct) { + std::cerr << " FAIL: warm-start statistics did not match native run\n"; + client.finish(); + return false; + } + + ClientMessage finish_request; + finish_request.mutable_finish_simulation(); + const auto finish = client.call(finish_request).finish_simulation(); + if (finish.statistics().jobs_submitted() != 2 || + finish.statistics().jobs_completed() != 2) { + std::cerr << " FAIL: warm-start finish statistics changed\n"; + client.finish(); + return false; + } + } catch (const std::exception &e) { + std::cerr << " FAIL: " << e.what() << "\n"; + client.finish(); + return false; + } + + const grpc::Status status = client.finish(); + if (!status.ok()) { + std::cerr << " FAIL: RPC failed: " << status.error_message() << "\n"; + return false; + } + std::cout << " PASSED\n"; + return true; +} + int main(int argc, char **argv) { if (argc < 3) { std::cerr << "Usage: " << argv[0] @@ -361,16 +464,20 @@ int main(int argc, char **argv) { << " must have a valid header and zero data " "rows -\n" << " this test's whole point is appending jobs the server never " - "loaded.\n"; + "loaded. warm_start_grpc.csv must be in the same directory.\n"; return 1; } std::string server_address = argv[1]; std::string trace_file = argv[2]; + const std::string warm_trace = + (std::filesystem::path(trace_file).parent_path() / "warm_start_grpc.csv") + .string(); bool ok = true; ok &= test_single_append(server_address, trace_file); ok &= test_batch_append(server_address, trace_file); ok &= test_backfill_window(server_address, trace_file); + ok &= test_warm_start(server_address, warm_trace); if (!ok) { std::cerr << "SOME TESTS FAILED\n"; diff --git a/tests/test_pcon_trace.cpp b/tests/test_pcon_trace.cpp index 32763cf..f3ea56c 100644 --- a/tests/test_pcon_trace.cpp +++ b/tests/test_pcon_trace.cpp @@ -22,6 +22,10 @@ constexpr const char *simulation_input_path = "/tmp/dr_evt_pcon_simulation_input.csv"; constexpr const char *simulation_output_path = "/tmp/dr_evt_pcon_simulation_output.csv"; +constexpr const char *warm_input_path = "/tmp/dr_evt_pcon_warm_input.csv"; +constexpr const char *warm_resource_path = + "/tmp/dr_evt_pcon_warm_resources.csv"; +constexpr const char *warm_job_path = "/tmp/dr_evt_pcon_warm_jobs.csv"; bool expect_line(std::istream &input, const std::string &expected) { std::string actual; @@ -69,6 +73,61 @@ int main() { passed = false; } + { + std::ofstream input(warm_input_path); + input << "job_submit_time,begin_time,end_time,num_nodes,exit_status,q_id," + "time_limit,avgpcon,minpcon,maxpcon\n" + << "0,1,5,2,0,1,4,1.5,2,3\n" + << "3,3,4,2,0,1,1,0.5,1,1.5\n"; + } + + try { + dr_evt::Sim_Params params; + params.m_infile = warm_input_path; + params.m_trace_type = dr_evt::TraceType::PCON; + params.m_total_nodes = 4; + params.m_sim_start_time = 3; + params.m_trace_format = "simple"; + params.m_timestamp_format = "epoch"; + params.m_run_time_mode = dr_evt::RunTimeMode::ACTUAL; + params.set_outfile(warm_job_path); + + dr_evt::PconSimulation simulation(params); + simulation.run(); + const auto stats = simulation.get_statistics(); + if (stats.jobs_completed != 1 || stats.resource_area != 6.0 || + stats.utilization != 0.75) { + std::cerr << "warm-start accounting mismatch: completed=" + << stats.jobs_completed << " area=" << stats.resource_area + << " utilization=" << stats.utilization << '\n'; + passed = false; + } + simulation.write_simulated_trace(); + simulation.write_resource_trace(warm_resource_path); + + std::ifstream resources(warm_resource_path); + passed &= expect_line( + resources, "time,free_nodes,allocated_nodes,avgpcon,minpcon,maxpcon"); + passed &= expect_line(resources, "3,2,2,1.500000,2.000000,3.000000"); + passed &= expect_line(resources, "3,0,4,2.000000,3.000000,4.500000"); + passed &= expect_line(resources, "4,2,2,1.500000,2.000000,3.000000"); + passed &= expect_line(resources, "5,4,0,0.000000,0.000000,0.000000"); + + std::ifstream jobs(warm_job_path); + passed &= expect_line( + jobs, "job_submit_time,begin_time,end_time,num_nodes,exit_status,q_id," + "time_limit"); + passed &= expect_line(jobs, "3,3,4,2,0,1,1"); + std::string unexpected; + if (std::getline(jobs, unexpected)) { + std::cerr << "unexpected warm-start job line: " << unexpected << '\n'; + passed = false; + } + } catch (const std::exception &error) { + std::cerr << error.what() << '\n'; + passed = false; + } + { std::ofstream input(simulation_input_path); input << "job_submit_time,num_nodes,q_id,time_limit,avgpcon,minpcon," @@ -108,5 +167,8 @@ int main() { std::remove(output_path); std::remove(simulation_input_path); std::remove(simulation_output_path); + std::remove(warm_input_path); + std::remove(warm_resource_path); + std::remove(warm_job_path); return passed ? EXIT_SUCCESS : EXIT_FAILURE; } diff --git a/tests/test_python_api.py b/tests/test_python_api.py index 806615e..b772138 100755 --- a/tests/test_python_api.py +++ b/tests/test_python_api.py @@ -77,6 +77,20 @@ def create_test_trace(filename, jobs): f.write(f"{job[0]},{job[1]},{job[2]}\n") +def create_replay_trace(filename): + """Create a replay trace spanning a nonzero warm-start boundary.""" + with open(filename, 'w') as f: + f.write("job_submit_time,begin_time,end_time,num_nodes," + "exit_status,time_limit\n") + f.write("0,0,5,2,0,5\n") # completed history + f.write("1,2,10,3,0,8\n") # ends exactly at t + f.write("2,4,15,2,0,11\n") # active history + f.write("3,5,15,3,0,10\n") # active history + f.write("4,12,14,1,0,2\n") # inherited waiter: excluded + f.write("10,10,14,4,0,4\n") # boundary arrival + f.write("11,12,14,5,0,2\n") # future arrival + + def test_module_import(result): """Test 1: Module import and version""" print("\n1. Module Import") @@ -550,6 +564,49 @@ def test_batch_mode(result): os.unlink(trace_file.name) +def test_warm_start_batch_mode(result): + """Python run() exposes replay-based warm-start classification/accounting.""" + print("\n10. Warm-start Batch Mode") + + trace_file = tempfile.NamedTemporaryFile(mode='w', suffix='.csv', + delete=False) + trace_file.close() + try: + create_replay_trace(trace_file.name) + params = dr_evt.SimParams() + params.infile = trace_file.name + params.total_nodes = 10 + params.sim_start_time = 10.0 + params.trace_format = "simple" + params.timestamp_format = "epoch" + params.run_time_mode = dr_evt.RunTimeMode.ACTUAL + + sim = dr_evt.Simulation(params) + sim.run() + stats = sim.get_statistics() + + # Only the boundary/future workload is counted. Completed history, + # the exact-boundary departure, active seeds, and inherited waiting + # work are all absent from job accounting. + assert stats.jobs_submitted == 2 + assert stats.jobs_completed == 2 + assert stats.jobs_running == 0 + assert stats.jobs_waiting == 0 + assert stats.total_nodes == 10 + assert stats.nodes_in_use == 0 + assert stats.nodes_available == 10 + assert abs(stats.resource_area - 51.0) < 1e-12 + assert abs(stats.utilization - 0.85) < 1e-12 + assert abs(stats.avg_wait_time - 1.5) < 1e-12 + assert abs(stats.avg_turnaround_time - 4.5) < 1e-12 + assert abs(stats.makespan - 16.0) < 1e-12 + result.record_pass("Native replay warm start") + except Exception as e: + result.record_fail("Warm-start batch mode", str(e)) + finally: + os.unlink(trace_file.name) + + def main(): print("="*60) print("DR_EVT Python API Test Suite") @@ -570,6 +627,7 @@ def main(): test_backfill_policies(result) test_priority_policies(result) test_batch_mode(result) + test_warm_start_batch_mode(result) # Print summary and exit return result.summary() diff --git a/tests/test_queue_input.cpp b/tests/test_queue_input.cpp index b6a29d1..d02612c 100644 --- a/tests/test_queue_input.cpp +++ b/tests/test_queue_input.cpp @@ -17,6 +17,7 @@ #include #include +#include #include #include #include @@ -158,6 +159,40 @@ void parse_and_check(const std::string &format, bool replay, } } +std::vector load_simple(const std::string &name, + const std::string &contents) { + const std::string path = "/tmp/test_runtime_validation_" + name + ".csv"; + write_file(path, contents); + Data_Columns columns("simple", "epoch", "UTC"); + assert(columns.check_header(path)); + std::vector records; + assert(load(path, columns, records) == EXIT_SUCCESS); + return records; +} + +void test_input_runtime_validation() { + const auto replay = + load_simple("replay", + "job_submit_time,begin_time,end_time,num_nodes,time_limit," + "actual_run_time\n" + "0,0.1,0.3,1,1,0.2\n" // equal within timestamp precision + "0,1,3,1,2,2.1\n" // greater than observed interval + "0,1,3,1,2,1.9\n"); // less than observed interval + assert(replay.size() == 1u); + assert(std::fabs(replay.front().get_actual_run_time() - 0.2) < 1.0e-12); + + const auto simulation = + load_simple("simulation", + "job_submit_time,num_nodes,time_limit,actual_run_time\n" + "0,1,10,9\n" // below the requested limit + "1,1,10,10\n" // equal to the requested limit + "2,1,10,11\n" // exceeds the requested limit + "3,1,10,nan\n"); + assert(simulation.size() == 2u); + assert(simulation[0].get_actual_run_time() == 9.0); + assert(simulation[1].get_actual_run_time() == 10.0); +} + } // namespace int main() { @@ -173,6 +208,7 @@ int main() { parse_and_check(format, replay, false); } } + test_input_runtime_validation(); std::cout << "queue input parser tests passed\n"; return EXIT_SUCCESS; } diff --git a/tests/test_trace_tools.py b/tests/test_trace_tools.py new file mode 100644 index 0000000..33ef0c9 --- /dev/null +++ b/tests/test_trace_tools.py @@ -0,0 +1,176 @@ +#!/usr/bin/env python3 +"""Exact-output checks for the capacity-analysis and warm-start helpers.""" + +import csv +import json +import os +from pathlib import Path +import subprocess +import sys +import tempfile + + +ROOT = Path(__file__).resolve().parents[1] + + +def run(*args): + return subprocess.run( + [sys.executable, *map(str, args)], + cwd=ROOT, + check=True, + capture_output=True, + text=True, + ) + + +def read_rows(path): + with path.open(newline="") as stream: + return list(csv.DictReader(stream)) + + +def test_warm_start(work): + output = work / "warm.csv" + result = run( + ROOT / "scripts/prepare_warm_start_trace.py", + ROOT / "tests/test_traces/tools/warm_start_history.csv", + output, + "--workload-trace", + ROOT / "tests/test_traces/tools/warm_start_workload.csv", + "--start-time", + "10", + "--total-nodes", + "100", + ) + assert "initialization_jobs=3\n" in result.stdout + assert read_rows(output) == [ + {"job_submit_time": "10", "num_nodes": "3", "time_limit": "10", + "actual_run_time": "10", "initialization": "1", + "source_job_index": "1"}, + {"job_submit_time": "10", "num_nodes": "1", "time_limit": "1", + "actual_run_time": "1", "initialization": "1", + "source_job_index": "3"}, + # begin_time == start_time is deliberately included. + {"job_submit_time": "10", "num_nodes": "5", "time_limit": "2", + "actual_run_time": "2", "initialization": "1", + "source_job_index": "4"}, + {"job_submit_time": "10", "num_nodes": "5", "time_limit": "2", + "actual_run_time": "2", "initialization": "0", + "source_job_index": "2"}, + ] + + +def test_capacity_detection(work): + history = work / "history.csv" + history.write_text( + "job_submit_time,begin_time,end_time,num_nodes,q_id\n" + "0,0,10,80,1\n" + "0,20,30,10,1\n" + ) + candidates = work / "candidates.csv" + schedule = work / "capacity.csv" + result = run( + ROOT / "scripts/detect_capacity_periods.py", + history, + candidates, + "--capacity-schedule", + schedule, + "--total-nodes", + "100", + "--utilization-threshold", + "0.25", + "--minimum-duration", + "10", + "--capacity-headroom", + "1", + "--queue-id", + "1", + ) + assert "candidate_periods=1\n" in result.stdout + assert read_rows(candidates) == [{ + "start_time": "0", "end_time": "20", "duration_seconds": "20", + "inferred_nodes": "80", + "reason": "low_allocation_with_backlog+no_starts_with_backlog", + "peak_allocated": "80", "peak_waiting_jobs": "1", + }] + assert read_rows(schedule) == [ + {"time": "0", "total_nodes": "80"}, + {"time": "20", "total_nodes": "100"}, + ] + + +def test_queue_pause_simulator_format(work): + report = work / "maintenance.json" + timeline = work / "timeline.csv" + with_reduced = work / "with_reduced.csv" + without_reduced = work / "without_reduced.csv" + + report.write_text(json.dumps({ + "normal_capacity": {"inferred_nodes": 100}, + "periods": [ + {"start": 3600, "end": 10800, + "state": "queue_pause_or_maintenance"}, + {"start": 10800, "end": 14400, "state": "reduced_capacity", + "potential_effective_capacity_lower_nodes": 60}, + ], + })) + timeline.write_text( + "start,end\n" + "0,3600\n" + "3600,7200\n" + "7200,10800\n" + "10800,14400\n" + ) + + run( + ROOT / "scripts/detect_queue_pause/build_resource_capacity_trace.py", + "--report", report, + "--timeline", timeline, + "--capacity", "100", + "--with-reduced-output", with_reduced, + "--without-reduced-output", without_reduced, + "--simulator-format", + ) + + assert read_rows(with_reduced) == [ + {"time": "0", "total_nodes": "100"}, + {"time": "3600", "total_nodes": "0"}, + {"time": "10800", "total_nodes": "60"}, + # The final inferred reduction must not persist past the timeline. + {"time": "14400", "total_nodes": "100"}, + ] + assert read_rows(without_reduced) == [ + {"time": "0", "total_nodes": "100"}, + {"time": "3600", "total_nodes": "0"}, + {"time": "10800", "total_nodes": "100"}, + ] + + # CTest supplies the built simulator so this verifies that the generated + # file is accepted directly, without renaming columns or preprocessing. + simulator = os.environ.get("DR_EVT_SIMULATOR") + if simulator: + jobs = work / "jobs.csv" + jobs.write_text("job_submit_time,num_nodes,time_limit\n0,1,1\n") + command = [ + simulator, str(jobs), "--total_nodes", "100", + "--capacity_schedule", str(with_reduced), + "--run_time_mode", "limit", + "--outfile", str(work / "simulated.csv"), + "--resource_trace", str(work / "resources.csv"), + ] + result = subprocess.run( + command, cwd=ROOT, capture_output=True, text=True + ) + assert result.returncode == 0, result.stdout + result.stderr + + +def main(): + with tempfile.TemporaryDirectory(prefix="dr_evt_trace_tools_") as tmp: + work = Path(tmp) + test_warm_start(work) + test_capacity_detection(work) + test_queue_pause_simulator_format(work) + print("Trace tool tests passed") + + +if __name__ == "__main__": + main() diff --git a/tests/test_traces/feature/capacity_schedule.capacity.csv b/tests/test_traces/feature/capacity_schedule.capacity.csv new file mode 100644 index 0000000..5515cc1 --- /dev/null +++ b/tests/test_traces/feature/capacity_schedule.capacity.csv @@ -0,0 +1,4 @@ +time,total_nodes +5,60 +20,100 +1000,25 diff --git a/tests/test_traces/feature/capacity_schedule.csv b/tests/test_traces/feature/capacity_schedule.csv new file mode 100644 index 0000000..45a0034 --- /dev/null +++ b/tests/test_traces/feature/capacity_schedule.csv @@ -0,0 +1,4 @@ +job_submit_time,num_nodes,time_limit +0,60,10 +0,40,30 +5,40,5 diff --git a/tests/test_traces/feature/capacity_schedule.expected_output.csv b/tests/test_traces/feature/capacity_schedule.expected_output.csv new file mode 100644 index 0000000..7e47aee --- /dev/null +++ b/tests/test_traces/feature/capacity_schedule.expected_output.csv @@ -0,0 +1,4 @@ +job_id,start_time,end_time +0,0,10 +1,0,30 +2,20,25 diff --git a/tests/test_traces/feature/capacity_schedule.expected_resources.csv b/tests/test_traces/feature/capacity_schedule.expected_resources.csv new file mode 100644 index 0000000..07ea36b --- /dev/null +++ b/tests/test_traces/feature/capacity_schedule.expected_resources.csv @@ -0,0 +1,9 @@ +time,nodes_used,nodes_free +0,60,40 +0,100,0 +5,100,0 +10,40,20 +20,40,60 +20,80,20 +25,40,60 +30,0,100 diff --git a/tests/test_traces/feature/capacity_schedule.flags b/tests/test_traces/feature/capacity_schedule.flags new file mode 100644 index 0000000..8ceaa86 --- /dev/null +++ b/tests/test_traces/feature/capacity_schedule.flags @@ -0,0 +1 @@ +--capacity_schedule tests/test_traces/feature/capacity_schedule.capacity.csv diff --git a/tests/test_traces/feature/warm_start_grpc.csv b/tests/test_traces/feature/warm_start_grpc.csv new file mode 100644 index 0000000..7edfd74 --- /dev/null +++ b/tests/test_traces/feature/warm_start_grpc.csv @@ -0,0 +1,9 @@ +job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit +0,0,5,2,0,5 +1,2,10,3,0,8 +2,4,15,2,0,11 +3,5,15,3,0,10 +4,12,14,1,0,2 +9,10,11,1,0,1 +10,10,14,4,0,4 +11,12,14,5,0,2 diff --git a/tests/test_traces/feature/warm_start_native.csv b/tests/test_traces/feature/warm_start_native.csv new file mode 100644 index 0000000..30a6d37 --- /dev/null +++ b/tests/test_traces/feature/warm_start_native.csv @@ -0,0 +1,6 @@ +job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit +0,0,20,40,0,20 +10,30,80,60,0,50 +20,60,65,10,0,5 +50,50,70,40,0,20 +55,60,90,60,0,30 diff --git a/tests/test_traces/feature/warm_start_native.expected_output.csv b/tests/test_traces/feature/warm_start_native.expected_output.csv new file mode 100644 index 0000000..73368f9 --- /dev/null +++ b/tests/test_traces/feature/warm_start_native.expected_output.csv @@ -0,0 +1,3 @@ +job_id,start_time,end_time +0,50,70 +1,80,110 diff --git a/tests/test_traces/feature/warm_start_native.expected_resources.csv b/tests/test_traces/feature/warm_start_native.expected_resources.csv new file mode 100644 index 0000000..edfd01f --- /dev/null +++ b/tests/test_traces/feature/warm_start_native.expected_resources.csv @@ -0,0 +1,7 @@ +time,nodes_used,nodes_free +50,60,40 +50,100,0 +70,60,40 +80,0,100 +80,60,40 +110,0,100 diff --git a/tests/test_traces/feature/warm_start_native.flags b/tests/test_traces/feature/warm_start_native.flags new file mode 100644 index 0000000..a18f5ab --- /dev/null +++ b/tests/test_traces/feature/warm_start_native.flags @@ -0,0 +1 @@ +--sim_start_time 50 --run_time_mode actual --job_store_capacity 5 --job_flush_interval 1 diff --git a/tests/test_traces/tools/warm_start_history.csv b/tests/test_traces/tools/warm_start_history.csv new file mode 100644 index 0000000..2edbba9 --- /dev/null +++ b/tests/test_traces/tools/warm_start_history.csv @@ -0,0 +1,6 @@ +job_submit_time,begin_time,end_time,num_nodes,time_limit +0,0,10,4,10 +0,5,20,3,20 +7,12,18,2,11 +8,9,11,1,3 +10,10,12,5,2 diff --git a/tests/test_traces/tools/warm_start_workload.csv b/tests/test_traces/tools/warm_start_workload.csv new file mode 100644 index 0000000..a6a8c81 --- /dev/null +++ b/tests/test_traces/tools/warm_start_workload.csv @@ -0,0 +1,4 @@ +job_submit_time,num_nodes,time_limit,actual_run_time +7,2,11,6 +8,1,3,2 +10,5,2,2 diff --git a/tests/test_warm_start.cpp b/tests/test_warm_start.cpp new file mode 100644 index 0000000..f070a79 --- /dev/null +++ b/tests/test_warm_start.cpp @@ -0,0 +1,665 @@ +/****************************************************************************** + * Copyright 2023 Lawrence Livermore National Security, LLC * + * See the top-level LICENSE file for details. * + * * + * SPDX-License-Identifier: MIT * + ******************************************************************************/ + +#define DR_EVT_HAS_CONFIG 1 +#include "sim/sim.hpp" + +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include +#include + +using namespace dr_evt; + +namespace { + +constexpr double kTolerance = 1e-4; + +struct JobRow { + double submit; + double begin; + double end; + unsigned nodes; +}; + +struct ResourceRow { + unsigned free; + unsigned allocated; +}; + +struct RunResult { + Simulation::Statistics stats; + std::vector jobs; + std::map settled_resources; +}; + +std::filesystem::path work_dir; +size_t run_number = 0; + +void expect_close(double actual, double expected) { + if (std::fabs(actual - expected) > kTolerance) { + std::cerr << "expected " << expected << ", got " << actual << '\n'; + std::abort(); + } +} + +void write_file(const std::filesystem::path &path, const std::string &text) { + std::ofstream output(path); + output << text; + assert(output.good()); +} + +std::vector split_csv(const std::string &line) { + std::vector fields; + std::stringstream input(line); + std::string field; + while (std::getline(input, field, ',')) { + fields.push_back(field); + } + return fields; +} + +std::vector read_jobs(const std::filesystem::path &path) { + std::ifstream input(path); + assert(input.good()); + std::string line; + assert(std::getline(input, line)); + std::vector jobs; + while (std::getline(input, line)) { + if (line.empty()) { + continue; + } + const auto fields = split_csv(line); + assert(fields.size() >= 4); + jobs.push_back({std::stod(fields[0]), std::stod(fields[1]), + std::stod(fields[2]), + static_cast(std::stoul(fields[3]))}); + } + return jobs; +} + +std::map +read_settled_resources(const std::filesystem::path &path, + double earliest = -1.0) { + std::ifstream input(path); + assert(input.good()); + std::string line; + assert(std::getline(input, line)); + std::map rows; + while (std::getline(input, line)) { + if (line.empty()) { + continue; + } + const auto fields = split_csv(line); + assert(fields.size() >= 3); + const double time = std::stod(fields[0]); + if (time + kTolerance >= earliest) { + // Assignment deliberately keeps only the final, settled sample when + // several arrivals/departures/capacity changes share one timestamp. + rows[time] = {static_cast(std::stoul(fields[1])), + static_cast(std::stoul(fields[2]))}; + } + } + return rows; +} + +RunResult run_trace(const std::string &trace, double sim_start_time, + unsigned total_nodes, + PriorityPolicy priority = PriorityPolicy::FCFS, + BackfillPolicy backfill = BackfillPolicy::EASY, + QueueImplementation queue = QueueImplementation::CIRCULAR, + const std::string &capacity = {}, + RunTimeMode run_time_mode = RunTimeMode::ACTUAL, + DistributionType run_time_distribution = + DistributionType::NORMAL, + double run_time_scale = 1.0, + double run_time_stddev = 0.0, + double max_time = -1.0) { + const std::string stem = "run_" + std::to_string(run_number++); + const auto trace_path = work_dir / (stem + ".csv"); + const auto job_path = work_dir / (stem + ".jobs.csv"); + const auto resource_path = work_dir / (stem + ".resources.csv"); + write_file(trace_path, trace); + + Sim_Params params; + params.m_infile = trace_path.string(); + params.m_total_nodes = total_nodes; + params.m_sim_start_time = sim_start_time; + params.m_trace_format = "simple"; + params.m_timestamp_format = "epoch"; + params.m_run_time_mode = run_time_mode; + params.m_run_time_distribution = run_time_distribution; + params.m_run_time_scale = run_time_scale; + params.m_run_time_stddev = run_time_stddev; + if (max_time >= 0.0) { + params.m_max_time = max_time; + params.m_is_time_set = true; + } + params.m_priority_policy = priority; + params.m_backfill_policy = backfill; + params.m_queue_impl = queue; + params.m_block_size = 4; + params.m_job_flush_interval = 100000; + params.m_resource_history_capacity = 100000; + params.m_msec_output = true; + params.set_outfile(job_path.string()); + params.set_resource_trace(resource_path.string()); + if (!capacity.empty()) { + const auto capacity_path = work_dir / (stem + ".capacity.csv"); + write_file(capacity_path, capacity); + params.m_capacity_schedule = capacity_path.string(); + } + + Simulation simulation(params); + simulation.run(); + const auto stats = simulation.get_statistics(); + simulation.write_simulated_trace(); + simulation.write_resource_trace(resource_path.string()); + return {stats, read_jobs(job_path), read_settled_resources(resource_path)}; +} + +void expect_job(const JobRow &job, double submit, double begin, double end, + unsigned nodes) { + expect_close(job.submit, submit); + expect_close(job.begin, begin); + expect_close(job.end, end); + assert(job.nodes == nodes); +} + +void test_boundary_classification_and_statistics() { + const std::string trace = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n" + "0,0,5,2,0,5\n" // completed before t + "1,2,10,3,0,8\n" // departure exactly at t + "2,4,15,2,0,11\n" // active seed, simultaneous departure + "3,5,15,3,0,10\n" // active seed, simultaneous departure + "4,12,14,1,0,2\n" // inherited waiter: excluded + "9,10,11,1,0,1\n" // inherited waiter beginning at t: excluded + "10,10,14,4,0,4\n" // boundary arrival: admitted + "11,12,14,5,0,2\n"; // future arrival: rescheduled + + const auto result = run_trace(trace, 10.0, 10); + assert(result.jobs.size() == 2); + expect_job(result.jobs[0], 10.0, 10.0, 14.0, 4); + expect_job(result.jobs[1], 11.0, 14.0, 16.0, 5); + + const auto &stats = result.stats; + assert(stats.jobs_submitted == 2); + assert(stats.jobs_completed == 2); + assert(stats.jobs_running == 0); + assert(stats.jobs_waiting == 0); + expect_close(stats.current_time, 16.0); + assert(stats.total_nodes == 10); + assert(stats.nodes_in_use == 0); + assert(stats.nodes_available == 10); + expect_close(stats.resource_area, 51.0); + expect_close(stats.utilization, 0.85); + expect_close(stats.avg_wait_time, 1.5); + expect_close(stats.avg_turnaround_time, 4.5); + expect_close(stats.makespan, 16.0); + + const std::map expected = { + {10.0, 9}, {14.0, 10}, {15.0, 5}, {16.0, 0}}; + assert(result.settled_resources.size() == expected.size()); + for (const auto &[time, allocated] : expected) { + assert(result.settled_resources.at(time).allocated == allocated); + } +} + +void test_max_time_is_an_inclusive_event_boundary() { + const std::string simulation_trace = + "job_submit_time,num_nodes,time_limit,actual_run_time\n" + "0,2,5,5\n" + "5,2,10,10\n" + "20,1,1,1\n"; + const auto simulation = + run_trace(simulation_trace, 0.0, 2, PriorityPolicy::FCFS, + BackfillPolicy::EASY, QueueImplementation::CIRCULAR, {}, + RunTimeMode::ACTUAL, DistributionType::NORMAL, 1.0, 0.0, 5.0); + expect_close(simulation.stats.current_time, 5.0); + assert(simulation.stats.jobs_completed == 1); + assert(simulation.stats.jobs_running == 1); + assert(simulation.stats.nodes_in_use == 2); + assert(simulation.jobs.size() == 1); + expect_job(simulation.jobs[0], 0.0, 0.0, 5.0, 2); + assert(simulation.settled_resources.at(5.0).allocated == 2); + + const std::string replay_trace = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n" + "0,0,5,2,0,5\n" + "4,5,10,2,0,5\n" + "11,11,12,1,0,1\n"; + const auto replay = + run_trace(replay_trace, 0.0, 4, PriorityPolicy::FCFS, + BackfillPolicy::EASY, QueueImplementation::CIRCULAR, {}, + RunTimeMode::ACTUAL, DistributionType::NORMAL, 1.0, 0.0, 5.0); + expect_close(replay.stats.current_time, 5.0); + assert(replay.stats.jobs_completed == 1); + assert(replay.stats.jobs_running == 1); + assert(replay.stats.nodes_in_use == 2); + assert(replay.jobs.size() == 1); + expect_job(replay.jobs[0], 0.0, 0.0, 5.0, 2); + + const std::string warm_trace = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n" + "0,2,15,2,0,13\n" + "10,20,24,2,0,4\n"; + const auto warm = + run_trace(warm_trace, 10.0, 4, PriorityPolicy::FCFS, + BackfillPolicy::EASY, QueueImplementation::CIRCULAR, {}, + RunTimeMode::ACTUAL, DistributionType::NORMAL, 1.0, 0.0, + 12.0); + expect_close(warm.stats.current_time, 12.0); + assert(warm.stats.jobs_completed == 0); + assert(warm.stats.jobs_running == 2); + assert(warm.stats.nodes_in_use == 4); + assert(warm.jobs.empty()); +} + +void test_no_live_history_and_empty_tail() { + const std::string future = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n" + "0,1,4,3,0,3\n" + "12,13,15,2,0,2\n"; + const auto with_future = run_trace(future, 10.0, 5); + assert(with_future.jobs.size() == 1); + expect_job(with_future.jobs[0], 12.0, 12.0, 14.0, 2); + assert(with_future.settled_resources.at(10.0).allocated == 0); + assert(with_future.stats.jobs_submitted == 1); + assert(with_future.stats.jobs_completed == 1); + expect_close(with_future.stats.resource_area, 4.0); + + const std::string finished = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n" + "0,1,3,2,0,2\n" + "1,3,7,4,0,4\n"; + const auto empty = run_trace(finished, 20.0, 8); + assert(empty.jobs.empty()); + assert(empty.stats.jobs_submitted == 0); + assert(empty.stats.jobs_completed == 0); + expect_close(empty.stats.current_time, 20.0); + expect_close(empty.stats.resource_area, 0.0); + expect_close(empty.stats.utilization, 0.0); + assert(empty.settled_resources.size() == 1); + assert(empty.settled_resources.at(20.0).allocated == 0); + + const std::string header_only = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n"; + const auto no_records = run_trace(header_only, 7.0, 4); + assert(no_records.jobs.empty()); + assert(no_records.stats.jobs_submitted == 0); + assert(no_records.stats.jobs_completed == 0); + assert(no_records.settled_resources.size() == 1); + assert(no_records.settled_resources.at(7.0).allocated == 0); +} + +void test_zero_start_preserves_full_replay() { + const std::string trace = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n" + "0,2,5,2,0,3\n" + "1,10,12,1,0,2\n"; + + // Zero is the disabled sentinel for replay-based warm start. Replay therefore + // preserves both historical schedules instead of rescheduling them from 0. + const auto result = run_trace(trace, 0.0, 4); + assert(result.jobs.size() == 2); + expect_job(result.jobs[0], 0.0, 2.0, 5.0, 2); + expect_job(result.jobs[1], 1.0, 10.0, 12.0, 1); +} + +void test_staggered_and_simultaneous_historical_departures() { + const std::string trace = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n" + "0,1,11,2,0,10\n" + "1,2,13,3,0,11\n" + "2,3,13,4,0,10\n" + "10,10,15,1,0,5\n"; + const auto result = run_trace(trace, 10.0, 10); + assert(result.jobs.size() == 1); + expect_job(result.jobs[0], 10.0, 10.0, 15.0, 1); + assert(result.settled_resources.at(10.0).allocated == 10); + assert(result.settled_resources.at(11.0).allocated == 8); + assert(result.settled_resources.at(13.0).allocated == 1); + assert(result.settled_resources.at(15.0).allocated == 0); + expect_close(result.stats.resource_area, 28.0); + expect_close(result.stats.utilization, 0.56); +} + +void test_capacity_boundaries_and_overcommit() { + const std::string trace = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n" + "0,1,14,6,0,13\n" + "10,10,13,2,0,3\n"; + const std::string capacity = + "time,total_nodes\n" + "0,10\n" // before t + "5,8\n" // before t + "10,4\n" // exactly at t: inherited allocation now overcommits + "12,10\n"; // after t: wakes the waiting boundary job + + const auto result = + run_trace(trace, 10.0, 10, PriorityPolicy::FCFS, BackfillPolicy::EASY, + QueueImplementation::CIRCULAR, capacity); + assert(result.jobs.size() == 1); + expect_job(result.jobs[0], 10.0, 12.0, 15.0, 2); + assert(result.stats.jobs_submitted == 1); + assert(result.stats.jobs_completed == 1); + expect_close(result.stats.resource_area, 30.0); + // Effective capacity is 6 while the reduction to 4 drains (t=10..12), + // then 10 through completion: capacity area = 6*2 + 10*3 = 42. + expect_close(result.stats.utilization, 30.0 / 42.0); + + const auto &at_t = result.settled_resources.at(10.0); + assert(at_t.allocated == 6 && at_t.free == 0); + const auto &after_t = result.settled_resources.at(12.0); + assert(after_t.allocated == 8 && after_t.free == 2); + assert(result.settled_resources.at(14.0).allocated == 2); + assert(result.settled_resources.at(15.0).allocated == 0); +} + +void test_capacity_change_at_final_warm_departure() { + const std::string trace = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n" + "0,1,15,4,0,14\n" // final warmup departure + "10,12,15,6,0,3\n"; // needs the capacity increase at t=15 + const std::string capacity = + "time,total_nodes\n" + "0,4\n" + "15,8\n"; // simultaneous with the last warmup departure + + const auto result = + run_trace(trace, 10.0, 8, PriorityPolicy::FCFS, BackfillPolicy::EASY, + QueueImplementation::CIRCULAR, capacity); + assert(result.jobs.size() == 1); + expect_job(result.jobs[0], 10.0, 15.0, 18.0, 6); + + const auto &at_boundary = result.settled_resources.at(10.0); + assert(at_boundary.allocated == 4 && at_boundary.free == 0); + const auto &at_transition = result.settled_resources.at(15.0); + assert(at_transition.allocated == 6 && at_transition.free == 2); + const auto &at_completion = result.settled_resources.at(18.0); + assert(at_completion.allocated == 0 && at_completion.free == 8); + + expect_close(result.stats.resource_area, 38.0); + expect_close(result.stats.utilization, 38.0 / 44.0); +} + +void test_fractional_boundary() { + const std::string trace = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n" + "0,9.500,10.250,2,0,1\n" // ends exactly at t + "1,9.750,10.750,3,0,1\n" // inherited occupancy + "10.250,10.250,10.750,2,0,1\n" + "10.500,10.500,11.250,3,0,1\n"; + const auto result = run_trace(trace, 10.25, 5); + assert(result.jobs.size() == 2); + expect_job(result.jobs[0], 10.25, 10.25, 10.75, 2); + expect_job(result.jobs[1], 10.5, 10.75, 11.5, 3); + assert(result.settled_resources.at(10.25).allocated == 5); + assert(result.settled_resources.at(10.75).allocated == 3); + assert(result.settled_resources.at(11.5).allocated == 0); + expect_close(result.stats.resource_area, 4.75); + expect_close(result.stats.avg_wait_time, 0.125); + expect_close(result.stats.avg_turnaround_time, 0.75); +} + +void test_runtime_mode_applies_only_to_simulated_jobs() { + const std::string trace = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n" + "0,2,20,1,0,18\n" // warmup job: fixed historical departure + "10,12,14,1,0,7\n"; // simulated job: observed runtime 2, limit 7 + + const auto actual = + run_trace(trace, 10.0, 2, PriorityPolicy::FCFS, BackfillPolicy::EASY, + QueueImplementation::CIRCULAR, {}, RunTimeMode::ACTUAL); + assert(actual.jobs.size() == 1); + expect_job(actual.jobs[0], 10.0, 10.0, 12.0, 1); + + const auto limit = + run_trace(trace, 10.0, 2, PriorityPolicy::FCFS, BackfillPolicy::EASY, + QueueImplementation::CIRCULAR, {}, RunTimeMode::LIMIT); + assert(limit.jobs.size() == 1); + expect_job(limit.jobs[0], 10.0, 10.0, 17.0, 1); + + const auto distribution = run_trace( + trace, 10.0, 2, PriorityPolicy::FCFS, BackfillPolicy::EASY, + QueueImplementation::CIRCULAR, {}, RunTimeMode::DISTRIBUTION, + DistributionType::NORMAL, 0.5, 0.0); + assert(distribution.jobs.size() == 1); + expect_job(distribution.jobs[0], 10.0, 10.0, 13.5, 1); + + // Runtime mode affects only the simulated job. The warmup job retains its + // recorded end time in all runs and remains allocated through t=20. + assert(actual.settled_resources.at(20.0).allocated == 0); + assert(limit.settled_resources.at(20.0).allocated == 0); + assert(distribution.settled_resources.at(20.0).allocated == 0); +} + +std::vector without_seed_jobs(const std::vector &jobs, + size_t seed_count) { + assert(jobs.size() >= seed_count); + return {jobs.begin() + static_cast(seed_count), jobs.end()}; +} + +void compare_equivalent(const std::string &warm_trace, + const std::string &reference_trace, + double sim_start_time, + unsigned total_nodes, size_t seed_count, + PriorityPolicy priority, BackfillPolicy backfill, + QueueImplementation queue) { + const auto warm = + run_trace(warm_trace, sim_start_time, total_nodes, priority, backfill, + queue); + const auto reference = + run_trace(reference_trace, 0.0, total_nodes, priority, backfill, queue); + const auto reference_jobs = without_seed_jobs(reference.jobs, seed_count); + assert(warm.jobs.size() == reference_jobs.size()); + for (size_t i = 0; i < warm.jobs.size(); ++i) { + expect_close(warm.jobs[i].submit, reference_jobs[i].submit); + expect_close(warm.jobs[i].begin, reference_jobs[i].begin); + expect_close(warm.jobs[i].end, reference_jobs[i].end); + assert(warm.jobs[i].nodes == reference_jobs[i].nodes); + } + + const auto reference_after_t = [&]() { + std::map rows; + for (const auto &[time, row] : reference.settled_resources) { + if (time + kTolerance >= sim_start_time) { + rows[time] = row; + } + } + return rows; + }(); + assert(warm.settled_resources.size() == reference_after_t.size()); + for (const auto &[time, row] : warm.settled_resources) { + const auto &reference_row = reference_after_t.at(time); + assert(row.free == reference_row.free); + assert(row.allocated == reference_row.allocated); + } + expect_close(warm.stats.resource_area, reference.stats.resource_area); + const double reference_from_boundary = + reference.stats.resource_area / (static_cast(total_nodes) * + (reference.stats.makespan - + sim_start_time)); + expect_close(warm.stats.utilization, reference_from_boundary); +} + +void test_every_policy_and_queue() { + const std::string warm = + "job_submit_time,begin_time,end_time,num_nodes,exit_status,time_limit\n" + "0,1,4,1,0,10\n" + "1,2,15,4,0,10\n" + "2,6,12,3,0,10\n" + "3,11,14,2,0,10\n" + "10,10,14,3,0,10\n" + "10,11,13,4,0,10\n" + "11,12,15,2,0,10\n" + "12,13,14,5,0,10\n"; + const std::string reference = + "job_submit_time,num_nodes,time_limit,actual_run_time\n" + "10,4,10,5\n" + "10,3,10,2\n" + "10,3,10,4\n" + "10,4,10,2\n" + "11,2,10,3\n" + "12,5,10,1\n"; + + const std::vector priorities = { + PriorityPolicy::FCFS, PriorityPolicy::FCFS_ALT, + PriorityPolicy::FCFS_CONSERVATIVE, PriorityPolicy::SJF, + PriorityPolicy::LJF}; + const std::vector backfills = { + BackfillPolicy::NONE, BackfillPolicy::EASY, BackfillPolicy::CONSERVATIVE}; + const std::vector queues = { + QueueImplementation::CIRCULAR, QueueImplementation::DEQUE, + QueueImplementation::MULTIMAP, QueueImplementation::BLOCK}; + + for (const auto priority : priorities) { + const auto supported_queue = priority == PriorityPolicy::FCFS_CONSERVATIVE + ? QueueImplementation::DEQUE + : priority == PriorityPolicy::FCFS_ALT + ? QueueImplementation::MULTIMAP + : QueueImplementation::CIRCULAR; + for (const auto backfill : backfills) { + compare_equivalent(warm, reference, 10.0, 10, 2, priority, backfill, + supported_queue); + } + } + for (const auto queue : queues) { + for (const auto backfill : backfills) { + compare_equivalent(warm, reference, 10.0, 10, 2, PriorityPolicy::FCFS, + backfill, queue); + } + } +} + +void test_seeded_random_differential() { + constexpr double t = 20.5; + for (unsigned seed : {7u, 29u, 101u, 4099u}) { + std::mt19937 rng(seed); + std::uniform_int_distribution nodes(1, 4); + std::uniform_int_distribution duration(1, 7); + std::uniform_int_distribution offset(0, 5); + + std::ostringstream warm; + warm << "job_submit_time,begin_time,end_time,num_nodes,exit_status," + "time_limit\n"; + // Finished history and inherited waiters deliberately have no reference + // counterpart. They should be observationally absent after the boundary. + warm << "0,1,4," << nodes(rng) << ",0,3\n"; + + std::ostringstream reference; + reference << "job_submit_time,num_nodes,time_limit,actual_run_time\n"; + unsigned active_nodes = 0; + constexpr size_t active_count = 3; + for (size_t i = 0; i < active_count; ++i) { + const unsigned n = nodes(rng); + const unsigned remaining = 2 + offset(rng); + const double begin = 10.0 + static_cast(i); + const double end = t + remaining; + active_nodes += n; + warm << i + 1 << ',' << begin << ',' << end << ',' << n << ",0," << 10 + << "\n"; + reference << t << ',' << n << ",10," << remaining << "\n"; + } + warm << "4,21,24,1,0,3\n"; + warm << "5,20.5,22,1,0,2\n"; + + constexpr size_t workload_count = 9; + for (size_t i = 0; i < workload_count; ++i) { + const double submit = t + 0.5 * offset(rng); + const unsigned n = nodes(rng); + const unsigned d = duration(rng); + // begin/end are observed history only; warm start must retain duration + // while replacing both with the counterfactual scheduler's result. + warm << submit << ',' << submit + 3 << ',' << submit + 3 + d << ',' << n + << ",0,10\n"; + reference << submit << ',' << n << ",10," << d << "\n"; + } + const unsigned total_nodes = std::max(12u, active_nodes); + compare_equivalent(warm.str(), reference.str(), t, total_nodes, + active_count, PriorityPolicy::FCFS, BackfillPolicy::EASY, + QueueImplementation::CIRCULAR); + } +} + +void test_rejected_modes() { + const auto simulation_path = work_dir / "non_replay.csv"; + write_file(simulation_path, "job_submit_time,num_nodes,time_limit\n0,1,1\n"); + Sim_Params non_replay; + non_replay.m_infile = simulation_path.string(); + non_replay.m_total_nodes = 2; + non_replay.m_sim_start_time = 1.0; + non_replay.m_trace_format = "simple"; + non_replay.m_timestamp_format = "epoch"; + bool rejected = false; + try { + Simulation simulation(non_replay); + simulation.run(); + } catch (const std::runtime_error &e) { + rejected = std::string(e.what()).find("replay-format") != std::string::npos; + } + assert(rejected); + + const auto replay_path = work_dir / "replay_for_list.csv"; + write_file(replay_path, + "job_submit_time,begin_time,end_time,num_nodes,exit_status," + "time_limit\n0,0,1,1,0,1\n"); + Sim_Params progressive; + progressive.m_infile = replay_path.string(); + progressive.m_infile_list = (work_dir / "traces.list").string(); + progressive.m_sim_start_time = 1.0; + progressive.m_total_nodes = 2; + progressive.m_trace_format = "simple"; + progressive.m_timestamp_format = "epoch"; + rejected = false; + try { + Simulation simulation(progressive); + simulation.run(); + } catch (const std::runtime_error &e) { + rejected = std::string(e.what()).find("infile_list") != std::string::npos; + } + assert(rejected); +} + +} // namespace + +int main() { + const auto suffix = + std::chrono::steady_clock::now().time_since_epoch().count(); + work_dir = std::filesystem::temp_directory_path() / + ("dr_evt_warm_start_" + std::to_string(suffix)); + std::filesystem::create_directory(work_dir); + try { + test_boundary_classification_and_statistics(); + test_max_time_is_an_inclusive_event_boundary(); + test_no_live_history_and_empty_tail(); + test_zero_start_preserves_full_replay(); + test_staggered_and_simultaneous_historical_departures(); + test_capacity_boundaries_and_overcommit(); + test_capacity_change_at_final_warm_departure(); + test_fractional_boundary(); + test_runtime_mode_applies_only_to_simulated_jobs(); + test_every_policy_and_queue(); + test_seeded_random_differential(); + test_rejected_modes(); + } catch (...) { + std::filesystem::remove_all(work_dir); + throw; + } + std::filesystem::remove_all(work_dir); + std::cout << "Warm-start tests passed\n"; + return EXIT_SUCCESS; +}