Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
8 changes: 8 additions & 0 deletions .github/workflows/ci.yml
Original file line number Diff line number Diff line change
Expand Up @@ -122,6 +122,7 @@ jobs:
dlopen_tests,
rss_tests,
stack_tests,
interpreter_tests,
]
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
Expand All @@ -136,6 +137,13 @@ jobs:
- name: Install additional allocators
run: sudo apt-get install -y libmimalloc-dev libjemalloc-dev

# The perf trampoline needs Python 3.12+.
- name: Install Python with perf trampoline support
if: matrix.test == 'interpreter_tests'
uses: actions/setup-python@5fda3b95a4ea91299a34e894583c3862153e4b97 # v7.0.0
with:
python-version: "3.12"

- name: Run tests
env:
RUST_LOG: debug
Expand Down
2 changes: 2 additions & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

5 changes: 3 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -160,11 +160,12 @@ codspeed exec --mode simulation -- ./my-binary

### Memory

Tracks heap allocations (peak usage, count, allocation size) with eBPF profiling.
Tracks native heap allocations (peak usage, count, allocation size) with eBPF profiling.

**Best for:** Memory optimization, leak detection, constrained environments

**Supported:** Rust, C/C++ with libc, jemalloc, mimalloc
**Supported:** Rust, C/C++, Python (3.12+), and Node.js programs that allocate
through libc, jemalloc, or mimalloc.
Comment thread
not-matthias marked this conversation as resolved.

```bash
codspeed exec --mode memory -- ./my-binary
Expand Down
3 changes: 3 additions & 0 deletions crates/exec-harness/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -22,6 +22,9 @@ runner-shared = { path = "../runner-shared" }
tempfile = { workspace = true }
object = { workspace = true }

[dev-dependencies]
rstest = { workspace = true }

[build-dependencies]
cc = "1"

Expand Down
4 changes: 0 additions & 4 deletions crates/exec-harness/src/analysis/mod.rs
Original file line number Diff line number Diff line change
Expand Up @@ -54,12 +54,8 @@ pub fn perform_with_valgrind(commands: Vec<BenchmarkCommand>) -> Result<()> {
cmd.args(&benchmark_cmd.command[1..]);
// Use LD_PRELOAD to inject instrumentation into the child process
cmd.env("LD_PRELOAD", preload_lib_path);
// Make sure python processes output perf maps. This is usually done by `pytest-codspeed`
cmd.env("PYTHONPERFSUPPORT", "1");
cmd.env(constants::URI_ENV, &name_and_uri.uri);

crate::node::set_node_options(&mut cmd);

let mut child = cmd.spawn().context("Failed to spawn command")?;

let status = child.wait().context("Failed to execute command")?;
Expand Down
31 changes: 10 additions & 21 deletions crates/exec-harness/src/lib.rs
Original file line number Diff line number Diff line change
@@ -1,23 +1,15 @@
use clap::ValueEnum;
use prelude::*;
use serde::{Deserialize, Serialize};
use std::io::{self, BufRead};

pub mod analysis;
pub mod constants;
pub mod node;
pub mod prelude;
mod runtime_env;
mod uri;
pub mod walltime;

#[derive(ValueEnum, Clone, Copy, Debug, Serialize, Deserialize, PartialEq)]
#[serde(rename_all = "lowercase")]
pub enum MeasurementMode {
Walltime,
Memory,
#[value(alias = "instrumentation")]
Simulation,
}
pub use runner_shared::measurement_mode::MeasurementMode;

/// A single benchmark command for stdin mode input.
///
Expand Down Expand Up @@ -70,17 +62,14 @@ pub fn execute_benchmarks(
commands: Vec<BenchmarkCommand>,
measurement_mode: Option<MeasurementMode>,
) -> Result<()> {
let measurement_mode = measurement_mode.unwrap_or(MeasurementMode::Walltime);
// SAFETY: exec-harness is single-threaded here. No thread has been spawned
// yet and InstrumentHooks is only created later, inside `perform`.
unsafe { runtime_env::apply(measurement_mode)? };

match measurement_mode {
Some(MeasurementMode::Walltime) | None => {
walltime::perform(commands)?;
}
Some(MeasurementMode::Memory) => {
analysis::perform(commands)?;
}
Some(MeasurementMode::Simulation) => {
analysis::perform_with_valgrind(commands)?;
}
MeasurementMode::Walltime => walltime::perform(commands),
MeasurementMode::Memory => analysis::perform(commands),
MeasurementMode::Simulation => analysis::perform_with_valgrind(commands),
}

Ok(())
}
18 changes: 0 additions & 18 deletions crates/exec-harness/src/node.rs

This file was deleted.

23 changes: 23 additions & 0 deletions crates/exec-harness/src/runtime_env/mod.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,23 @@
//! Runtime env for benchmark processes started by exec-harness: the shared
//! [`runner_shared::runtime_env::env`] vars plus the `node` wrapper on `PATH`.

use crate::MeasurementMode;

mod node;

const PATH_ENV: &str = "PATH";

/// Applies the runtime env and the node wrapper to the current process, so
/// every child inherits them. Existing values are overwritten: `mode` may come
/// from a CLI flag that differs from an inherited `CODSPEED_RUNNER_MODE`.
///
/// # Safety
/// Must be called while the process is single-threaded.
pub(crate) unsafe fn apply(mode: MeasurementMode) -> anyhow::Result<()> {
for (key, value) in runner_shared::runtime_env::env(mode) {
unsafe { std::env::set_var(key, value) };
}
let path = std::env::var_os(PATH_ENV).unwrap_or_default();
unsafe { std::env::set_var(PATH_ENV, node::path_with_node_wrapper(&path)?) };
Ok(())
}
68 changes: 68 additions & 0 deletions crates/exec-harness/src/runtime_env/node/mod.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,68 @@
//! `node` wrapper for `codspeed exec`.
//!
//! The V8 flags we need slow node down, so they should only apply to the
//! benchmark process.
//!
//! In `codspeed run`, the runner's `introspected_nodejs` wrapper does this:
//! codspeed-node requests the flags through introspection, so non-benchmark
//! node processes are left alone.
//!
//! In `codspeed exec`, there is no codspeed-node and so no way to know which
//! process is the benchmark. This wrapper therefore adds the flags to every
//! `node` call. It mirrors codspeed-node's `getV8Flags()`.
//!
//! The two wrappers are never on PATH together (`enable_introspection` is
//! false for exec-harness targets).
Comment thread
not-matthias marked this conversation as resolved.

use std::ffi::{OsStr, OsString};
use std::os::unix::fs::PermissionsExt;
use std::path::PathBuf;
use std::sync::Mutex;

const NODE_WRAPPER_SCRIPT: &str = include_str!("node.sh");
const WRAPPER_DIR_PREFIX: &str = "codspeed_node_wrapper";
const WRAPPER_FILE_NAME: &str = "node";
const EXECUTABLE_MODE: u32 = 0o755;

/// Folder of the wrapper installed by this process, once it is installed.
static INSTALLED_DIR: Mutex<Option<PathBuf>> = Mutex::new(None);

/// Writes the `node` wrapper script and returns its folder.
///
/// The folder is a fresh temp dir with a random name, created exclusively and
/// writable only by us: a shared, predictable path could be pre-created by
/// another user, who could then replace the script that ends up first on
/// `PATH`. It is kept after the process exits, since benchmarks execute the
/// wrapper until then.
///
/// A process installs at most once. A second writer in the same process would
/// race with concurrent `fork`+`exec` calls: the child inherits the writable
/// descriptor and then fails to execute the script with `ETXTBSY`.
fn install_wrapper() -> std::io::Result<PathBuf> {
let mut installed = INSTALLED_DIR.lock().unwrap_or_else(|e| e.into_inner());
if let Some(dir) = installed.as_ref() {
return Ok(dir.clone());
}

let dir = tempfile::Builder::new()
.prefix(WRAPPER_DIR_PREFIX)
.tempdir()?
.keep();
Comment thread
not-matthias marked this conversation as resolved.
let script = dir.join(WRAPPER_FILE_NAME);
std::fs::write(&script, NODE_WRAPPER_SCRIPT)?;
std::fs::set_permissions(&script, std::fs::Permissions::from_mode(EXECUTABLE_MODE))?;

*installed = Some(dir.clone());
Ok(dir)
}

/// Installs the `node` wrapper and returns `path` with its folder prepended.
pub(super) fn path_with_node_wrapper(path: &OsStr) -> anyhow::Result<OsString> {
let wrapper_dir = install_wrapper()?;
Ok(std::env::join_paths(
std::iter::once(wrapper_dir).chain(std::env::split_paths(path)),
)?)
}

#[cfg(test)]
mod tests;
117 changes: 117 additions & 0 deletions crates/exec-harness/src/runtime_env/node/node.sh
Original file line number Diff line number Diff line change
@@ -0,0 +1,117 @@
#!/usr/bin/env bash
# CodSpeed `node` wrapper.
#
# Installed on PATH in front of the real node. Runs the real node with the V8
# flags that the codspeed-node integration would request, based on
# CODSPEED_RUNNER_MODE and the node major version.
#
# Mirrors getV8Flags() and the mode mapping of getInstrumentMode():
# https://github.com/CodSpeedHQ/codspeed-node/blob/main/packages/core/src/introspection.ts
# https://github.com/CodSpeedHQ/codspeed-node/blob/main/packages/core/src/runnerMode.ts
set -euo pipefail

# In simulation mode this script runs under valgrind with the CodSpeed preload
# library in LD_PRELOAD, which reports a benchmark result from every process
# that loads it. Helper processes must not load it, so it is removed here and
# restored for the final exec. For the same reason the script avoids command
# substitutions: a forked subshell would report a result when it exits.
codspeed_preload="${LD_PRELOAD:-}"
unset LD_PRELOAD

wrapper_script="${BASH_SOURCE[0]}"

# Sets `real_node` to the first `node` on PATH that is not this script.
find_real_node() {
local entry
local IFS=':'
set -o noglob
for entry in $PATH; do
if [[ -z "$entry" || ! -x "$entry/node" || "$entry/node" -ef "$wrapper_script" ]]; then
continue
fi
real_node="$entry/node"
return 0
done
return 1
}

# Sets `major` from the given node binary's version ("v22.12.0" -> 22).
node_major_version() {
local version
# The only subprocess: node runs without the preload library (see above).
version="$("$1" --version)"
version="${version#v}"
major="${version%%.*}"
}

# Flags for simulation and memory mode: make execution deterministic.
add_analysis_flags() {
flags+=(
--hash-seed=1
--random-seed=1
--no-opt
--predictable
--predictable-gc-schedule
--expose-gc
--no-concurrent-sweeping
--max-old-space-size=4096
)
if (( major < 18 )); then
flags+=(--no-randomize-hashes)
fi
if (( major < 20 )); then
flags+=(--no-scavenge-task)
else
# V8 11.3 renamed --scavenge-task to --minor-gc-task
flags+=(--no-minor-gc-task)
fi
if (( major >= 24 )); then
# --no-opt only disables TurboFan. Maglev is enabled by default from
# V8 13.6 and keeps tiering up hot functions.
flags+=(--no-maglev)
fi
}

# Flags for walltime mode: emit JIT symbols for the profiler.
add_walltime_flags() {
flags+=(--perf-prof)
if [[ -n "${CODSPEED_V8_LOG:-}" ]]; then
flags+=(
--log-code
--no-log-source-code
--no-logfile-per-isolate
"--logfile=${CODSPEED_V8_LOG}/codspeed-v8-%p.log"
)
else
flags+=(--perf-basic-prof)
fi
}

find_real_node || {
echo "codspeed: node not found in PATH" >&2
exit 1
}
node_major_version "$real_node"

flags=(
--interpreted-frames-native-stack
--allow-natives-syntax
)
case "${CODSPEED_RUNNER_MODE:-walltime}" in
instrumentation | simulation) add_analysis_flags ;;
memory)
add_analysis_flags
# Memory flamegraphs name JS frames through /tmp/perf-<pid>.map.
flags+=(--perf-basic-prof)
;;
walltime) add_walltime_flags ;;
*)
echo "codspeed: unknown CODSPEED_RUNNER_MODE '${CODSPEED_RUNNER_MODE}'" >&2
exit 1
;;
esac

if [[ -n "$codspeed_preload" ]]; then
export LD_PRELOAD="$codspeed_preload"
fi
exec "$real_node" "${flags[@]}" "$@"
Loading
Loading