Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
1 change: 1 addition & 0 deletions Cargo.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

5 changes: 4 additions & 1 deletion Makefile
Original file line number Diff line number Diff line change
@@ -1,4 +1,4 @@
.PHONY: help lint format type-check security workflow-lint license-check third-party-notices third-party-notices-check cargo-deny-licenses test pre-push pre-push-clean pre-push-preflight pre-push-fast clean test-tck docstring-coverage test-network benchmark test-perf test-perf-xs test-perf-slow test-perf-large coverage coverage-rust coverage-python coverage-node coverage-quick coverage-report coverage-diff coverage-strict check-coverage check-coverage-rust check-coverage-python check-coverage-node check-patch-coverage test-durations test-analytics docs-serve docs-build docs-clean cargo-build codspeed-build codspeed-build-walltime codspeed-run bench-traversal bench-fixed-hop-limit bench-fixed-hop-livejournal bench-m4-entry bench-adjacency-200m m4-entry-matrix-check durability-isolation-check native-consumers release-load-matrix-check release-load-matrix bulk-construction-conformance-check bulk-construction-conformance cargo-test cargo-check cargo-clippy cargo-fmt cargo-fmt-check clean-builds clean-builds-all pnpm-install pnpm-build install build release-version-check package-license-verify publish-dry-run publish-dry-run-npm publish-dry-run-docs publish-dry-run-python publish-dry-run-cargo record-release-artifacts clean-env-verify-check clean-env-verify-preflight clean-env-verify
.PHONY: help lint format type-check security workflow-lint license-check third-party-notices third-party-notices-check cargo-deny-licenses test pre-push pre-push-clean pre-push-preflight pre-push-fast clean test-tck docstring-coverage test-network benchmark test-perf test-perf-xs test-perf-slow test-perf-large coverage coverage-rust coverage-python coverage-node coverage-quick coverage-report coverage-diff coverage-strict check-coverage check-coverage-rust check-coverage-python check-coverage-node check-patch-coverage test-durations test-analytics docs-serve docs-build docs-clean cargo-build codspeed-build codspeed-build-walltime codspeed-run bench-traversal bench-tck-scenarios bench-fixed-hop-limit bench-fixed-hop-livejournal bench-m4-entry bench-adjacency-200m m4-entry-matrix-check durability-isolation-check native-consumers release-load-matrix-check release-load-matrix bulk-construction-conformance-check bulk-construction-conformance cargo-test cargo-check cargo-clippy cargo-fmt cargo-fmt-check clean-builds clean-builds-all pnpm-install pnpm-build install build release-version-check package-license-verify publish-dry-run publish-dry-run-npm publish-dry-run-docs publish-dry-run-python publish-dry-run-cargo record-release-artifacts clean-env-verify-check clean-env-verify-preflight clean-env-verify

help: ## Show this help message
@grep -E '^[a-zA-Z_-]+:.*?## .*$$' $(MAKEFILE_LIST) | sort | awk 'BEGIN {FS = ":.*?## "}; {printf "\033[36m%-20s\033[0m %s\n", $$1, $$2}'
Expand Down Expand Up @@ -311,6 +311,9 @@ codspeed-run: ## Run the CodSpeed benchmarks locally (requires the codspeed CLI
bench-traversal: ## Run the #767 traversal scaling Divan benchmarks (release, manual; see benchmarks/traversal_scaling.md)
cargo bench -p graphforge-exec --bench traversal_scaling -- --sample-count 5

bench-tck-scenarios: ## Run the #1653 per-scenario openCypher TCK Divan benchmark (manual; raw results under CODSPEED_ENV)
cargo bench -p graphforge-api --bench tck_scenarios

bench-merge-scaling: ## Run the #1400 node MERGE scaling Divan benchmarks (release, manual)
cargo bench -p graphforge-exec --bench merge_scaling -- --sample-count 5

Expand Down
16 changes: 16 additions & 0 deletions config/benchmark-measurement-inventory.json
Original file line number Diff line number Diff line change
Expand Up @@ -55,6 +55,22 @@
"owner_issue": null,
"notes": "Node MERGE scaling wall-clock workloads; topology-read bound stays in merge_scaling_bench integration tests."
},
{
"path": "crates/graphforge-api/benches/tck_scenarios/main.rs",
"boundary": "in_process",
"authority": "divan",
"disposition": "framework_authority",
"owner_issue": null,
"notes": "openCypher TCK per-scenario Divan benchmark (#1653): normalized corpus and pooled fixture setup; timing from Divan only, serialized as CodSpeed walltime raw_results under CODSPEED_ENV."
},
{
"path": "crates/graphforge-api/benches/tck_scenarios/runner.rs",
"boundary": "in_process",
"authority": "divan",
"disposition": "framework_authority",
"owner_issue": null,
"notes": "TCK scenario execution through the registered cucumber steps; the timed region is the whole scenario and each verdict is checked outside it, so a failing scenario aborts before Divan records timing."
},
{
"path": "crates/graphforge-exec/tests/persistent_adjacency.rs",
"boundary": "in_process",
Expand Down
7 changes: 7 additions & 0 deletions crates/graphforge-api/Cargo.toml
Original file line number Diff line number Diff line change
Expand Up @@ -54,6 +54,7 @@ uuid = { workspace = true }
# Assessment-only IPC codec experiments; production writers remain unchanged.
arrow = { workspace = true, features = ["ipc_compression"] }
cucumber = { workspace = true }
divan = { workspace = true }
tokio = { workspace = true }
graphforge-cypher = { path = "../graphforge-cypher" }
graphforge-storage = { path = "../graphforge-storage", features = ["test-failpoints", "test-support"] }
Expand Down Expand Up @@ -213,5 +214,11 @@ required-features = ["research", "search"]
name = "bdd"
harness = false

# In-process Divan benchmark over the same TCK scenarios, step functions and
# pooled fixture as the `bdd` run (#1653). Divan's own harness drives `main`.
[[bench]]
name = "tck_scenarios"
harness = false

[lints]
workspace = true
52 changes: 52 additions & 0 deletions crates/graphforge-api/benches/tck_scenarios/main.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,52 @@
//! In-process Divan benchmark over the openCypher TCK scenarios (#1653).
//!
//! Parses the same ephemeral normalized corpus as the Cucumber correctness run
//! (`tests/bdd/main.rs`) and times each scenario, executed through the same
//! registered step functions, pooled fixture and clear-on-lease semantics. See
//! `runner.rs` for the timed region and the fail-closed verdict check.
//!
//! ```bash
//! cargo bench -p graphforge-api --bench tck_scenarios # measure
//! cargo bench -p graphforge-api --bench tck_scenarios -- --test # test mode
//! ```
//!
//! Divan does the measuring. Machine-readable per-scenario output is CodSpeed's
//! walltime `raw_results`, written only when `CODSPEED_ENV` is set; see
//! `docs/development/benchmarking.md`. Test mode runs every scenario once and
//! emits no timing. `TCK_ONLY=<substr>` restricts the corpus as it does for the
//! Cucumber run.

#[cfg(feature = "search")]
#[path = "../../tests/bdd/api_steps.rs"]
mod api_steps;
#[path = "../../tests/bdd/corpus.rs"]
mod corpus;
#[path = "../../tests/bdd/fixture.rs"]
mod fixture;
mod runner;
#[path = "../../tests/bdd/tck_steps.rs"]
mod tck_steps;
#[path = "../../tests/bdd/world.rs"]
mod world;

use world::GraphForgeWorld;

fn main() {
let workspace_root = std::path::Path::new(env!("CARGO_MANIFEST_DIR"))
.join("../..")
.canonicalize()
.expect("workspace root must exist");
let (normalized, cases) = runner::load_normalized(&workspace_root.join("tests/tck/features"));
eprintln!(
"TCK scenario benchmark: {} scenarios, fixture profile pooled-isolated-serial-v1, concurrency {}",
cases.len(),
fixture::TCK_CONCURRENCY
);
runner::install_corpus(cases);

let fixture_guard = fixture::activate();
divan::main();
runner::assert_fixture_profile();
drop(fixture_guard);
drop(normalized);
}
259 changes: 259 additions & 0 deletions crates/graphforge-api/benches/tck_scenarios/runner.rs
Original file line number Diff line number Diff line change
@@ -0,0 +1,259 @@
//! Scenario loading, execution and the Divan benchmark over the TCK corpus.
//!
//! Every step runs through the step functions registered against
//! [`GraphForgeWorld`] and resolved with `GraphForgeWorld::collection().find()`,
//! the same registry the Cucumber correctness run uses. There is no second
//! semantics engine.
//!
//! One timed iteration is one whole scenario: world construction, feature and
//! rule backgrounds, every scenario step (including `Given an empty graph`,
//! which leases and clears the pooled fixture) and the fixture release that the
//! Cucumber run performs in its `after` hook. Each iteration's pass/fail verdict
//! is recorded inside the timed region and checked outside it, before the next
//! sample starts and before the benchmark function returns. Divan emits a
//! benchmark's timing only after that function returns, so a failing scenario
//! panics first and never yields timing.

use std::cell::RefCell;
use std::fmt;
use std::panic::AssertUnwindSafe;
use std::path::Path;
use std::sync::{Arc, OnceLock};

use cucumber::gherkin;
use cucumber::{Parser as _, World as _};
use futures::{FutureExt as _, StreamExt as _};

use crate::GraphForgeWorld;

/// One expanded TCK scenario, keyed `<feature>:<line>:<name>` exactly as the
/// Cucumber runner and `tests/tck/passing_baseline.txt` key it.
#[derive(Clone)]
pub struct ScenarioCase {
key: String,
feature: Arc<gherkin::Feature>,
rule: Option<Arc<gherkin::Rule>>,
scenario: Arc<gherkin::Scenario>,
}

impl ScenarioCase {
fn new(
feature: &Arc<gherkin::Feature>,
rule: Option<&Arc<gherkin::Rule>>,
scenario: &gherkin::Scenario,
) -> Self {
Self {
key: crate::corpus::scenario_key(&feature.name, scenario.position.line, &scenario.name),
feature: Arc::clone(feature),
rule: rule.cloned(),
scenario: Arc::new(scenario.clone()),
}
}

/// Steps in execution order: feature background, rule background, then
/// the scenario's own steps (the order Cucumber's runner uses).
fn steps(&self) -> impl Iterator<Item = &gherkin::Step> {
let feature_background = self.feature.background.iter().flat_map(|b| &b.steps);
let rule_background = self
.rule
.iter()
.flat_map(|rule| rule.background.iter().flat_map(|b| &b.steps));
feature_background
.chain(rule_background)
.chain(&self.scenario.steps)
}
}

/// Divan names each benchmark `scenario[<key>]` from this display value.
impl fmt::Display for ScenarioCase {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
f.write_str(&self.key)
}
}

/// Why a scenario did not pass. Cucumber reports the same three cases as a
/// failed or skipped step, and the correctness run counts none of them as
/// passing.
#[derive(Debug)]
pub struct ScenarioFailure {
pub key: String,
pub step: String,
pub reason: String,
}

impl fmt::Display for ScenarioFailure {
fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
write!(
f,
"TCK benchmark scenario failed; no timing is recorded: {} at step `{}`: {}",
self.key, self.step, self.reason
)
}
}

/// Parse a normalized feature tree with Cucumber's own parser, which expands
/// scenario outlines exactly as the correctness run does. Any parse error, an
/// empty corpus or a duplicate key fails closed.
pub fn load_scenarios(root: &Path) -> Vec<ScenarioCase> {
let parsed: Vec<_> = futures::executor::block_on(
cucumber::parser::Basic::default()
.parse(root, cucumber::parser::basic::Cli::default())
.collect(),
);
let mut cases = Vec::new();
for feature in parsed {
let feature =
Arc::new(feature.unwrap_or_else(|error| panic!("TCK feature parse error: {error}")));
for scenario in &feature.scenarios {
cases.push(ScenarioCase::new(&feature, None, scenario));
}
for rule in &feature.rules {
let rule = Arc::new(rule.clone());
for scenario in &rule.scenarios {
cases.push(ScenarioCase::new(&feature, Some(&rule), scenario));
}
}
}
assert!(
!cases.is_empty(),
"no TCK scenarios found under {}",
root.display()
);
let mut keys = std::collections::BTreeSet::new();
for case in &cases {
assert!(
keys.insert(case.key.as_str()),
"duplicate TCK scenario key {}",
case.key
);
}
cases
}

/// Normalize the feature tree under `source` into a fresh temporary directory,
/// exactly as the Cucumber run does, and load its scenarios. The directory is
/// returned so it outlives the run.
pub fn load_normalized(source: &Path) -> (tempfile::TempDir, Vec<ScenarioCase>) {
let normalized = tempfile::TempDir::new().expect("temp dir for normalized TCK corpus");
crate::corpus::copy_features_normalized(source, normalized.path());
let cases = load_scenarios(normalized.path());
(normalized, cases)
}

static CORPUS: OnceLock<Vec<ScenarioCase>> = OnceLock::new();

/// Install the scenario set the Divan benchmark iterates. Once per process.
pub fn install_corpus(cases: Vec<ScenarioCase>) {
assert!(
CORPUS.set(cases).is_ok(),
"the TCK benchmark corpus is installed once per process"
);
}

/// Divan argument source: the installed corpus. Uninstalled fails closed
/// rather than benchmarking an empty set.
fn scenarios() -> Vec<ScenarioCase> {
CORPUS
.get()
.expect("install_corpus must run before Divan")
.clone()
}

fn collection() -> &'static cucumber::step::Collection<GraphForgeWorld> {
static COLLECTION: OnceLock<cucumber::step::Collection<GraphForgeWorld>> = OnceLock::new();
COLLECTION.get_or_init(GraphForgeWorld::collection)
}

/// The same multi-thread runtime flavour `#[tokio::main]` gives the Cucumber run.
fn runtime() -> &'static tokio::runtime::Runtime {
static RUNTIME: OnceLock<tokio::runtime::Runtime> = OnceLock::new();
RUNTIME.get_or_init(|| {
tokio::runtime::Builder::new_multi_thread()
.enable_all()
.build()
.expect("TCK benchmark runtime")
})
}

fn panic_message(payload: &(dyn std::any::Any + Send)) -> String {
payload
.downcast_ref::<String>()
.cloned()
.or_else(|| payload.downcast_ref::<&str>().map(|s| (*s).to_owned()))
.unwrap_or_else(|| "non-string panic payload".to_owned())
}

async fn run_steps(
world: &mut GraphForgeWorld,
case: &ScenarioCase,
) -> Result<(), ScenarioFailure> {
let failure = |step: &gherkin::Step, reason: String| ScenarioFailure {
key: case.key.clone(),
step: format!("{}{}", step.keyword, step.value),
reason,
};
for step in case.steps() {
let (step_fn, _captures, _location, context) = match collection().find(step) {
Ok(Some(found)) => found,
Ok(None) => return Err(failure(step, "no registered step matches".to_owned())),
Err(error) => return Err(failure(step, format!("ambiguous step: {error}"))),
};
if let Err(payload) = AssertUnwindSafe(step_fn(world, context))
.catch_unwind()
.await
{
return Err(failure(step, panic_message(payload.as_ref())));
}
}
Ok(())
}

/// Execute one whole scenario and return its verdict.
pub fn execute(case: &ScenarioCase) -> Result<(), ScenarioFailure> {
runtime().block_on(async {
let mut world = GraphForgeWorld::new()
.await
.unwrap_or_else(|error| panic!("GraphForgeWorld::new: {error}"));
let verdict = run_steps(&mut world, case).await;
// The Cucumber run's `after` hook: return the fixture to the pool,
// which clears it on the next lease.
crate::fixture::release(&mut world.forge);
verdict
})
}

/// Fail closed on any recorded non-passing verdict. Runs outside the timed
/// region, before Divan can record the sample it belongs to.
fn require_passed(verdicts: &RefCell<Vec<Result<(), ScenarioFailure>>>) {
for verdict in verdicts.borrow_mut().drain(..) {
if let Err(failure) = verdict {
panic!("{failure}");
}
}
}

/// One Divan benchmark per TCK scenario, named `scenario[<feature>:<line>:<name>]`.
///
/// `sample_size = 1` makes each sample exactly one scenario execution; the
/// sample count is a default that `--sample-count` overrides.
#[divan::bench(args = scenarios(), sample_count = 10, sample_size = 1)]
fn scenario(bencher: divan::Bencher, case: &ScenarioCase) {
let verdicts = RefCell::new(Vec::with_capacity(1));
bencher
.with_inputs(|| require_passed(&verdicts))
.bench_local_values(|()| {
let verdict = execute(case);
verdicts.borrow_mut().push(verdict);
});
require_passed(&verdicts);
}

/// The fixture profile both runners share: one pooled engine per concurrency slot.
pub fn assert_fixture_profile() {
let created = crate::fixture::created_count();
assert!(
created <= crate::fixture::TCK_CONCURRENCY,
"TCK fixture pool created {created} engines for concurrency {}",
crate::fixture::TCK_CONCURRENCY
);
}
Loading
Loading