diff --git a/Cargo.lock b/Cargo.lock index 61c907ea9..49b38cce0 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -2158,6 +2158,7 @@ version = "0.5.2" dependencies = [ "arrow", "bytes", + "codspeed-divan-compat", "cucumber", "datafusion", "fs4 1.1.0", diff --git a/Makefile b/Makefile index 8af5cee50..3099f5ce4 100644 --- a/Makefile +++ b/Makefile @@ -1,4 +1,4 @@ -.PHONY: help lint format type-check security workflow-lint license-check third-party-notices third-party-notices-check cargo-deny-licenses test pre-push pre-push-clean pre-push-preflight pre-push-fast clean test-tck docstring-coverage test-network benchmark test-perf test-perf-xs test-perf-slow test-perf-large coverage coverage-rust coverage-python coverage-node coverage-quick coverage-report coverage-diff coverage-strict check-coverage check-coverage-rust check-coverage-python check-coverage-node check-patch-coverage test-durations test-analytics docs-serve docs-build docs-clean cargo-build codspeed-build codspeed-build-walltime codspeed-run bench-traversal bench-fixed-hop-limit bench-fixed-hop-livejournal bench-m4-entry bench-adjacency-200m m4-entry-matrix-check durability-isolation-check native-consumers release-load-matrix-check release-load-matrix bulk-construction-conformance-check bulk-construction-conformance cargo-test cargo-check cargo-clippy cargo-fmt cargo-fmt-check clean-builds clean-builds-all pnpm-install pnpm-build install build release-version-check package-license-verify publish-dry-run publish-dry-run-npm publish-dry-run-docs publish-dry-run-python publish-dry-run-cargo record-release-artifacts clean-env-verify-check clean-env-verify-preflight clean-env-verify +.PHONY: help lint format type-check security workflow-lint license-check third-party-notices third-party-notices-check cargo-deny-licenses test pre-push pre-push-clean pre-push-preflight pre-push-fast clean test-tck docstring-coverage test-network benchmark test-perf test-perf-xs test-perf-slow test-perf-large coverage coverage-rust coverage-python coverage-node coverage-quick coverage-report coverage-diff coverage-strict check-coverage check-coverage-rust check-coverage-python check-coverage-node check-patch-coverage test-durations test-analytics docs-serve docs-build docs-clean cargo-build codspeed-build codspeed-build-walltime codspeed-run bench-traversal bench-tck-scenarios bench-fixed-hop-limit bench-fixed-hop-livejournal bench-m4-entry bench-adjacency-200m m4-entry-matrix-check durability-isolation-check native-consumers release-load-matrix-check release-load-matrix bulk-construction-conformance-check bulk-construction-conformance cargo-test cargo-check cargo-clippy cargo-fmt cargo-fmt-check clean-builds clean-builds-all pnpm-install pnpm-build install build release-version-check package-license-verify publish-dry-run publish-dry-run-npm publish-dry-run-docs publish-dry-run-python publish-dry-run-cargo record-release-artifacts clean-env-verify-check clean-env-verify-preflight clean-env-verify help: ## Show this help message @grep -E '^[a-zA-Z_-]+:.*?## .*$$' $(MAKEFILE_LIST) | sort | awk 'BEGIN {FS = ":.*?## "}; {printf "\033[36m%-20s\033[0m %s\n", $$1, $$2}' @@ -311,6 +311,9 @@ codspeed-run: ## Run the CodSpeed benchmarks locally (requires the codspeed CLI bench-traversal: ## Run the #767 traversal scaling Divan benchmarks (release, manual; see benchmarks/traversal_scaling.md) cargo bench -p graphforge-exec --bench traversal_scaling -- --sample-count 5 +bench-tck-scenarios: ## Run the #1653 per-scenario openCypher TCK Divan benchmark (manual; raw results under CODSPEED_ENV) + cargo bench -p graphforge-api --bench tck_scenarios + bench-merge-scaling: ## Run the #1400 node MERGE scaling Divan benchmarks (release, manual) cargo bench -p graphforge-exec --bench merge_scaling -- --sample-count 5 diff --git a/config/benchmark-measurement-inventory.json b/config/benchmark-measurement-inventory.json index d48352956..4ed403a26 100644 --- a/config/benchmark-measurement-inventory.json +++ b/config/benchmark-measurement-inventory.json @@ -55,6 +55,22 @@ "owner_issue": null, "notes": "Node MERGE scaling wall-clock workloads; topology-read bound stays in merge_scaling_bench integration tests." }, + { + "path": "crates/graphforge-api/benches/tck_scenarios/main.rs", + "boundary": "in_process", + "authority": "divan", + "disposition": "framework_authority", + "owner_issue": null, + "notes": "openCypher TCK per-scenario Divan benchmark (#1653): normalized corpus and pooled fixture setup; timing from Divan only, serialized as CodSpeed walltime raw_results under CODSPEED_ENV." + }, + { + "path": "crates/graphforge-api/benches/tck_scenarios/runner.rs", + "boundary": "in_process", + "authority": "divan", + "disposition": "framework_authority", + "owner_issue": null, + "notes": "TCK scenario execution through the registered cucumber steps; the timed region is the whole scenario and each verdict is checked outside it, so a failing scenario aborts before Divan records timing." + }, { "path": "crates/graphforge-exec/tests/persistent_adjacency.rs", "boundary": "in_process", diff --git a/crates/graphforge-api/Cargo.toml b/crates/graphforge-api/Cargo.toml index 912c98521..69129b77a 100644 --- a/crates/graphforge-api/Cargo.toml +++ b/crates/graphforge-api/Cargo.toml @@ -54,6 +54,7 @@ uuid = { workspace = true } # Assessment-only IPC codec experiments; production writers remain unchanged. arrow = { workspace = true, features = ["ipc_compression"] } cucumber = { workspace = true } +divan = { workspace = true } tokio = { workspace = true } graphforge-cypher = { path = "../graphforge-cypher" } graphforge-storage = { path = "../graphforge-storage", features = ["test-failpoints", "test-support"] } @@ -213,5 +214,11 @@ required-features = ["research", "search"] name = "bdd" harness = false +# In-process Divan benchmark over the same TCK scenarios, step functions and +# pooled fixture as the `bdd` run (#1653). Divan's own harness drives `main`. +[[bench]] +name = "tck_scenarios" +harness = false + [lints] workspace = true diff --git a/crates/graphforge-api/benches/tck_scenarios/main.rs b/crates/graphforge-api/benches/tck_scenarios/main.rs new file mode 100644 index 000000000..f0ac6d37a --- /dev/null +++ b/crates/graphforge-api/benches/tck_scenarios/main.rs @@ -0,0 +1,52 @@ +//! In-process Divan benchmark over the openCypher TCK scenarios (#1653). +//! +//! Parses the same ephemeral normalized corpus as the Cucumber correctness run +//! (`tests/bdd/main.rs`) and times each scenario, executed through the same +//! registered step functions, pooled fixture and clear-on-lease semantics. See +//! `runner.rs` for the timed region and the fail-closed verdict check. +//! +//! ```bash +//! cargo bench -p graphforge-api --bench tck_scenarios # measure +//! cargo bench -p graphforge-api --bench tck_scenarios -- --test # test mode +//! ``` +//! +//! Divan does the measuring. Machine-readable per-scenario output is CodSpeed's +//! walltime `raw_results`, written only when `CODSPEED_ENV` is set; see +//! `docs/development/benchmarking.md`. Test mode runs every scenario once and +//! emits no timing. `TCK_ONLY=` restricts the corpus as it does for the +//! Cucumber run. + +#[cfg(feature = "search")] +#[path = "../../tests/bdd/api_steps.rs"] +mod api_steps; +#[path = "../../tests/bdd/corpus.rs"] +mod corpus; +#[path = "../../tests/bdd/fixture.rs"] +mod fixture; +mod runner; +#[path = "../../tests/bdd/tck_steps.rs"] +mod tck_steps; +#[path = "../../tests/bdd/world.rs"] +mod world; + +use world::GraphForgeWorld; + +fn main() { + let workspace_root = std::path::Path::new(env!("CARGO_MANIFEST_DIR")) + .join("../..") + .canonicalize() + .expect("workspace root must exist"); + let (normalized, cases) = runner::load_normalized(&workspace_root.join("tests/tck/features")); + eprintln!( + "TCK scenario benchmark: {} scenarios, fixture profile pooled-isolated-serial-v1, concurrency {}", + cases.len(), + fixture::TCK_CONCURRENCY + ); + runner::install_corpus(cases); + + let fixture_guard = fixture::activate(); + divan::main(); + runner::assert_fixture_profile(); + drop(fixture_guard); + drop(normalized); +} diff --git a/crates/graphforge-api/benches/tck_scenarios/runner.rs b/crates/graphforge-api/benches/tck_scenarios/runner.rs new file mode 100644 index 000000000..90b35bc4f --- /dev/null +++ b/crates/graphforge-api/benches/tck_scenarios/runner.rs @@ -0,0 +1,259 @@ +//! Scenario loading, execution and the Divan benchmark over the TCK corpus. +//! +//! Every step runs through the step functions registered against +//! [`GraphForgeWorld`] and resolved with `GraphForgeWorld::collection().find()`, +//! the same registry the Cucumber correctness run uses. There is no second +//! semantics engine. +//! +//! One timed iteration is one whole scenario: world construction, feature and +//! rule backgrounds, every scenario step (including `Given an empty graph`, +//! which leases and clears the pooled fixture) and the fixture release that the +//! Cucumber run performs in its `after` hook. Each iteration's pass/fail verdict +//! is recorded inside the timed region and checked outside it, before the next +//! sample starts and before the benchmark function returns. Divan emits a +//! benchmark's timing only after that function returns, so a failing scenario +//! panics first and never yields timing. + +use std::cell::RefCell; +use std::fmt; +use std::panic::AssertUnwindSafe; +use std::path::Path; +use std::sync::{Arc, OnceLock}; + +use cucumber::gherkin; +use cucumber::{Parser as _, World as _}; +use futures::{FutureExt as _, StreamExt as _}; + +use crate::GraphForgeWorld; + +/// One expanded TCK scenario, keyed `::` exactly as the +/// Cucumber runner and `tests/tck/passing_baseline.txt` key it. +#[derive(Clone)] +pub struct ScenarioCase { + key: String, + feature: Arc, + rule: Option>, + scenario: Arc, +} + +impl ScenarioCase { + fn new( + feature: &Arc, + rule: Option<&Arc>, + scenario: &gherkin::Scenario, + ) -> Self { + Self { + key: crate::corpus::scenario_key(&feature.name, scenario.position.line, &scenario.name), + feature: Arc::clone(feature), + rule: rule.cloned(), + scenario: Arc::new(scenario.clone()), + } + } + + /// Steps in execution order: feature background, rule background, then + /// the scenario's own steps (the order Cucumber's runner uses). + fn steps(&self) -> impl Iterator { + let feature_background = self.feature.background.iter().flat_map(|b| &b.steps); + let rule_background = self + .rule + .iter() + .flat_map(|rule| rule.background.iter().flat_map(|b| &b.steps)); + feature_background + .chain(rule_background) + .chain(&self.scenario.steps) + } +} + +/// Divan names each benchmark `scenario[]` from this display value. +impl fmt::Display for ScenarioCase { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + f.write_str(&self.key) + } +} + +/// Why a scenario did not pass. Cucumber reports the same three cases as a +/// failed or skipped step, and the correctness run counts none of them as +/// passing. +#[derive(Debug)] +pub struct ScenarioFailure { + pub key: String, + pub step: String, + pub reason: String, +} + +impl fmt::Display for ScenarioFailure { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + write!( + f, + "TCK benchmark scenario failed; no timing is recorded: {} at step `{}`: {}", + self.key, self.step, self.reason + ) + } +} + +/// Parse a normalized feature tree with Cucumber's own parser, which expands +/// scenario outlines exactly as the correctness run does. Any parse error, an +/// empty corpus or a duplicate key fails closed. +pub fn load_scenarios(root: &Path) -> Vec { + let parsed: Vec<_> = futures::executor::block_on( + cucumber::parser::Basic::default() + .parse(root, cucumber::parser::basic::Cli::default()) + .collect(), + ); + let mut cases = Vec::new(); + for feature in parsed { + let feature = + Arc::new(feature.unwrap_or_else(|error| panic!("TCK feature parse error: {error}"))); + for scenario in &feature.scenarios { + cases.push(ScenarioCase::new(&feature, None, scenario)); + } + for rule in &feature.rules { + let rule = Arc::new(rule.clone()); + for scenario in &rule.scenarios { + cases.push(ScenarioCase::new(&feature, Some(&rule), scenario)); + } + } + } + assert!( + !cases.is_empty(), + "no TCK scenarios found under {}", + root.display() + ); + let mut keys = std::collections::BTreeSet::new(); + for case in &cases { + assert!( + keys.insert(case.key.as_str()), + "duplicate TCK scenario key {}", + case.key + ); + } + cases +} + +/// Normalize the feature tree under `source` into a fresh temporary directory, +/// exactly as the Cucumber run does, and load its scenarios. The directory is +/// returned so it outlives the run. +pub fn load_normalized(source: &Path) -> (tempfile::TempDir, Vec) { + let normalized = tempfile::TempDir::new().expect("temp dir for normalized TCK corpus"); + crate::corpus::copy_features_normalized(source, normalized.path()); + let cases = load_scenarios(normalized.path()); + (normalized, cases) +} + +static CORPUS: OnceLock> = OnceLock::new(); + +/// Install the scenario set the Divan benchmark iterates. Once per process. +pub fn install_corpus(cases: Vec) { + assert!( + CORPUS.set(cases).is_ok(), + "the TCK benchmark corpus is installed once per process" + ); +} + +/// Divan argument source: the installed corpus. Uninstalled fails closed +/// rather than benchmarking an empty set. +fn scenarios() -> Vec { + CORPUS + .get() + .expect("install_corpus must run before Divan") + .clone() +} + +fn collection() -> &'static cucumber::step::Collection { + static COLLECTION: OnceLock> = OnceLock::new(); + COLLECTION.get_or_init(GraphForgeWorld::collection) +} + +/// The same multi-thread runtime flavour `#[tokio::main]` gives the Cucumber run. +fn runtime() -> &'static tokio::runtime::Runtime { + static RUNTIME: OnceLock = OnceLock::new(); + RUNTIME.get_or_init(|| { + tokio::runtime::Builder::new_multi_thread() + .enable_all() + .build() + .expect("TCK benchmark runtime") + }) +} + +fn panic_message(payload: &(dyn std::any::Any + Send)) -> String { + payload + .downcast_ref::() + .cloned() + .or_else(|| payload.downcast_ref::<&str>().map(|s| (*s).to_owned())) + .unwrap_or_else(|| "non-string panic payload".to_owned()) +} + +async fn run_steps( + world: &mut GraphForgeWorld, + case: &ScenarioCase, +) -> Result<(), ScenarioFailure> { + let failure = |step: &gherkin::Step, reason: String| ScenarioFailure { + key: case.key.clone(), + step: format!("{}{}", step.keyword, step.value), + reason, + }; + for step in case.steps() { + let (step_fn, _captures, _location, context) = match collection().find(step) { + Ok(Some(found)) => found, + Ok(None) => return Err(failure(step, "no registered step matches".to_owned())), + Err(error) => return Err(failure(step, format!("ambiguous step: {error}"))), + }; + if let Err(payload) = AssertUnwindSafe(step_fn(world, context)) + .catch_unwind() + .await + { + return Err(failure(step, panic_message(payload.as_ref()))); + } + } + Ok(()) +} + +/// Execute one whole scenario and return its verdict. +pub fn execute(case: &ScenarioCase) -> Result<(), ScenarioFailure> { + runtime().block_on(async { + let mut world = GraphForgeWorld::new() + .await + .unwrap_or_else(|error| panic!("GraphForgeWorld::new: {error}")); + let verdict = run_steps(&mut world, case).await; + // The Cucumber run's `after` hook: return the fixture to the pool, + // which clears it on the next lease. + crate::fixture::release(&mut world.forge); + verdict + }) +} + +/// Fail closed on any recorded non-passing verdict. Runs outside the timed +/// region, before Divan can record the sample it belongs to. +fn require_passed(verdicts: &RefCell>>) { + for verdict in verdicts.borrow_mut().drain(..) { + if let Err(failure) = verdict { + panic!("{failure}"); + } + } +} + +/// One Divan benchmark per TCK scenario, named `scenario[::]`. +/// +/// `sample_size = 1` makes each sample exactly one scenario execution; the +/// sample count is a default that `--sample-count` overrides. +#[divan::bench(args = scenarios(), sample_count = 10, sample_size = 1)] +fn scenario(bencher: divan::Bencher, case: &ScenarioCase) { + let verdicts = RefCell::new(Vec::with_capacity(1)); + bencher + .with_inputs(|| require_passed(&verdicts)) + .bench_local_values(|()| { + let verdict = execute(case); + verdicts.borrow_mut().push(verdict); + }); + require_passed(&verdicts); +} + +/// The fixture profile both runners share: one pooled engine per concurrency slot. +pub fn assert_fixture_profile() { + let created = crate::fixture::created_count(); + assert!( + created <= crate::fixture::TCK_CONCURRENCY, + "TCK fixture pool created {created} engines for concurrency {}", + crate::fixture::TCK_CONCURRENCY + ); +} diff --git a/crates/graphforge-api/tests/bdd/corpus.rs b/crates/graphforge-api/tests/bdd/corpus.rs new file mode 100644 index 000000000..59d345ed1 --- /dev/null +++ b/crates/graphforge-api/tests/bdd/corpus.rs @@ -0,0 +1,97 @@ +//! Ephemeral normalization of the vendored openCypher TCK corpus. +//! +//! Shared by the Cucumber runner (`tests/bdd/main.rs`) and the in-process Divan +//! TCK benchmark (`benches/tck_scenarios/`), so both execute the same scenario +//! set from the same normalized feature text. + +/// The scenario key used by the passing baseline and every per-scenario +/// report: `::`. +/// +/// Deliberately keyed by FEATURE NAME (unique per TCK file), not the file +/// path: the normalized corpus lives under a temp dir whose canonicalization +/// differs by platform (macOS `/private` symlinks), which would make keys, and +/// thus the baseline, non-portable between local and CI. +pub fn scenario_key(feature_name: &str, line: usize, scenario_name: &str) -> String { + format!("{feature_name}:{line}:{scenario_name}") +} + +/// The `TCK_ONLY` local-iteration substring filter, or `None` if unset/empty. +/// An empty value is treated as unset so a stray `TCK_ONLY=` can't silently +/// bypass the baseline gate (`contains("")` is always true). +pub fn tck_only_filter() -> Option { + std::env::var("TCK_ONLY") + .ok() + .filter(|s| !s.trim().is_empty()) +} + +/// Recursively copy the vendored TCK feature tree from `src` into `dst`, rewriting +/// only block-leading `And`/`But` continuation keywords to `Given` (see +/// [`normalize_leading_continuations`]). The vendored source files are never modified. +pub fn copy_features_normalized(src: &std::path::Path, dst: &std::path::Path) { + for entry in std::fs::read_dir(src).expect("read TCK feature dir") { + let path = entry.expect("dir entry").path(); + let target = dst.join(path.file_name().expect("entry file name")); + if path.is_dir() { + std::fs::create_dir_all(&target).expect("create temp subdir"); + copy_features_normalized(&path, &target); + } else if path.extension().is_some_and(|e| e == "feature") { + // Local iteration: `TCK_ONLY=` restricts the corpus to + // feature files whose path contains `` (e.g. + // `TCK_ONLY=Temporal`) for a fast subset run. Unset in CI → the + // whole corpus. The baseline gate is skipped when set (see `main`), + // since a subset can't satisfy the whole-corpus baseline. + if let Some(filter) = tck_only_filter() + && !path.to_string_lossy().contains(&filter) + { + continue; + } + let content = std::fs::read_to_string(&path).expect("read feature file"); + std::fs::write(&target, normalize_leading_continuations(&content)) + .expect("write temp feature"); + } + } +} + +/// Rewrite a step that is the FIRST step of its `Scenario`/`Scenario Outline`/ +/// `Background`/`Rule`/`Example` block and uses the `And`/`But` continuation keyword +/// into `Given`. The Rust `gherkin` parser rejects a block-leading `And`/`But`, while +/// cucumber-js accepts it; semantics are unchanged (the continuation inherits `Given`). +/// Only block-leading steps are touched — `And`/`But` after a concrete step are left as-is. +pub fn normalize_leading_continuations(content: &str) -> String { + let mut out = String::with_capacity(content.len() + 64); + let mut awaiting_first_step = false; + for line in content.lines() { + let trimmed = line.trim_start(); + let is_header = trimmed.starts_with("Scenario:") + || trimmed.starts_with("Scenario Outline:") + || trimmed.starts_with("Background:") + || trimmed.starts_with("Rule:") + || trimmed.starts_with("Example:"); + let is_step = ["Given ", "When ", "Then ", "And ", "But "] + .iter() + .any(|kw| trimmed.starts_with(kw)); + if is_header { + awaiting_first_step = true; + out.push_str(line); + } else if awaiting_first_step + && (trimmed.starts_with("And ") || trimmed.starts_with("But ")) + { + let indent = &line[..line.len() - trimmed.len()]; + let rest = trimmed + .strip_prefix("And ") + .or_else(|| trimmed.strip_prefix("But ")) + .expect("And/But prefix present"); + out.push_str(indent); + out.push_str("Given "); + out.push_str(rest); + awaiting_first_step = false; + } else { + if is_step { + awaiting_first_step = false; + } + out.push_str(line); + } + out.push('\n'); + } + out +} diff --git a/crates/graphforge-api/tests/bdd/main.rs b/crates/graphforge-api/tests/bdd/main.rs index ca03dd022..333c03476 100644 --- a/crates/graphforge-api/tests/bdd/main.rs +++ b/crates/graphforge-api/tests/bdd/main.rs @@ -13,65 +13,23 @@ #[cfg(feature = "search")] mod api_steps; +mod corpus; mod fixture; mod tck_steps; mod timing; +mod world; use cucumber::{World, WriterExt}; use futures::FutureExt; +use corpus::{copy_features_normalized, tck_only_filter}; +use world::GraphForgeWorld; + use timing::{ ScenarioTimer, ScenarioTiming, Suite, annotation_messages, baseline_candidate, build_report, escape_github_command, load_baseline, load_policy, non_passing_scenario_keys, write_artifacts, }; -/// Shared cucumber [`World`] for the GraphForge BDD suites (public API + TCK). -#[derive(Debug, Default, World)] -pub struct GraphForgeWorld { - /// The forge instance under test (None until a Given step creates it). - pub forge: Option, - /// Owns a persistent-project fixture directory for lifecycle scenarios. - pub persistent_fixture: Option, - /// Owns an ontology fixture directory for load scenarios. - pub ontology_fixture: Option, - /// Ontology fixture path selected by the Given step. - pub ontology_path: Option, - /// Last error returned by a When step. - pub last_error: Option, - /// Stable public code for the last typed Rust facade error. - pub last_error_code: Option<&'static str>, - /// Typed planning InvalidType rejection; its public code is GF_VALIDATION. - pub last_compile_type_error: bool, - /// Last metadata collection returned by labels or relationship_types. - pub last_names: Option>, - /// Last scalar returned by node_count. - pub last_count: Option, - /// Last explanation returned by explain. - pub last_explanation: Option, - /// Last Arrow-backed result returned by `execute()`. - pub last_exec: Option, - /// Most recent Arrow result returned by an analyst verb. - pub last_algorithm_result: Option, - /// Previous analyst result, retained for comparison scenarios. - pub previous_algorithm_result: Option, - /// Query parameters bound by openCypher TCK `And parameters are:` steps. - pub params: std::collections::HashMap, - /// Node handles by name. - pub nodes: std::collections::HashMap, - /// Most recently created node handle for result-focused assertions. - pub last_node_handle: Option, - /// Most recently created edge handle for result-focused assertions. - pub last_edge_handle: Option, - /// Number of explicit public index calls made in this scenario. - pub index_calls: usize, - /// Stored query/index vector for find/index scenarios. - pub stored_vector: Option>, - /// Caller-defined vector space used by find/index fixtures. - pub stored_space: Option, - /// Stored node UUID (hex or hyphenated) for index upsert scenarios. - pub stored_paper_id: Option, -} - #[tokio::main] async fn main() { // Resolve paths relative to the workspace root so the runner works from @@ -454,19 +412,11 @@ impl cucumber::Writer for ScenarioCollector { Feature::Rule(_, Rule::Scenario(scenario, retry)) => (scenario, retry), _ => return, }; - let key = format!( - "{}:{}:{}", - feature.name, scenario.position.line, scenario.name - ); + let key = corpus::scenario_key(&feature.name, scenario.position.line, &scenario.name); let attempt = retry.retries.map_or(0, |retries| retries.current); let active_key = (key.clone(), attempt); match retry.event { Scenario::Started => { - // Key by FEATURE NAME (unique per TCK file) + line + scenario - // name. Deliberately NOT the file path: the normalized corpus - // lives under a temp dir whose canonicalization differs by - // platform (macOS `/private` symlinks), which would make keys — - // and thus the baseline — non-portable between local and CI. self.timer .start( self.suite, @@ -577,84 +527,3 @@ fn load_passing_baseline(path: &std::path::Path) -> std::collections::BTreeSet Option { - std::env::var("TCK_ONLY") - .ok() - .filter(|s| !s.trim().is_empty()) -} - -fn copy_features_normalized(src: &std::path::Path, dst: &std::path::Path) { - for entry in std::fs::read_dir(src).expect("read TCK feature dir") { - let path = entry.expect("dir entry").path(); - let target = dst.join(path.file_name().expect("entry file name")); - if path.is_dir() { - std::fs::create_dir_all(&target).expect("create temp subdir"); - copy_features_normalized(&path, &target); - } else if path.extension().is_some_and(|e| e == "feature") { - // Local iteration: `TCK_ONLY=` restricts the corpus to - // feature files whose path contains `` (e.g. - // `TCK_ONLY=Temporal`) for a fast subset run. Unset in CI → the - // whole corpus. The baseline gate is skipped when set (see `main`), - // since a subset can't satisfy the whole-corpus baseline. - if let Some(filter) = tck_only_filter() - && !path.to_string_lossy().contains(&filter) - { - continue; - } - let content = std::fs::read_to_string(&path).expect("read feature file"); - std::fs::write(&target, normalize_leading_continuations(&content)) - .expect("write temp feature"); - } - } -} - -/// Rewrite a step that is the FIRST step of its `Scenario`/`Scenario Outline`/ -/// `Background`/`Rule`/`Example` block and uses the `And`/`But` continuation keyword -/// into `Given`. The Rust `gherkin` parser rejects a block-leading `And`/`But`, while -/// cucumber-js accepts it; semantics are unchanged (the continuation inherits `Given`). -/// Only block-leading steps are touched — `And`/`But` after a concrete step are left as-is. -fn normalize_leading_continuations(content: &str) -> String { - let mut out = String::with_capacity(content.len() + 64); - let mut awaiting_first_step = false; - for line in content.lines() { - let trimmed = line.trim_start(); - let is_header = trimmed.starts_with("Scenario:") - || trimmed.starts_with("Scenario Outline:") - || trimmed.starts_with("Background:") - || trimmed.starts_with("Rule:") - || trimmed.starts_with("Example:"); - let is_step = ["Given ", "When ", "Then ", "And ", "But "] - .iter() - .any(|kw| trimmed.starts_with(kw)); - if is_header { - awaiting_first_step = true; - out.push_str(line); - } else if awaiting_first_step - && (trimmed.starts_with("And ") || trimmed.starts_with("But ")) - { - let indent = &line[..line.len() - trimmed.len()]; - let rest = trimmed - .strip_prefix("And ") - .or_else(|| trimmed.strip_prefix("But ")) - .expect("And/But prefix present"); - out.push_str(indent); - out.push_str("Given "); - out.push_str(rest); - awaiting_first_step = false; - } else { - if is_step { - awaiting_first_step = false; - } - out.push_str(line); - } - out.push('\n'); - } - out -} diff --git a/crates/graphforge-api/tests/bdd/world.rs b/crates/graphforge-api/tests/bdd/world.rs new file mode 100644 index 000000000..d3b1da08e --- /dev/null +++ b/crates/graphforge-api/tests/bdd/world.rs @@ -0,0 +1,54 @@ +//! The shared cucumber [`World`] for the GraphForge BDD suites. +//! +//! Shared by the Cucumber runner (`tests/bdd/main.rs`) and the in-process Divan +//! TCK benchmark (`benches/tck_scenarios/`), which both execute scenarios +//! through the step functions registered against this type. + +use cucumber::World; + +/// Shared cucumber [`World`] for the GraphForge BDD suites (public API + TCK). +#[derive(Debug, Default, World)] +pub struct GraphForgeWorld { + /// The forge instance under test (None until a Given step creates it). + pub forge: Option, + /// Owns a persistent-project fixture directory for lifecycle scenarios. + pub persistent_fixture: Option, + /// Owns an ontology fixture directory for load scenarios. + pub ontology_fixture: Option, + /// Ontology fixture path selected by the Given step. + pub ontology_path: Option, + /// Last error returned by a When step. + pub last_error: Option, + /// Stable public code for the last typed Rust facade error. + pub last_error_code: Option<&'static str>, + /// Typed planning InvalidType rejection; its public code is GF_VALIDATION. + pub last_compile_type_error: bool, + /// Last metadata collection returned by labels or relationship_types. + pub last_names: Option>, + /// Last scalar returned by node_count. + pub last_count: Option, + /// Last explanation returned by explain. + pub last_explanation: Option, + /// Last Arrow-backed result returned by `execute()`. + pub last_exec: Option, + /// Most recent Arrow result returned by an analyst verb. + pub last_algorithm_result: Option, + /// Previous analyst result, retained for comparison scenarios. + pub previous_algorithm_result: Option, + /// Query parameters bound by openCypher TCK `And parameters are:` steps. + pub params: std::collections::HashMap, + /// Node handles by name. + pub nodes: std::collections::HashMap, + /// Most recently created node handle for result-focused assertions. + pub last_node_handle: Option, + /// Most recently created edge handle for result-focused assertions. + pub last_edge_handle: Option, + /// Number of explicit public index calls made in this scenario. + pub index_calls: usize, + /// Stored query/index vector for find/index scenarios. + pub stored_vector: Option>, + /// Caller-defined vector space used by find/index fixtures. + pub stored_space: Option, + /// Stored node UUID (hex or hyphenated) for index upsert scenarios. + pub stored_paper_id: Option, +} diff --git a/crates/graphforge-api/tests/tck_scenario_bench.rs b/crates/graphforge-api/tests/tck_scenario_bench.rs new file mode 100644 index 000000000..7c48ae3dd --- /dev/null +++ b/crates/graphforge-api/tests/tck_scenario_bench.rs @@ -0,0 +1,218 @@ +//! Direct tests for the in-process Divan TCK benchmark (#1653). +//! +//! Each case re-runs this binary as a child that installs a small feature +//! corpus and runs the real `benches/tck_scenarios` Divan benchmark in-process +//! with `CODSPEED_ENV` set, so CodSpeed's walltime `raw_results` are written to +//! a temporary workspace root. The parent then inspects the exit status and the +//! raw results: +//! +//! * a passing scenario in bench mode yields one raw result keyed by scenario +//! (the known positive that makes the negative cases meaningful); +//! * a deliberately failing step aborts the run and its scenario yields no +//! timing; +//! * test mode executes the scenario and yields no performance evidence. + +#[cfg(feature = "search")] +#[path = "bdd/api_steps.rs"] +mod api_steps; +#[path = "bdd/corpus.rs"] +mod corpus; +#[path = "bdd/fixture.rs"] +mod fixture; +#[path = "../benches/tck_scenarios/runner.rs"] +mod runner; +#[path = "bdd/tck_steps.rs"] +mod tck_steps; +#[path = "bdd/world.rs"] +mod world; + +use std::path::{Path, PathBuf}; +use std::process::{Command, Output}; + +use world::GraphForgeWorld; + +const CHILD_MODE: &str = "GF_TCK_BENCH_CHILD_MODE"; +const CHILD_CORPUS: &str = "GF_TCK_BENCH_CHILD_CORPUS"; +const CHILD_SAMPLES: u32 = 3; + +const PASSING_KEY: &str = "BenchFault:3:[1] Passing scenario"; +const FAILING_KEY: &str = "BenchFault:13:[2] Deliberately failing scenario"; + +const PASSING_SCENARIO: &str = r#"Feature: BenchFault + + Scenario: [1] Passing scenario + Given an empty graph + When executing query: + """ + RETURN 1 AS x + """ + Then the result should be, in any order: + | x | + | 1 | +"#; + +/// Appended after `PASSING_SCENARIO`: a registered step whose assertion fails, +/// because `RETURN 1` is not empty. +const FAILING_SCENARIO: &str = r#" + Scenario: [2] Deliberately failing scenario + Given an empty graph + When executing query: + """ + RETURN 1 AS x + """ + Then the result should be empty +"#; + +/// Subprocess entry point. It does nothing unless a parent test launched it. +#[test] +#[ignore = "subprocess entry point for the Divan benchmark tests in this file"] +fn divan_child() { + let mode = std::env::var(CHILD_MODE).expect("child mode"); + let corpus = PathBuf::from(std::env::var_os(CHILD_CORPUS).expect("child corpus")); + let (_normalized, cases) = runner::load_normalized(&corpus); + runner::install_corpus(cases); + let guard = fixture::activate(); + let divan = divan::Divan::default().sample_count(CHILD_SAMPLES); + match mode.as_str() { + "bench" => divan.run_benches(), + "test" => divan.test_benches(), + other => panic!("unknown child mode {other}"), + } + runner::assert_fixture_profile(); + drop(guard); +} + +struct ChildRun { + output: Output, + raw_results: Vec, +} + +impl ChildRun { + fn stderr(&self) -> String { + String::from_utf8_lossy(&self.output.stderr).into_owned() + } + + fn names(&self) -> Vec<&str> { + self.raw_results + .iter() + .map(|result| result["name"].as_str().expect("raw result name")) + .collect() + } +} + +fn run_child(mode: &str, feature: &str) -> ChildRun { + let scratch = tempfile::TempDir::new().expect("scratch dir"); + let corpus = scratch.path().join("features"); + let workspace = scratch.path().join("workspace"); + std::fs::create_dir_all(&corpus).expect("corpus dir"); + std::fs::create_dir_all(&workspace).expect("workspace dir"); + std::fs::write(corpus.join("BenchFault.feature"), feature).expect("write feature"); + + let output = Command::new(std::env::current_exe().expect("test binary")) + .args([ + "divan_child", + "--exact", + "--include-ignored", + "--nocapture", + "--test-threads=1", + ]) + // A local `TCK_ONLY` filter would drop this corpus's feature file. + .env_remove("TCK_ONLY") + .env(CHILD_MODE, mode) + .env(CHILD_CORPUS, &corpus) + .env("CODSPEED_ENV", "local") + .env("CODSPEED_CARGO_WORKSPACE_ROOT", &workspace) + .output() + .expect("spawn Divan child"); + ChildRun { + output, + raw_results: raw_results(&workspace), + } +} + +/// Every walltime raw result CodSpeed's Divan integration wrote under `root`. +fn raw_results(root: &Path) -> Vec { + let dir = root.join("target/codspeed/walltime/raw_results/divan"); + let Ok(entries) = std::fs::read_dir(&dir) else { + return Vec::new(); + }; + entries + .map(|entry| { + let path = entry.expect("raw result entry").path(); + serde_json::from_slice(&std::fs::read(&path).expect("read raw result")) + .expect("raw result JSON") + }) + .collect() +} + +#[test] +fn passing_scenario_emits_walltime_raw_results_keyed_by_scenario() { + let run = run_child("bench", PASSING_SCENARIO); + assert!( + run.output.status.success(), + "bench failed:\n{}", + run.stderr() + ); + assert_eq!(run.names(), [format!("scenario[{PASSING_KEY}]")]); + let stats = &run.raw_results[0]["stats"]; + assert_eq!(stats["rounds"], u64::from(CHILD_SAMPLES)); + assert_eq!(stats["iter_per_round"], 1); + assert!(stats["min_ns"].as_f64().expect("min_ns") > 0.0); +} + +#[test] +fn failing_step_aborts_the_bench_without_recording_timing() { + let run = run_child("bench", &format!("{PASSING_SCENARIO}{FAILING_SCENARIO}")); + let stderr = run.stderr(); + assert!( + !run.output.status.success(), + "a failing scenario must abort the bench:\n{stderr}" + ); + assert!( + stderr.contains(&format!( + "TCK benchmark scenario failed; no timing is recorded: {FAILING_KEY} at step \ + `Then the result should be empty`" + )), + "the abort must come from the failing step's verdict:\n{stderr}" + ); + let failing = format!("scenario[{FAILING_KEY}]"); + assert!( + !run.names().contains(&failing.as_str()), + "the failing scenario yielded timing: {:?}", + run.names() + ); +} + +#[test] +fn test_mode_executes_scenarios_without_performance_evidence() { + let run = run_child("test", PASSING_SCENARIO); + assert!( + run.output.status.success(), + "test mode failed:\n{}", + run.stderr() + ); + assert!( + run.raw_results.is_empty(), + "test mode wrote performance evidence: {:?}", + run.names() + ); + + // Test mode still executes, and so still enforces, every scenario's verdict. + let run = run_child("test", &format!("{PASSING_SCENARIO}{FAILING_SCENARIO}")); + assert!( + !run.output.status.success(), + "test mode must fail on a failing scenario:\n{}", + run.stderr() + ); + assert!(run.raw_results.is_empty()); +} + +#[test] +fn scenario_keys_match_the_cucumber_runner() { + let scratch = tempfile::TempDir::new().expect("scratch dir"); + let feature = format!("{PASSING_SCENARIO}{FAILING_SCENARIO}"); + std::fs::write(scratch.path().join("BenchFault.feature"), feature).expect("write feature"); + let cases = runner::load_scenarios(scratch.path()); + let keys: Vec = cases.iter().map(ToString::to_string).collect(); + assert_eq!(keys, [PASSING_KEY, FAILING_KEY]); +} diff --git a/docs/development/benchmarking.md b/docs/development/benchmarking.md index fce18fa4c..1817a147e 100644 --- a/docs/development/benchmarking.md +++ b/docs/development/benchmarking.md @@ -46,6 +46,47 @@ statistics in the scanned in-process benchmark surfaces (`crates/*/benches/`, when adding a reviewed legacy exception or completing a migration; stale entries fail closed. +### CodSpeed walltime raw results are Divan evidence + +Maintainer decision on #1467 (2026-09-30): when a Divan target runs with +`CODSPEED_ENV` set, the per-benchmark walltime `raw_results` JSON that the +CodSpeed Divan integration writes +(`$CODSPEED_CARGO_WORKSPACE_ROOT/target/codspeed/walltime/raw_results/divan/*.json`) +is accepted as Divan evidence. Divan does the measuring; CodSpeed only +serializes the samples Divan collected. This does not make CodSpeed a merge +authority, and it does not admit any hand-written timer. Divan test mode +(`--test`, or `cargo test` on a bench target) runs each benchmark once and +writes no raw results, so it is never performance evidence. + +### openCypher TCK scenario benchmark + +`crates/graphforge-api/benches/tck_scenarios/` (#1653) is the in-process +per-scenario measurement boundary for the TCK. It parses the same ephemeral +normalized corpus as the Cucumber correctness run, and executes every step +through the step functions registered on `GraphForgeWorld` with the same pooled +fixture and clear-on-lease semantics. One timed iteration is a whole scenario, +including `Given an empty graph`, so fixture reset cost stays visible. Each +iteration's verdict is checked outside the timed region; a failing scenario +aborts the run before Divan records its timing. Benchmarks are named +`scenario[::]`, the key `tests/tck/passing_baseline.txt` +uses. The default is 10 samples of one scenario execution each. + +```bash +cargo bench -p graphforge-api --bench tck_scenarios -- --test # run every scenario once, no timing +CODSPEED_ENV=local CODSPEED_CARGO_WORKSPACE_ROOT="$PWD" \ + cargo bench -p graphforge-api --bench tck_scenarios # whole corpus, raw results +TCK_ONLY=Delete5 cargo bench -p graphforge-api --bench tck_scenarios # feature-file subset +``` + +`make bench-tck-scenarios` runs the whole corpus with the default sample count. + +The target has no thresholds, baseline or comparison; those belong to the +provenance-gated consumer of #1654. It is not part of the PR CI Gate: +`cargo test` and nextest do not run bench targets by default. +`tests/tck_scenario_bench.rs` covers it there, running the benchmark in +subprocesses: a passing scenario yields a keyed raw result, a failing step +aborts without one, and test mode writes none. + Validation: ```bash