diff --git a/.github/workflows/gq-logic-tests.yml b/.github/workflows/gq-logic-tests.yml index a9473c85c..e42877fc1 100644 --- a/.github/workflows/gq-logic-tests.yml +++ b/.github/workflows/gq-logic-tests.yml @@ -144,4 +144,5 @@ jobs: - name: Run GQ logic tests if: needs.classify_changes.outputs.run_full_ci == 'true' - run: cargo test -p omnigraph-engine --test gq_logic_tests --locked -- --nocapture + # The corpus crate: format self-tests, corpus layout check, one test per case. + run: cargo test -p omnigraph-gqt --locked -- --nocapture diff --git a/AGENTS.md b/AGENTS.md index e1064e16d..535451ce3 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -34,7 +34,8 @@ Tools that support `@` imports include these automatically: - Storage substrate: Lance 11.0.0 - Workspace: compiler, storage, engine (`omnigraph-engine` package), policy, API types, cluster, CLI, server, Azure admission wrapper, benchmark harness, - and `omnigraph-dst` (deterministic simulation testing; needs + `omnigraph-gqt` (the `.gqt` logic-test corpus and its runner; one libtest + test per case), and `omnigraph-dst` (deterministic simulation testing; needs `--cfg tokio_unstable`, set by its crate-local `.cargo/config.toml` when cargo runs from the crate dir — compiles empty without it) - License: MIT @@ -152,7 +153,7 @@ Set `OMNIGRAPH_UPDATE_OPENAPI=1` only when the drift is intentional. - For a bug, reproduce the predicted failure at the tier the regression rule below names, then fix the root cause and prove the regression turns green. - Query-behavior tests default to `.gqt` logic tests under - `crates/omnigraph/tests/gq_logic_tests/`; a Rust test needs a reason the + `crates/omnigraph-gqt/cases/`; a Rust test needs a reason the logic test format cannot express (mechanism assertions, scale symptoms, process environment, concurrency). - Every issue fix lands a regression test at the cheapest tier that catches diff --git a/Cargo.lock b/Cargo.lock index 15b6bffbf..56a2f2e8c 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1233,6 +1233,12 @@ dependencies = [ "either", ] +[[package]] +name = "camino" +version = "1.2.5" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "bb1307f12aa967b5a58416e87b3653360e0fd614a016b6e970db08fecbb1b80d" + [[package]] name = "cbc" version = "0.1.2" @@ -1474,7 +1480,7 @@ version = "3.1.1" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "faf9468729b8cbcea668e36183cb69d317348c2e08e994829fb56ebfdfbaac34" dependencies = [ - "windows-sys 0.52.0", + "windows-sys 0.61.2", ] [[package]] @@ -2437,6 +2443,18 @@ dependencies = [ "sqlparser", ] +[[package]] +name = "datatest-stable" +version = "0.3.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "a867d7322eb69cf3a68a5426387a25b45cb3b9c5ee41023ee6cea92e2afadd82" +dependencies = [ + "camino", + "fancy-regex", + "libtest-mimic", + "walkdir", +] + [[package]] name = "defmt" version = "1.1.1" @@ -2656,9 +2674,15 @@ source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "39cab71617ae0d63f51a36d69f866391735b51691dbda63cf6f96d042b63efeb" dependencies = [ "libc", - "windows-sys 0.52.0", + "windows-sys 0.61.2", ] +[[package]] +name = "escape8259" +version = "0.5.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "5692dd7b5a1978a5aeb0ce83b7655c58ca8efdcb79d21036ea249da95afec2c6" + [[package]] name = "ethnum" version = "1.5.3" @@ -2708,6 +2732,17 @@ dependencies = [ "tokio", ] +[[package]] +name = "fancy-regex" +version = "0.14.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6e24cb5a94bcae1e5408b0effca5cd7172ea3c5755049c5f3af4cd283a165298" +dependencies = [ + "bit-set", + "regex-automata", + "regex-syntax", +] + [[package]] name = "fast-float2" version = "0.2.3" @@ -3384,7 +3419,7 @@ dependencies = [ "libc", "percent-encoding", "pin-project-lite", - "socket2 0.5.10", + "socket2 0.6.4", "system-configuration", "tokio", "tower-service", @@ -4502,6 +4537,18 @@ dependencies = [ "rayon", ] +[[package]] +name = "libtest-mimic" +version = "0.8.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "14e6ba06f0ade6e504aff834d7c34298e5155c6baca353cc6a4aaff2f9fd7f33" +dependencies = [ + "anstream", + "anstyle", + "clap", + "escape8259", +] + [[package]] name = "link-section" version = "0.18.3" @@ -5249,6 +5296,19 @@ dependencies = [ "url", ] +[[package]] +name = "omnigraph-gqt" +version = "0.10.0" +dependencies = [ + "datatest-stable", + "futures", + "omnigraph-compiler", + "omnigraph-engine", + "serde_json", + "tempfile", + "tokio", +] + [[package]] name = "omnigraph-policy" version = "0.10.0" @@ -6130,7 +6190,7 @@ dependencies = [ "quinn-udp", "rustc-hash", "rustls 0.23.41", - "socket2 0.5.10", + "socket2 0.6.4", "thiserror", "tokio", "tracing", @@ -6169,9 +6229,9 @@ dependencies = [ "cfg_aliases", "libc", "once_cell", - "socket2 0.5.10", + "socket2 0.6.4", "tracing", - "windows-sys 0.52.0", + "windows-sys 0.61.2", ] [[package]] @@ -6742,7 +6802,7 @@ dependencies = [ "errno", "libc", "linux-raw-sys", - "windows-sys 0.52.0", + "windows-sys 0.61.2", ] [[package]] @@ -6812,7 +6872,7 @@ dependencies = [ "security-framework", "security-framework-sys", "webpki-root-certs", - "windows-sys 0.52.0", + "windows-sys 0.61.2", ] [[package]] @@ -7567,7 +7627,7 @@ dependencies = [ "getrandom 0.4.3", "once_cell", "rustix", - "windows-sys 0.52.0", + "windows-sys 0.61.2", ] [[package]] @@ -8395,7 +8455,7 @@ version = "0.1.11" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "c2a7b1c03c876122aa43f3020e6c3c3ee5c05081c9a00739faf7503aeba10d22" dependencies = [ - "windows-sys 0.52.0", + "windows-sys 0.61.2", ] [[package]] diff --git a/Cargo.toml b/Cargo.toml index 227a92347..ef932ec1e 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -12,6 +12,7 @@ members = [ "crates/omnigraph-policy", "crates/omnigraph-server", "crates/omnigraph-dst", + "crates/omnigraph-gqt", "tools/omnigraph-vocabulary-guard", ] default-members = [ diff --git a/crates/omnigraph-gqt/Cargo.toml b/crates/omnigraph-gqt/Cargo.toml new file mode 100644 index 000000000..bdd57d24c --- /dev/null +++ b/crates/omnigraph-gqt/Cargo.toml @@ -0,0 +1,32 @@ +[package] +name = "omnigraph-gqt" +version = "0.10.0" +edition = "2024" +description = "GQ logic-test corpus (`cases/*.gqt`) and its runner; one named test per case." +license = "MIT" +repository = "https://github.com/ModernRelay/omnigraph" +homepage = "https://github.com/ModernRelay/omnigraph" +publish = false + +# No doc examples; skip the empty doctest target on every `cargo test -p`. +[lib] +doctest = false + +[dependencies] +futures = { workspace = true } +omnigraph = { package = "omnigraph-engine", path = "../omnigraph", version = "0.10.0" } +omnigraph-compiler = { path = "../omnigraph-compiler", version = "0.10.0" } +serde_json = { workspace = true } +tempfile = { workspace = true } +tokio = { workspace = true } + +[dev-dependencies] +datatest-stable = "0.3" + +# One libtest test per `cases/*.gqt`; mechanism and flags: tests/gq_logic_tests.rs. +[[test]] +name = "gq_logic_tests" +harness = false + +[lints] +workspace = true diff --git a/crates/omnigraph-gqt/README.md b/crates/omnigraph-gqt/README.md new file mode 100644 index 000000000..abc6a1c83 --- /dev/null +++ b/crates/omnigraph-gqt/README.md @@ -0,0 +1,31 @@ +# omnigraph-gqt + +The GQ logic-test corpus and its runner. Never part of a release build: +`publish = false`, not a workspace default member, not in `release.yml`. A +bare `cargo test` at the workspace root therefore skips it; `-p omnigraph-gqt` +or `--workspace` reaches it. + +- `cases/*.gqt`: the corpus. One file is one case: a `.pg` schema, JSONL + seed rows, and steps (queries, mutations, restarts, loops) with expected + outcomes. Format, refusal set, and comparison semantics: RFC 0045 + (`docs/rfcs/0045-gq-logic-tests.md`). A regression for a merged fix is + `issue_NNN_.gqt`; `scripts/check-fix-regression.py` looks for it + here. +- `src/lib.rs`: the runner (case parsing, execution against a fresh + temporary store, row comparison, bless). Format self-tests and the corpus + layout check are its unit tests (`src/tests.rs`). +- `tests/gq_logic_tests.rs`: one libtest test per case, named + `case::.gqt`, registered at run time by `datatest-stable` + (`harness = false`). A new case file is picked up without any Rust change. + +```bash +cargo test -p omnigraph-gqt # everything +cargo test -p omnigraph-gqt --test gq_logic_tests issue_563 # cases whose name contains issue_563 +cargo test -p omnigraph-gqt --test gq_logic_tests -- --list # one line per case +cargo test -p omnigraph-gqt --test gq_logic_tests -- --test-threads=2 +OMNIGRAPH_GQ_BLESS=1 cargo test -p omnigraph-gqt --test gq_logic_tests my_case # rewrite the failing expect +``` + +`OMNIGRAPH_GQ_CASE_TIMEOUT_SECS=` (default 10) bounds each case's wall +time; a case over budget belongs in a `heavy-repro:` `#[ignore]`d Rust test, +not here. diff --git a/crates/omnigraph/tests/gq_logic_tests/issue_563_aggregate_uncapped.gqt b/crates/omnigraph-gqt/cases/issue_563_aggregate_uncapped.gqt similarity index 98% rename from crates/omnigraph/tests/gq_logic_tests/issue_563_aggregate_uncapped.gqt rename to crates/omnigraph-gqt/cases/issue_563_aggregate_uncapped.gqt index 067015e1f..52a150b56 100644 --- a/crates/omnigraph/tests/gq_logic_tests/issue_563_aggregate_uncapped.gqt +++ b/crates/omnigraph-gqt/cases/issue_563_aggregate_uncapped.gqt @@ -2,7 +2,7 @@ # red_on: 2026-08-29, pre-fix build: the bm25 scan cap (limit 2 x 4 = 8 rows) fed count() only the capped window, so total was 8, not 20. # notes: an aggregate's value is computed over the matching rows, so the BM25 # notes: scan cap must never apply to aggregate returns. Rust twin: -# notes: bm25_ordered_aggregate_counts_all_matches_not_the_capped_scan (tests/search.rs). +# notes: bm25_ordered_aggregate_counts_all_matches_not_the_capped_scan (crates/omnigraph/tests/search.rs). # notes: Seed geometry: 20 chunks all matching "needle", strict tf gradient # notes: (tf = 20 - c) so BM25 has a real order with no ties. diff --git a/crates/omnigraph/tests/gq_logic_tests/issue_563_underfill_retry.gqt b/crates/omnigraph-gqt/cases/issue_563_underfill_retry.gqt similarity index 99% rename from crates/omnigraph/tests/gq_logic_tests/issue_563_underfill_retry.gqt rename to crates/omnigraph-gqt/cases/issue_563_underfill_retry.gqt index 6626475b3..e9fc81896 100644 --- a/crates/omnigraph/tests/gq_logic_tests/issue_563_underfill_retry.gqt +++ b/crates/omnigraph-gqt/cases/issue_563_underfill_retry.gqt @@ -5,7 +5,7 @@ # notes: the uncapped retry ran. Passes on pre-cap code too (an unbounded scan # notes: finds everything): this case guards the retry GIVEN the cap, not the cap # notes: itself; the cap is the aggregate case's job. Rust twin: -# notes: bm25_join_fills_limit_when_capped_scan_underfills_issue_563 (tests/search.rs). +# notes: bm25_join_fills_limit_when_capped_scan_underfills_issue_563 (crates/omnigraph/tests/search.rs). --- schema node Chunk { diff --git a/crates/omnigraph/tests/gq_logic_tests/issue_603_negation_string_predicate_outer_var.gqt b/crates/omnigraph-gqt/cases/issue_603_negation_string_predicate_outer_var.gqt similarity index 100% rename from crates/omnigraph/tests/gq_logic_tests/issue_603_negation_string_predicate_outer_var.gqt rename to crates/omnigraph-gqt/cases/issue_603_negation_string_predicate_outer_var.gqt diff --git a/crates/omnigraph/tests/gq_logic_tests/mutation_error_typed.gqt b/crates/omnigraph-gqt/cases/mutation_error_typed.gqt similarity index 100% rename from crates/omnigraph/tests/gq_logic_tests/mutation_error_typed.gqt rename to crates/omnigraph-gqt/cases/mutation_error_typed.gqt diff --git a/crates/omnigraph/tests/gq_logic_tests/order_clause_aggregate_refused.gqt b/crates/omnigraph-gqt/cases/order_clause_aggregate_refused.gqt similarity index 100% rename from crates/omnigraph/tests/gq_logic_tests/order_clause_aggregate_refused.gqt rename to crates/omnigraph-gqt/cases/order_clause_aggregate_refused.gqt diff --git a/crates/omnigraph/tests/gq_logic_tests/ordered_two_key_sort.gqt b/crates/omnigraph-gqt/cases/ordered_two_key_sort.gqt similarity index 100% rename from crates/omnigraph/tests/gq_logic_tests/ordered_two_key_sort.gqt rename to crates/omnigraph-gqt/cases/ordered_two_key_sort.gqt diff --git a/crates/omnigraph/tests/gq_logic_tests/restart_survives_reopen.gqt b/crates/omnigraph-gqt/cases/restart_survives_reopen.gqt similarity index 100% rename from crates/omnigraph/tests/gq_logic_tests/restart_survives_reopen.gqt rename to crates/omnigraph-gqt/cases/restart_survives_reopen.gqt diff --git a/crates/omnigraph/tests/gq_logic_tests.rs b/crates/omnigraph-gqt/src/lib.rs similarity index 54% rename from crates/omnigraph/tests/gq_logic_tests.rs rename to crates/omnigraph-gqt/src/lib.rs index 2227d7049..ef2b355e9 100644 --- a/crates/omnigraph/tests/gq_logic_tests.rs +++ b/crates/omnigraph-gqt/src/lib.rs @@ -1,15 +1,15 @@ -//! GQ logic tests: walks `tests/gq_logic_tests/*.gqt` and runs each case -//! against a fresh temporary store (init, load, index, then the steps in -//! order). The file format, refusal set, comparison semantics, and bless -//! workflow are specified in `docs/rfcs/0045-gq-logic-tests.md`. +//! GQ logic tests: the `.gqt` corpus under `cases/` and the runner that +//! executes one case against a fresh temporary store (init, load, index, +//! then the steps in order). The file format, refusal set, comparison +//! semantics, and bless workflow are specified in +//! `docs/rfcs/0045-gq-logic-tests.md`. //! -//! To libtest the whole walker is one test; case concurrency comes from a -//! `JoinSet` bounded by a semaphore (`OMNIGRAPH_GQ_JOBS=` overrides the -//! default of the machine's available parallelism), and every case runs under -//! a per-case budget (`OMNIGRAPH_GQ_CASE_TIMEOUT_SECS=`, default 10). -//! `OMNIGRAPH_GQ_LOGIC_TESTS=[,]` restricts the run to -//! matching case files; `OMNIGRAPH_GQ_BLESS=1` rewrites the failing step's -//! `--- expect` rows in place. +//! The test target `tests/gq_logic_tests.rs` registers every case file as +//! its own test (`datatest-stable`); how cases are selected, listed, and +//! run concurrently is documented there. Every case runs under a per-case +//! wall-time budget (`OMNIGRAPH_GQ_CASE_TIMEOUT_SECS=`, default 10) via +//! [`run_case_bounded`]. `OMNIGRAPH_GQ_BLESS=1` rewrites the failing +//! step's `--- expect` rows in place. //! //! Layout of the `run_query_step` future (an engine query under the traversal //! task-local, the timeout, and `catch_unwind`) exceeds the default @@ -23,9 +23,8 @@ use std::fmt::Write as _; use std::future::Future; use std::panic::AssertUnwindSafe; use std::path::{Path, PathBuf}; -use std::pin::Pin; use std::sync::Arc; -use std::sync::atomic::{AtomicU64, AtomicUsize, Ordering}; +use std::sync::atomic::{AtomicU64, Ordering}; use std::time::{Duration, Instant}; use futures::FutureExt as _; @@ -38,12 +37,10 @@ use omnigraph_compiler::schema::ast::{Annotation, PropDecl, SchemaDecl}; use omnigraph_compiler::schema::parser::parse_schema; use omnigraph_compiler::{JsonParamMode, json_params_to_param_map}; use serde_json::Value; -use tokio::sync::Semaphore; -use tokio::task::JoinSet; -const CASE_TIMEOUT_ENV: &str = "OMNIGRAPH_GQ_CASE_TIMEOUT_SECS"; -const DEFAULT_CASE_TIMEOUT_SECS: u64 = 10; -const JOBS_ENV: &str = "OMNIGRAPH_GQ_JOBS"; +pub const CASE_TIMEOUT_ENV: &str = "OMNIGRAPH_GQ_CASE_TIMEOUT_SECS"; +pub const DEFAULT_CASE_TIMEOUT_SECS: u64 = 10; +pub const BLESS_ENV: &str = "OMNIGRAPH_GQ_BLESS"; #[derive(Debug)] struct Case { @@ -1508,40 +1505,68 @@ fn bless_rewrite(path: &Path, span: BodySpan, rows: &[String]) -> Result<(), Str .map_err(|e| format!("bless: cannot write case: {e}")) } -async fn run_case(path: PathBuf, bless: bool) -> Result<(), String> { - let stem = path - .file_stem() - .and_then(|s| s.to_str()) - .ok_or_else(|| "case file name is not utf-8".to_string())? - .to_string(); +/// Parses and executes one case file; `bless` rewrites a failing step's +/// `--- expect` rows in place and still reports the step (`expect +/// rewritten`), so the run stays red until the re-run confirms. +pub async fn run_case(path: PathBuf, bless: bool) -> Result<(), String> { + let stem = stem_of(&path); let text = std::fs::read_to_string(&path).map_err(|e| format!("cannot read case file: {e}"))?; let case = parse_case(&stem, &text).map_err(|e| format!("refused: {e}"))?; execute_case(&case, &path, bless).await } -fn corpus_root() -> PathBuf { - let manifest_dir = std::env::var("CARGO_MANIFEST_DIR").expect("CARGO_MANIFEST_DIR"); - PathBuf::from(manifest_dir) - .join("tests") - .join("gq_logic_tests") +/// The corpus directory, `cases/` beside this crate's manifest: the same +/// compile-time root the test target's `datatest_stable::harness!` resolves +/// `root = "cases"` against. +pub fn corpus_root() -> PathBuf { + PathBuf::from(env!("CARGO_MANIFEST_DIR")).join("cases") +} + +/// `OMNIGRAPH_GQ_BLESS=1` turns bless on; unset, empty, or `0` leaves it off. +/// +/// # Panics +/// +/// On any other value: the knob is refused, not ignored. +pub fn bless_from_env() -> bool { + match std::env::var(BLESS_ENV) { + Err(_) => false, + Ok(v) if v == "1" => true, + Ok(v) if v == "0" || v.is_empty() => false, + Ok(v) => panic!("{BLESS_ENV} takes 1 (or 0/unset), got `{v}`"), + } +} + +/// The per-case wall-time budget: `OMNIGRAPH_GQ_CASE_TIMEOUT_SECS` or the +/// default. +pub fn case_budget_from_env() -> Duration { + Duration::from_secs(env_positive(CASE_TIMEOUT_ENV).unwrap_or(DEFAULT_CASE_TIMEOUT_SECS)) } /// Splits the corpus dir into `.gqt` case files and foreign entries; a -/// foreign entry is a mis-renamed, nested, or dot-prefixed case that would -/// otherwise silently never run. A case is a top-level file whose name ends -/// in `.gqt` and does not start with `.`; `scripts/check-fix-regression.py` -/// (`corpus_case`) applies the same rule, and both self-tests walk one name -/// battery. Dot-prefixed entries without a `.gqt` extension (`.DS_Store`, -/// `.gitkeep`, and a file named exactly `.gqt`, which has no extension) -/// are neither cases nor foreign: they are skipped. -fn list_cases(root: &Path) -> (Vec, Vec) { +/// foreign entry is a mis-renamed, nested, symlinked, dot-prefixed, or +/// non-UTF-8-named case that would otherwise silently never run. The rule +/// that RUNS a case is the test target's `datatest_stable::harness!` pattern +/// (`tests/gq_logic_tests.rs`): a regular file (symlinks are not followed) +/// with a UTF-8 name that ends in `.gqt` and does not start with `.`. This +/// function mirrors that rule so `corpus_layout` refuses what the target +/// would skip; `scripts/check-fix-regression.py` (`corpus_case`) mirrors +/// the name half, and both self-tests walk one name battery. Dot-prefixed +/// entries without a `.gqt` extension (`.DS_Store`, `.gitkeep`, and a file +/// named exactly `.gqt`, which has no extension) are neither cases nor +/// foreign: they are skipped, as the target skips every hidden file. +pub fn list_cases(root: &Path) -> (Vec, Vec) { let mut files = Vec::new(); let mut foreign = Vec::new(); if let Ok(entries) = std::fs::read_dir(root) { for entry in entries.flatten() { let path = entry.path(); - let name = entry.file_name().to_string_lossy().to_string(); - let is_gqt = path.is_file() && path.extension().and_then(|s| s.to_str()) == Some("gqt"); + let Some(name) = entry.file_name().to_str().map(str::to_owned) else { + foreign.push(entry.file_name().to_string_lossy().into_owned()); + continue; + }; + let is_regular_file = entry.file_type().map(|t| t.is_file()).unwrap_or(false); + let is_gqt = + is_regular_file && path.extension().and_then(|s| s.to_str()) == Some("gqt"); if name.starts_with('.') && !is_gqt { continue; } @@ -1557,7 +1582,9 @@ fn list_cases(root: &Path) -> (Vec, Vec) { (files, foreign) } -fn stem_of(path: &Path) -> String { +/// The case name: the file stem, or `` for a name the corpus +/// rule refuses anyway. +pub fn stem_of(path: &Path) -> String { path.file_stem() .and_then(|s| s.to_str()) .unwrap_or("") @@ -1565,7 +1592,12 @@ fn stem_of(path: &Path) -> String { } /// A positive-integer environment override; unset or empty means none. -fn env_positive(name: &str) -> Option { +/// +/// # Panics +/// +/// On a value that is not a positive integer: the knob is refused, not +/// ignored. +pub fn env_positive(name: &str) -> Option { let value = std::env::var(name).ok()?; if value.trim().is_empty() { return None; @@ -1578,7 +1610,7 @@ fn env_positive(name: &str) -> Option { /// The production traversal path consults `OMNIGRAPH_TRAVERSAL_MODE`, so a set /// variable would silently decide which path an unpinned case exercises. -fn traversal_override_refusal(value: Option<&OsStr>) -> Option { +pub fn traversal_override_refusal(value: Option<&OsStr>) -> Option { value.map(|v| { format!( "OMNIGRAPH_TRAVERSAL_MODE={} is set; logic tests run the production traversal \ @@ -1588,7 +1620,7 @@ fn traversal_override_refusal(value: Option<&OsStr>) -> Option { }) } -fn panic_message(payload: &(dyn std::any::Any + Send)) -> String { +pub fn panic_message(payload: &(dyn std::any::Any + Send)) -> String { if let Some(s) = payload.downcast_ref::<&str>() { (*s).to_string() } else if let Some(s) = payload.downcast_ref::() { @@ -1598,1171 +1630,52 @@ fn panic_message(payload: &(dyn std::any::Any + Send)) -> String { } } -type CaseFuture = Pin> + Send>>; -type CaseRunner = Arc CaseFuture + Send + Sync>; - +/// What one case produced: its stem, wall time from the moment it started, +/// and the verdict (the error text carries the failing step's diff, the +/// refusal, the panic message, or the budget overrun). #[derive(Debug)] -struct CaseOutcome { - stem: String, - elapsed: Duration, - result: Result<(), String>, -} - -/// Runs every case as its own task with at most `permits` in flight (each -/// case holds a store and may build indexes) and `budget` of wall time per -/// case, timed from the moment it holds a permit; a case over budget is -/// dropped, store included, before its permit is released. A panic or a -/// timeout is an ordinary failed case, so the whole corpus always runs. -/// `report` sees each outcome as it completes; the returned list is sorted -/// by case name. -async fn run_bounded( - cases: Vec<(String, PathBuf)>, - permits: usize, - budget: Duration, - run: CaseRunner, - report: &(dyn Fn(&CaseOutcome) + Sync), -) -> Vec { - let semaphore = Arc::new(Semaphore::new(permits)); - let mut set: JoinSet = JoinSet::new(); - for (stem, path) in cases { - let semaphore = Arc::clone(&semaphore); - let run = Arc::clone(&run); - set.spawn(async move { - let _permit = semaphore - .acquire_owned() - .await - .expect("the case semaphore is never closed"); - let started = Instant::now(); - let case = AssertUnwindSafe(run(path)).catch_unwind(); - let result = match tokio::time::timeout(budget, case).await { - Ok(Ok(result)) => result, - Ok(Err(payload)) => Err(format!( - "case panicked: {}", - panic_message(payload.as_ref()) - )), - Err(_) => Err(format!( - "case exceeded its budget of {:.2}s while up to {permits} cases ran \ - concurrently ({CASE_TIMEOUT_ENV} overrides the default of \ - {DEFAULT_CASE_TIMEOUT_SECS}s, {JOBS_ENV} the concurrency; a case over \ - budget belongs in a `heavy-repro:` `#[ignore]`d test, not the corpus)", - budget.as_secs_f64() - )), - }; - CaseOutcome { - stem, - elapsed: started.elapsed(), - result, - } - }); - } - let mut outcomes = Vec::new(); - while let Some(joined) = set.join_next().await { - // The case itself is caught by `catch_unwind`; the task around it - // holds nothing that can panic. - let outcome = joined.expect("a case task never panics"); - report(&outcome); - outcomes.push(outcome); - } - outcomes.sort_by(|a, b| a.stem.cmp(&b.stem)); - outcomes -} - -#[tokio::test(flavor = "multi_thread")] -async fn gq_logic_tests() { - let root = corpus_root(); - let (mut files, foreign) = list_cases(&root); - assert!( - foreign.is_empty(), - "foreign entries under {}: {}; a mis-renamed, nested, or dot-prefixed case must never silently skip", - root.display(), - foreign.join(", ") - ); - assert!( - !files.is_empty(), - "no .gqt cases found under {}; a broken checkout must never read as green", - root.display() - ); - if let Ok(filter) = std::env::var("OMNIGRAPH_GQ_LOGIC_TESTS") { - let needles: Vec = filter - .split(',') - .map(str::trim) - .filter(|s| !s.is_empty()) - .map(str::to_string) - .collect(); - if !needles.is_empty() { - files.retain(|p| { - p.file_name() - .and_then(|s| s.to_str()) - .is_some_and(|name| needles.iter().any(|n| name.contains(n.as_str()))) - }); - assert!( - !files.is_empty(), - "OMNIGRAPH_GQ_LOGIC_TESTS={filter} matched no cases" - ); - } - } - let bless = match std::env::var("OMNIGRAPH_GQ_BLESS") { - Err(_) => false, - Ok(v) if v == "1" => true, - Ok(v) if v == "0" || v.is_empty() => false, - Ok(v) => panic!("OMNIGRAPH_GQ_BLESS takes 1 (or 0/unset), got `{v}`"), - }; - - if let Some(reason) = - traversal_override_refusal(std::env::var_os("OMNIGRAPH_TRAVERSAL_MODE").as_deref()) - { - panic!("{reason}"); - } - let permits = env_positive(JOBS_ENV) - .map(|n| { - usize::try_from(n) - .unwrap_or(usize::MAX) - .min(Semaphore::MAX_PERMITS) - }) - .unwrap_or_else(|| { - std::thread::available_parallelism() - .map(std::num::NonZeroUsize::get) - .unwrap_or(1) - }); - let budget = - Duration::from_secs(env_positive(CASE_TIMEOUT_ENV).unwrap_or(DEFAULT_CASE_TIMEOUT_SECS)); - - let total = files.len(); - let cases = files - .into_iter() - .map(|path| (stem_of(&path), path)) - .collect(); - let runner: CaseRunner = Arc::new(move |path| Box::pin(run_case(path, bless))); - let report = |outcome: &CaseOutcome| { - let secs = outcome.elapsed.as_secs_f64(); - match &outcome.result { - Ok(()) => println!("ok {} {secs:.2}s", outcome.stem), - Err(_) => println!("FAIL {} {secs:.2}s", outcome.stem), - } - }; - let outcomes = run_bounded(cases, permits, budget, runner, &report).await; - - let mut failures: Vec<(String, String)> = outcomes - .into_iter() - .filter_map(|outcome| { - let CaseOutcome { stem, result, .. } = outcome; - result.err().map(|detail| (stem, detail)) - }) - .collect(); - - if !failures.is_empty() { - failures.sort(); - let mut msg = format!( - "{} of {total} gq logic test cases failed:\n", - failures.len() - ); - for (stem, detail) in &failures { - let _ = write!(msg, "\n{stem}:\n {}\n", detail.replace('\n', "\n ")); - } - panic!("{msg}"); - } -} - -const HDR: &str = "# issue: none\n"; -const SCHEMA: &str = "--- schema\nnode Person {\n name: String @key\n}\n"; -const SEED: &str = "--- seed\n{\"type\":\"Person\",\"data\":{\"name\":\"alice\"}}\n"; -const QUERY: &str = - "--- query\nquery all() {\n match { $p: Person }\n return { $p.name }\n}\n"; -const EXPECT: &str = "--- expect unordered\n{\"p.name\": \"alice\"}\n"; -const MUTATE: &str = "--- mutate\nquery ins($n: String) {\n insert Person { name: $n }\n}\n"; -const PARAMS: &str = "--- params\n{\"n\": \"bob\"}\n"; -const EXPECT_OK: &str = "--- expect ok\n"; - -fn refusal(stem: &str, text: &str) -> String { - parse_case(stem, text).expect_err("expected the case to be refused") -} - -#[test] -fn parses_a_minimal_case() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}"); - let case = parse_case("minimal", &text).unwrap(); - assert_eq!(case.items.len(), 1); - assert!(!case.needs_indices); - assert_eq!(case.traversal, None); -} - -#[test] -fn header_notes_repeat_and_continuation_lines_are_refused() { - let text = format!( - "# issue: 7\n# red_on: 2026-01-01, the run\n# notes: returned 8,\n# notes: not 20.\n{SCHEMA}{SEED}{QUERY}{EXPECT}" - ); - parse_case("issue_7_notes", &text).unwrap(); - let text = format!( - "# issue: 7\n# red_on: 2026-01-01, the run\n# returned 8: not 20.\n{SCHEMA}{SEED}{QUERY}{EXPECT}" - ); - assert!(refusal("issue_7_x", &text).contains("unknown header key")); - // The three misspellings that the old prose branch swallowed silently. - for typo in [ - "# Traversal: indexed", - "# traversal : indexed", - "# traversal=csr", - ] { - let text = format!("{HDR}{typo}\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - let reason = refusal("x", &text); - assert!( - reason.contains("unknown header key") || reason.contains("not `# : `"), - "{typo}: {reason}" - ); - } -} - -/// Bounded exhaustive walk of the header-line typo space: key spelling, -/// separator, leading and trailing whitespace, and the gap before the value. -/// A line is accepted exactly when it equals the canonical -/// `# traversal: indexed`, so a future key inherits the same proof. -#[test] -fn header_lines_are_accepted_only_in_canonical_form() { - let keys = [ - "traversal", - "Traversal", - "TRAVERSAL", - "traversa1", - "traversal_", - " traversal", - ]; - let seps = [":", " :", "=", "", "::"]; - let leads = ["# ", "#", "# ", " # "]; - let gaps = [" ", "", " ", "\t"]; - let trails = ["", " ", "\t"]; - let canonical = canonical_header_line("traversal", "indexed"); - // Distinct lines: two lead/key pairs collide (`#` + ` traversal` = `# ` + - // `traversal`, and `# ` + ` traversal` = `# ` + `traversal`), 120 - // duplicates over the 1440 grid points, so the set is what gets walked. - let mut lines = std::collections::BTreeSet::new(); - for lead in leads { - for key in keys { - for sep in seps { - for gap in gaps { - for trail in trails { - lines.insert(format!("{lead}{key}{sep}{gap}indexed{trail}")); - } - } - } - } - } - assert_eq!(lines.len(), 1320, "the typo space is 1320 distinct lines"); - let mut accepted = 0; - for line in &lines { - let text = format!("{HDR}{line}\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - match parse_case("x", &text) { - Ok(case) => { - assert_eq!(line, &canonical, "accepted a non-canonical line"); - assert_eq!(case.traversal, Some("indexed")); - accepted += 1; - } - Err(e) => assert_ne!(line, &canonical, "refused the canonical line: {e}"), - } - } - assert_eq!(accepted, 1); -} - -#[test] -fn refuses_missing_issue_header() { - let text = format!("# notes: no anchor\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("# issue:")); -} - -#[test] -fn refuses_numbered_issue_without_red_on() { - let text = format!("# issue: 7\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("issue_7_x", &text).contains("red_on")); -} - -#[test] -fn refuses_header_line_without_a_key() { - let text = format!("# stray prose\n{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("not `# : `")); -} - -#[test] -fn refuses_unknown_header_key() { - let text = format!("{HDR}# owner: me\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("unknown header key")); -} - -#[test] -fn refuses_bad_traversal_mode() { - let text = format!("{HDR}# traversal: bogus\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("indexed")); -} - -#[test] -fn refuses_non_header_line_before_first_section() { - let text = format!("{HDR}stray\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("precede the first section")); -} - -#[test] -fn refuses_comment_line_in_seed() { - let text = format!("{HDR}{SCHEMA}--- seed\n# a comment\n{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("seed")); -} - -#[test] -fn refuses_comment_line_in_expect_body() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect unordered\n# nope\n"); - assert!(refusal("x", &text).contains("expect")); -} - -#[test] -fn refuses_seed_before_schema() { - let text = format!("{HDR}{SEED}{SCHEMA}{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("first section")); -} - -#[test] -fn refuses_missing_seed_section() { - let text = format!("{HDR}{SCHEMA}{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("second section")); -} - -#[test] -fn refuses_case_without_a_query_or_mutate_step() { - let text = format!("{HDR}{SCHEMA}{SEED}"); - assert!(refusal("x", &text).contains("at least one query or mutate step")); -} - -#[test] -fn refuses_restart_only_step_list() { - let text = format!("{HDR}{SCHEMA}{SEED}--- restart\n"); - assert!(refusal("x", &text).contains("at least one query or mutate step")); -} - -#[test] -fn refuses_second_declaration_in_one_section() { - let two = "--- query\nquery a() {\n match { $p: Person }\n return { $p.name }\n}\nquery b() {\n match { $p: Person }\n return { $p.name }\n}\n"; - let text = format!("{HDR}{SCHEMA}{SEED}{two}{EXPECT}"); - assert!(refusal("x", &text).contains("exactly one declaration")); -} - -#[test] -fn refuses_mutation_declaration_under_query() { - let text = format!( - "{HDR}{SCHEMA}{SEED}--- query\nquery ins($n: String) {{\n insert Person {{ name: $n }}\n}}\n{EXPECT}" - ); - assert!(refusal("x", &text).contains("use `--- mutate`")); -} - -#[test] -fn refuses_read_declaration_under_mutate() { - let text = format!( - "{HDR}{SCHEMA}{SEED}--- mutate\nquery all() {{\n match {{ $p: Person }}\n return {{ $p.name }}\n}}\n{EXPECT_OK}" - ); - assert!(refusal("x", &text).contains("use `--- query`")); -} - -#[test] -fn refuses_bare_expect() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect\n"); - assert!(refusal("x", &text).contains("mode word")); -} - -#[test] -fn refuses_error_expect_without_substring() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect error:\n"); - assert!(refusal("x", &text).contains("substring")); -} - -#[test] -fn refuses_error_expect_with_body() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect error: boom\nbody\n"); - assert!(refusal("x", &text).contains("carries no body")); -} - -#[test] -fn refuses_affected_expect_missing_a_count() { - let text = format!("{HDR}{SCHEMA}{SEED}{MUTATE}{PARAMS}--- expect affected: nodes=1\n"); - assert!(refusal("x", &text).contains("nodes= edges=")); -} - -#[test] -fn refuses_unknown_expect_mode() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect sorted\n"); - assert!(refusal("x", &text).contains("unknown expect mode")); -} - -#[test] -fn refuses_row_expect_on_a_mutate_step() { - let text = format!("{HDR}{SCHEMA}{SEED}{MUTATE}{PARAMS}{EXPECT}"); - assert!(refusal("x", &text).contains("carry no rows")); -} - -#[test] -fn refuses_ok_expect_on_a_query_step() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT_OK}"); - assert!(refusal("x", &text).contains("a query step takes")); -} - -#[test] -fn refuses_query_step_without_expect() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}"); - assert!(refusal("x", &text).contains("missing its `--- expect`")); -} - -#[test] -fn refuses_expect_with_no_step_to_bind_to() { - let text = format!("{HDR}{SCHEMA}{SEED}--- restart\n{EXPECT}{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("no query or mutate step to bind to")); -} - -#[test] -fn refuses_params_without_a_step() { - let text = format!("{HDR}{SCHEMA}{SEED}{PARAMS}{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("directly follow")); -} - -#[test] -fn refuses_second_params_for_one_step() { - let text = format!("{HDR}{SCHEMA}{SEED}{MUTATE}{PARAMS}{PARAMS}{EXPECT_OK}"); - assert!(refusal("x", &text).contains("second `--- params`")); -} - -#[test] -fn refuses_restart_with_a_body() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}--- restart\nstray\n"); - assert!(refusal("x", &text).contains("carries no body")); -} - -#[test] -fn refuses_unknown_section() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}--- teardown\n"); - assert!(refusal("x", &text).contains("unknown section")); -} - -#[test] -fn refuses_schema_out_of_position() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}{SCHEMA}"); - assert!(refusal("x", &text).contains("out of position")); -} - -#[test] -fn refuses_negative_loop_bound() { - let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i -1 2\n{QUERY}{EXPECT}--- endloop\n"); - assert!(refusal("x", &text).contains("non-negative")); -} - -#[test] -fn refuses_empty_loop_range() { - let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 3 3\n{QUERY}{EXPECT}--- endloop\n"); - assert!(refusal("x", &text).contains("empty loop range")); -} - -#[test] -fn refuses_foreach_without_values() { - let text = format!("{HDR}{SCHEMA}{SEED}--- foreach $x\n{QUERY}{EXPECT}--- endloop\n"); - assert!(refusal("x", &text).contains("no values")); -} - -#[test] -fn refuses_foreach_value_outside_charset() { - let text = format!("{HDR}{SCHEMA}{SEED}--- foreach $x a\"b\n{QUERY}{EXPECT}--- endloop\n"); - assert!(refusal("x", &text).contains("[A-Za-z0-9_.-]")); -} - -#[test] -fn refuses_bad_loop_variable_name() { - let text = format!("{HDR}{SCHEMA}{SEED}--- loop $I 0 2\n{QUERY}{EXPECT}--- endloop\n"); - assert!(refusal("x", &text).contains("$[a-z][a-z0-9_]*")); -} - -#[test] -fn refuses_nested_loops() { - let text = format!( - "{HDR}{SCHEMA}{SEED}--- loop $i 0 2\n--- loop $j 0 2\n{QUERY}{EXPECT}--- endloop\n--- endloop\n" - ); - assert!(refusal("x", &text).contains("may not nest")); -} - -#[test] -fn refuses_endloop_without_a_loop() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}--- endloop\n"); - assert!(refusal("x", &text).contains("without an open loop")); -} - -#[test] -fn refuses_unclosed_loop() { - let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 0 2\n{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("not closed")); -} - -#[test] -fn refuses_loop_enclosing_no_steps() { - let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 0 2\n--- endloop\n{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("enclosing no steps")); -} - -#[test] -fn refuses_substitution_marker_in_query_body() { - let query = "--- query\nquery all() {\n match { $p: Person }\n return { $p.name }\n}\n"; - let text = - format!("{HDR}{SCHEMA}{SEED}{query}{EXPECT}").replace("$p.name }", "$p.name } // ${i}"); - assert!(refusal("x", &text).contains("only inside a params or expect body")); -} - -#[test] -fn refuses_substitution_marker_in_seed() { - let text = format!( - "{HDR}{SCHEMA}--- seed\n{{\"type\":\"Person\",\"data\":{{\"name\":\"${{i}}\"}}}}\n{QUERY}{EXPECT}" - ); - assert!(refusal("x", &text).contains("only inside a params or expect body")); -} - -#[test] -fn refuses_substitution_outside_a_loop() { - let text = - format!("{HDR}{SCHEMA}{SEED}{MUTATE}--- params\n{{\"n\": \"${{who}}\"}}\n{EXPECT_OK}"); - assert!(refusal("x", &text).contains("outside a loop")); -} - -#[test] -fn refuses_substitution_naming_the_wrong_variable() { - let text = format!( - "{HDR}{SCHEMA}{SEED}--- foreach $who bob\n{MUTATE}--- params\n{{\"n\": \"${{other}}\"}}\n{EXPECT_OK}--- endloop\n" - ); - assert!(refusal("x", &text).contains("enclosing loop's variable")); -} - -#[test] -fn refuses_unterminated_substitution() { - let text = format!( - "{HDR}{SCHEMA}{SEED}--- foreach $who bob\n{MUTATE}--- params\n{{\"n\": \"${{who\n{EXPECT_OK}--- endloop\n" - ); - assert!(refusal("x", &text).contains("unterminated")); -} - -#[test] -fn refuses_file_name_disagreeing_with_issue_header() { - let text = format!("# issue: 7\n# red_on: 2026-01-01, red.\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("issue_8_wrong", &text).contains("disagrees")); -} - -#[test] -fn refuses_issue_prefix_without_number_or_short_name() { - let text = format!("# issue: 7\n# red_on: 2026-01-01, red.\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("issue_7", &text).contains("issue__")); - assert!(refusal("issue_x", &text).contains("issue__")); -} - -#[test] -fn refuses_feature_name_with_numbered_issue_header() { - let text = format!("# issue: 7\n# red_on: 2026-01-01, red.\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("feature_name", &text).contains("issue_7_")); -} - -#[test] -fn refuses_file_name_outside_charset() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("Bad-Name", &text).contains("[a-z0-9_]")); -} - -#[test] -fn refuses_ordered_expect_without_an_order_clause() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect ordered\n{{\"p.name\": \"alice\"}}\n"); - assert!(refusal("x", &text).contains("order` clause")); -} - -#[test] -fn refuses_embed_schema() { - let schema = "--- schema\nnode Doc {\n slug: String @key\n text: String\n vec: Vector(4) @embed(\"text\")\n}\n"; - let seed = "--- seed\n"; - let text = format!("{HDR}{schema}{seed}{QUERY}{EXPECT}") - .replace("$p: Person", "$p: Doc") - .replace("$p.name", "$p.slug"); - assert!(refusal("x", &text).contains("@embed")); -} - -#[test] -fn refuses_nearest_over_a_string_literal() { - let query = "--- query\nquery q() {\n match { $p: Person }\n return { $p.name }\n order { nearest($p.name, \"alpha\") }\n}\n"; - let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect unordered\n"); - assert!(refusal("x", &text).contains("string argument")); -} - -#[test] -fn refuses_nearest_over_a_string_param() { - let query = "--- query\nquery q($q: String) {\n match { $p: Person }\n return { $p.name }\n order { nearest($p.name, $q) }\n}\n"; - let text = format!( - "{HDR}{SCHEMA}{SEED}{query}--- params\n{{\"q\": \"alpha\"}}\n--- expect unordered\n" - ); - assert!(refusal("x", &text).contains("string argument")); -} - -#[test] -fn accepts_empty_expect_body_as_empty_result_assertion() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect unordered\n"); - parse_case("x", &text).unwrap(); -} - -#[test] -fn search_construct_sets_the_index_decision() { - let schema = "--- schema\nnode Doc {\n slug: String @key\n text: String @index\n}\n"; - let query = "--- query\nquery q($q: String) {\n match {\n $d: Doc\n search($d.text, $q)\n }\n return { $d.slug }\n}\n"; - let text = format!( - "{HDR}{schema}--- seed\n{query}--- params\n{{\"q\": \"needle\"}}\n--- expect unordered\n" - ); - let case = parse_case("x", &text).unwrap(); - assert!(case.needs_indices); -} - -#[test] -fn normalization_equates_integer_and_float_spellings() { - let a: Value = serde_json::from_str("{\"total\": 2}").unwrap(); - let b: Value = serde_json::from_str("{\"total\": 2.0}").unwrap(); - assert_eq!(canonical_json(&a), canonical_json(&b)); -} - -#[test] -fn normalization_does_not_collapse_large_integers() { - let a: Value = serde_json::from_str("{\"n\": 9007199254740993}").unwrap(); - let b: Value = serde_json::from_str("{\"n\": 9007199254740992}").unwrap(); - assert_ne!(canonical_json(&a), canonical_json(&b)); -} - -#[test] -fn normalization_ignores_noise_below_scale_12() { - let a: Value = serde_json::from_str("{\"x\": 0.1000000000000001}").unwrap(); - let b: Value = serde_json::from_str("{\"x\": 0.1}").unwrap(); - assert_eq!(canonical_json(&a), canonical_json(&b)); -} - -#[test] -fn canonical_form_sorts_object_keys_and_recurses() { - let a: Value = serde_json::from_str("{\"b\": [{\"z\": 1, \"a\": 2}], \"a\": null}").unwrap(); - assert_eq!(canonical_json(&a), "{\"a\":null,\"b\":[{\"a\":2,\"z\":1}]}"); -} - -#[test] -fn unordered_comparison_is_multiset_equality() { - let rows = |s: &str| -> Vec { - s.lines() - .map(|l| serde_json::from_str(l).unwrap()) - .collect() +pub struct CaseOutcome { + pub stem: String, + pub elapsed: Duration, + pub result: Result<(), String>, +} + +/// Runs `case` (the future for one case, named `stem`) under `budget` of +/// wall time, timed from its first poll; a case over budget is dropped, +/// store included. A panic or a timeout is an ordinary failed case, so a +/// corpus run always reaches every case. +pub async fn run_bounded(stem: &str, budget: Duration, case: F) -> CaseOutcome +where + F: Future>, +{ + let started = Instant::now(); + let case = AssertUnwindSafe(case).catch_unwind(); + let result = match tokio::time::timeout(budget, case).await { + Ok(Ok(result)) => result, + Ok(Err(payload)) => Err(format!( + "case panicked: {}", + panic_message(payload.as_ref()) + )), + Err(_) => Err(format!( + "case exceeded its budget of {:.2}s ({CASE_TIMEOUT_ENV} overrides the default of \ + {DEFAULT_CASE_TIMEOUT_SECS}s; libtest's --test-threads sets how many cases run \ + concurrently; a case over budget belongs in a `heavy-repro:` `#[ignore]`d test, \ + not the corpus)", + budget.as_secs_f64() + )), }; - let expected = rows("{\"n\": 1}\n{\"n\": 1}\n{\"n\": 2}"); - let actual = rows("{\"n\": 2}\n{\"n\": 1}\n{\"n\": 1}"); - compare_rows(&expected, &actual, false).unwrap(); - let missing_dup = rows("{\"n\": 1}\n{\"n\": 2}"); - assert!(compare_rows(&expected, &missing_dup, false).is_err()); -} - -#[test] -fn ordered_comparison_is_positional() { - let rows = |s: &str| -> Vec { - s.lines() - .map(|l| serde_json::from_str(l).unwrap()) - .collect() - }; - let expected = rows("{\"n\": 1}\n{\"n\": 2}"); - let swapped = rows("{\"n\": 2}\n{\"n\": 1}"); - assert!(compare_rows(&expected, &swapped, true).is_err()); - compare_rows(&expected, &expected.clone(), true).unwrap(); -} - -#[tokio::test] -async fn execution_reports_a_row_mismatch() { - let text = - format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect unordered\n{{\"p.name\": \"nobody\"}}\n"); - let case = parse_case("mismatch", &text).unwrap(); - let err = execute_case(&case, Path::new("unused.gqt"), false) - .await - .unwrap_err(); - assert!(err.contains("row mismatch"), "got: {err}"); - assert!(err.contains("step 1 (query)"), "got: {err}"); -} - -#[tokio::test] -async fn bless_rewrites_the_failing_expect_and_converges() { - let dir = tempfile::tempdir().unwrap(); - let path = dir.path().join("bless_case.gqt"); - let text = - format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect unordered\n{{\"p.name\": \"nobody\"}}\n"); - std::fs::write(&path, &text).unwrap(); - let case = parse_case("bless_case", &text).unwrap(); - let err = execute_case(&case, &path, true).await.unwrap_err(); - assert!(err.contains("expect rewritten"), "got: {err}"); - - let blessed = std::fs::read_to_string(&path).unwrap(); - assert!(blessed.contains("{\"p.name\":\"alice\"}"), "got: {blessed}"); - let case = parse_case("bless_case", &blessed).unwrap(); - execute_case(&case, &path, false).await.unwrap(); -} - -#[test] -fn refuses_duplicate_issue_header() { - let text = format!("{HDR}# issue: none\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("x", &text).contains("duplicate")); -} - -#[test] -fn refuses_empty_red_on_value() { - let text = format!("# issue: 7\n# red_on: \n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("issue_7_x", &text).contains("needs a value")); - let text = format!("# issue: 7\n# red_on:\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("issue_7_x", &text).contains("not `# : `")); -} - -#[test] -fn refuses_noncanonical_issue_header_number() { - let text = format!("# issue: 0563\n# red_on: 2026-01-01, red.\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("issue_563_x", &text).contains("no sign or leading zeros")); - let text = format!("# issue: +563\n# red_on: 2026-01-01, red.\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("issue_563_x", &text).contains("no sign or leading zeros")); -} - -#[test] -fn refuses_leading_zero_issue_digits() { - let text = format!("# issue: 7\n# red_on: 2026-01-01, red.\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - assert!(refusal("issue_007_x", &text).contains("leading zeros")); -} - -#[test] -fn refuses_arguments_on_bare_sections() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}--- restart now\n"); - assert!(refusal("x", &text).contains("takes no arguments")); - let junk_query = - "--- query fast\nquery all() {\n match { $p: Person }\n return { $p.name }\n}\n"; - let text = format!("{HDR}{SCHEMA}{SEED}{junk_query}{EXPECT}"); - assert!(refusal("x", &text).contains("takes no arguments")); -} - -#[test] -fn refuses_crlf_line_endings() { - let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}").replace('\n', "\r\n"); - assert!(refusal("x", &text).contains("line endings")); -} - -#[test] -fn refuses_loop_range_over_the_cap() { - let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 0 10001\n{QUERY}{EXPECT}--- endloop\n"); - assert!(refusal("x", &text).contains("10000 cap")); -} - -#[test] -fn refuses_signed_or_padded_numeric_tokens() { - let text = - format!("{HDR}{SCHEMA}{SEED}{MUTATE}{PARAMS}--- expect affected: nodes=+1 edges=0\n"); - assert!(refusal("x", &text).contains("nodes= edges=")); - let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 00 2\n{QUERY}{EXPECT}--- endloop\n"); - assert!(refusal("x", &text).contains("plain decimal")); -} - -#[test] -fn refuses_ok_expect_with_body() { - let text = format!("{HDR}{SCHEMA}{SEED}{MUTATE}{PARAMS}--- expect ok\nstray\n"); - assert!(refusal("x", &text).contains("carries no body")); -} - -#[test] -fn refuses_affected_expect_with_body() { - let text = - format!("{HDR}{SCHEMA}{SEED}{MUTATE}{PARAMS}--- expect affected: nodes=1 edges=0\nstray\n"); - assert!(refusal("x", &text).contains("carries no body")); -} - -#[test] -fn refuses_schema_inside_a_loop() { - let text = format!("{HDR}{SCHEMA}{SEED}--- foreach $x a\n{SCHEMA}{QUERY}{EXPECT}--- endloop\n"); - assert!(refusal("x", &text).contains("out of position")); -} - -#[test] -fn refuses_loop_headers_with_a_body() { - let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 0 2\nstray\n{QUERY}{EXPECT}--- endloop\n"); - assert!(refusal("x", &text).contains("carries no body")); - let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 0 2\n{QUERY}{EXPECT}--- endloop\nstray\n"); - assert!(refusal("x", &text).contains("carries no body")); -} - -#[test] -fn refuses_nearest_over_a_string_property() { - let query = "--- query\nquery q() {\n match { $p: Person }\n return { $p.name }\n order { nearest($p.name, $p.name) }\n}\n"; - let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect unordered\n"); - assert!(refusal("x", &text).contains("vector parameter")); -} - -#[test] -fn traversal_header_forces_index_builds() { - let text = format!("{HDR}# traversal: indexed\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); - let case = parse_case("x", &text).unwrap(); - assert!(case.needs_indices); - assert_eq!(case.traversal, Some("indexed")); -} - -#[test] -fn pin_violation_names_the_path_that_ran() { - assert_eq!(pin_violation("indexed", 3, 0, true), None); - assert_eq!(pin_violation("csr", 0, 2, true), None); - assert_eq!(pin_violation("indexed", 0, 0, false), None); - let v = pin_violation("indexed", 1, 2, false).unwrap(); - assert!( - v.contains("pinned `indexed`, ran `csr` on 2 expand(s)"), - "{v}" - ); - let v = pin_violation("csr", 4, 0, false).unwrap(); - assert!( - v.contains("pinned `csr`, ran `indexed` on 4 expand(s)"), - "{v}" - ); - // A step that must expand and shows nothing on the pinned path: the - // pin and the probes were dropped together. - let v = pin_violation("indexed", 0, 0, true).unwrap(); - assert!(v.contains("no expand ran on it"), "{v}"); - let v = pin_violation("csr", 0, 0, true).unwrap(); - assert!(v.contains("no expand ran on it"), "{v}"); -} - -#[test] -fn expects_expand_ignores_bound_edges_and_plain_bindings() { - let unbound = parse_query(TRAVERSAL_QUERY.trim_start_matches("--- query\n")).unwrap(); - assert!(expects_expand(&unbound.queries[0].match_clause)); - let bound = "query f($n: String) {\n match {\n $a: Person\n $a.name = $n\n \ - $a $k:knows $b\n }\n return { $b.name }\n}\n"; - let bound = parse_query(bound).unwrap(); - assert!(!expects_expand(&bound.queries[0].match_clause)); - let plain = parse_query(QUERY.trim_start_matches("--- query\n")).unwrap(); - assert!(!expects_expand(&plain.queries[0].match_clause)); - let negated = "query f() {\n match {\n $a: Person\n not { $a knows $x }\n }\n \ - return { $a.name }\n}\n"; - let negated = parse_query(negated).unwrap(); - assert!(!expects_expand(&negated.queries[0].match_clause)); -} - -/// A two-node, one-edge graph with a one-hop traversal, for the pin tests. -const TRAVERSAL_SCHEMA: &str = "--- schema\nnode Person {\n name: String @key\n}\n\n\ - edge Knows: Person -> Person {\n since: I64\n}\n"; -const TRAVERSAL_SEED: &str = "--- seed\n{\"type\":\"Person\",\"data\":{\"name\":\"alice\"}}\n\ - {\"type\":\"Person\",\"data\":{\"name\":\"bob\"}}\n\ - {\"edge\":\"Knows\",\"from\":\"alice\",\"to\":\"bob\",\"data\":{\"id\":\"k-1\",\"since\":2020}}\n"; -const TRAVERSAL_QUERY: &str = "--- query\nquery friends($n: String) {\n match {\n $a: Person\n \ - $a.name = $n\n $a knows $b\n }\n return { $b.name }\n}\n"; -const TRAVERSAL_PARAMS: &str = "--- params\n{\"n\": \"alice\"}\n"; -const TRAVERSAL_EXPECT: &str = "--- expect unordered\n{\"b.name\": \"bob\"}\n"; - -/// The pin reaches the executor on both paths: a pinned step runs its -/// expands on the pinned path only, and the probes see them (a zero count -/// on both paths would make the check vacuous). -#[tokio::test] -async fn pinned_step_runs_only_its_pinned_path() { - for (mode, expect_indexed) in [("indexed", true), ("csr", false)] { - let text = format!( - "{HDR}# traversal: {mode}\n{TRAVERSAL_SCHEMA}{TRAVERSAL_SEED}{TRAVERSAL_QUERY}{TRAVERSAL_PARAMS}{TRAVERSAL_EXPECT}" - ); - let case = parse_case("pinned", &text).unwrap(); - execute_case(&case, Path::new("unused.gqt"), false) - .await - .unwrap_or_else(|e| panic!("{mode}: {e}")); - - let (db, _uri, _dir) = open_case_store(&case).await.unwrap(); - let Some(Item::Step(Step::Query(step))) = case.items.first() else { - panic!("first item is the query step"); - }; - let params = build_params(step.params_raw.as_ref(), &step.ast_params, None).unwrap(); - let (outcome, counts) = under_traversal( - Some(mode), - db.query( - ReadTarget::branch("main"), - &step.source, - &step.name, - ¶ms, - ), - ) - .await; - outcome.unwrap(); - let counts = counts.unwrap(); - let (indexed, csr) = ( - counts.indexed.load(Ordering::Relaxed), - counts.csr.load(Ordering::Relaxed), - ); - if expect_indexed { - assert!( - indexed >= 1 && csr == 0, - "{mode}: indexed={indexed} csr={csr}" - ); - } else { - assert!( - csr >= 1 && indexed == 0, - "{mode}: indexed={indexed} csr={csr}" - ); - } + CaseOutcome { + stem: stem.to_string(), + elapsed: started.elapsed(), + result, } } -#[test] -fn refuses_ordered_expect_on_an_rrf_led_order() { - let query = "--- query\nquery q($v: Vector(4), $t: String) {\n match { $p: Person }\n \ - return { $p.name }\n order { rrf(nearest($p.vec, $v), bm25($p.name, $t)) }\n}\n"; - let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect ordered\n"); - let message = refusal("x", &text); - assert!(message.contains("led by `rrf()`"), "{message}"); - let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect unordered\n"); - parse_case("x", &text).unwrap(); -} - -#[test] -fn refuses_ordered_expect_with_an_aggregate_in_return() { - let query = "--- query\nquery q($t: String) {\n match { $p: Person\n search($p.name, $t) }\n \ - return { count($p) as total }\n order { bm25($p.name, $t) }\n}\n"; - let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect ordered\n"); - assert!(refusal("x", &text).contains("aggregate in its `return` list")); - let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect unordered\n"); - parse_case("x", &text).unwrap(); -} - -fn synthetic_cases(names: &[&str]) -> Vec<(String, PathBuf)> { - names - .iter() - .map(|n| ((*n).to_string(), PathBuf::from(format!("{n}.gqt")))) - .collect() -} - -fn no_report(_: &CaseOutcome) {} - -#[tokio::test(flavor = "multi_thread")] -async fn walker_bounds_cases_in_flight() { - let in_flight = Arc::new(AtomicUsize::new(0)); - let max_seen = Arc::new(AtomicUsize::new(0)); - let names: Vec = (0..12).map(|i| format!("c{i:02}")).collect(); - let cases = synthetic_cases(&names.iter().map(String::as_str).collect::>()); - // Every permit holder waits at a 3-party barrier, so the three are in - // flight together (max == 3) or the walker admitted fewer than three - // and the barrier never releases: the outer timeout turns that hang - // into a failure instead of a stall. Twelve cases = four generations. - let barrier = Arc::new(tokio::sync::Barrier::new(3)); - let (counter, max) = (Arc::clone(&in_flight), Arc::clone(&max_seen)); - let runner: CaseRunner = Arc::new(move |_| { - let (counter, max, barrier) = - (Arc::clone(&counter), Arc::clone(&max), Arc::clone(&barrier)); - Box::pin(async move { - let now = counter.fetch_add(1, Ordering::SeqCst) + 1; - max.fetch_max(now, Ordering::SeqCst); - barrier.wait().await; - counter.fetch_sub(1, Ordering::SeqCst); - Ok(()) - }) - }); - let outcomes = tokio::time::timeout( - Duration::from_secs(30), - run_bounded(cases, 3, Duration::from_secs(10), runner, &no_report), - ) - .await - .expect("fewer than three cases in flight: the barrier never released"); - assert_eq!(outcomes.len(), 12); - assert!(outcomes.iter().all(|o| o.result.is_ok()), "{outcomes:?}"); - assert_eq!(max_seen.load(Ordering::SeqCst), 3); - assert_eq!(in_flight.load(Ordering::SeqCst), 0); -} - -#[tokio::test(flavor = "multi_thread")] -async fn walker_reports_in_completion_order_and_returns_sorted() { - // Completion is forced into the order c2, c1, c0 by hand-offs, not by - // sleep lengths: c2 returns at once; c1 waits until c2 is reported; - // c0 waits until c1 is reported. - let gates: Arc<[tokio::sync::Notify; 2]> = - Arc::new([tokio::sync::Notify::new(), tokio::sync::Notify::new()]); - let runner_gates = Arc::clone(&gates); - let runner: CaseRunner = Arc::new(move |path| { - let gates = Arc::clone(&runner_gates); - let idx: usize = path.file_stem().unwrap().to_str().unwrap()[1..] - .parse() - .unwrap(); - Box::pin(async move { - if idx < 2 { - gates[idx].notified().await; - } - Ok(()) - }) - }); - let reported = std::sync::Mutex::new(Vec::new()); - let report = |o: &CaseOutcome| { - reported.lock().unwrap().push(o.stem.clone()); - match o.stem.as_str() { - "c2" => gates[1].notify_one(), - "c1" => gates[0].notify_one(), - _ => {} - } - }; - let cases = synthetic_cases(&["c0", "c1", "c2"]); - let outcomes = tokio::time::timeout( - Duration::from_secs(30), - run_bounded(cases, 3, Duration::from_secs(10), runner, &report), - ) - .await - .expect("a hand-off never arrived"); - let returned: Vec<&str> = outcomes.iter().map(|o| o.stem.as_str()).collect(); - assert_eq!(returned, ["c0", "c1", "c2"]); - assert_eq!(*reported.lock().unwrap(), ["c2", "c1", "c0"]); -} - -#[tokio::test(flavor = "multi_thread")] -async fn walker_budget_starts_at_the_permit_not_at_spawn() { - let runner: CaseRunner = Arc::new(|_| { - Box::pin(async { - tokio::time::sleep(Duration::from_millis(300)).await; - Ok(()) - }) - }); - let cases = synthetic_cases(&["a", "b", "c"]); - // One permit: the third case waits ~600 ms in the queue, past the - // 500 ms budget that its own 300 ms of work stays under; the margins - // are wide because libtest runs this beside the corpus walker. - let outcomes = run_bounded(cases, 1, Duration::from_millis(500), runner, &no_report).await; - assert!(outcomes.iter().all(|o| o.result.is_ok()), "{outcomes:?}"); -} - -#[tokio::test(flavor = "multi_thread")] -async fn walker_fails_a_case_over_its_budget_and_runs_the_rest() { - let runner: CaseRunner = Arc::new(|path| { - Box::pin(async move { - if path.starts_with("slow.gqt") { - tokio::time::sleep(Duration::from_secs(30)).await; - } - Ok(()) - }) - }); - let cases = synthetic_cases(&["slow", "quick"]); - let outcomes = run_bounded(cases, 1, Duration::from_millis(50), runner, &no_report).await; - let quick = &outcomes[0]; - assert_eq!(quick.stem, "quick"); - assert!(quick.result.is_ok(), "{quick:?}"); - let slow = &outcomes[1]; - assert_eq!(slow.stem, "slow"); - assert!( - slow.elapsed < Duration::from_secs(5), - "timeout did not cut the case short" - ); - let err = slow.result.as_ref().unwrap_err(); - assert!(err.contains("budget of 0.05s"), "{err}"); - assert!(err.contains("up to 1 cases"), "{err}"); - assert!(err.contains(CASE_TIMEOUT_ENV), "{err}"); -} - -#[tokio::test(flavor = "multi_thread")] -async fn walker_records_a_panicking_case_and_runs_the_rest() { - let runner: CaseRunner = Arc::new(|path| { - Box::pin(async move { - if path.starts_with("p.gqt") { - panic!("boom"); - } - Ok(()) - }) - }); - let cases = synthetic_cases(&["p", "q"]); - let outcomes = run_bounded(cases, 1, Duration::from_secs(10), runner, &no_report).await; - let err = outcomes[0].result.as_ref().unwrap_err(); - assert!(err.starts_with("case panicked: boom"), "{err}"); - assert!(outcomes[1].result.is_ok(), "{:?}", outcomes[1]); -} - -#[test] -fn walker_refuses_a_process_traversal_override() { - assert!(traversal_override_refusal(None).is_none()); - let reason = traversal_override_refusal(Some(OsStr::new("csr"))).unwrap(); - assert!(reason.contains("OMNIGRAPH_TRAVERSAL_MODE=csr"), "{reason}"); - assert!(reason.contains("# traversal:"), "{reason}"); -} - -/// Same name battery as `scripts/check-fix-regression.py --self-test` -/// (`corpus_case`): the walker and the gate must agree on what a case is. -#[test] -fn walker_flags_foreign_corpus_entries() { - let dir = tempfile::tempdir().unwrap(); - std::fs::write(dir.path().join("a.gqt"), "x").unwrap(); - std::fs::write(dir.path().join("b.txt"), "x").unwrap(); - std::fs::write(dir.path().join(".hidden.gqt"), "x").unwrap(); - std::fs::write(dir.path().join(".DS_Store"), "x").unwrap(); - std::fs::write(dir.path().join("c.GQT"), "x").unwrap(); - std::fs::create_dir(dir.path().join("nested")).unwrap(); - std::fs::write(dir.path().join("nested").join("d.gqt"), "x").unwrap(); - let (files, foreign) = list_cases(dir.path()); - assert_eq!(files, vec![dir.path().join("a.gqt")]); - assert_eq!( - foreign, - vec![ - ".hidden.gqt".to_string(), - "b.txt".to_string(), - "c.GQT".to_string(), - "nested".to_string() - ] - ); -} - -#[tokio::test] -async fn bless_refuses_cases_containing_loops() { - let dir = tempfile::tempdir().unwrap(); - let path = dir.path().join("bless_loop_case.gqt"); - let text = format!( - "{HDR}{SCHEMA}{SEED}--- foreach $x a b\n{QUERY}--- expect unordered\n{{\"p.name\": \"nobody\"}}\n--- endloop\n" - ); - std::fs::write(&path, &text).unwrap(); - let case = parse_case("bless_loop_case", &text).unwrap(); - let err = execute_case(&case, &path, true).await.unwrap_err(); - assert!(err.contains("bless: refused"), "got: {err}"); - assert_eq!(std::fs::read_to_string(&path).unwrap(), text); -} - -#[tokio::test] -async fn bless_never_rewrites_on_a_kind_mismatch() { - let dir = tempfile::tempdir().unwrap(); - let path = dir.path().join("bless_kind_case.gqt"); - let query = "--- query\nquery q() {\n match { $p: Person }\n return { $p.nope }\n}\n"; - let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect unordered\n{{\"p.nope\": \"x\"}}\n"); - std::fs::write(&path, &text).unwrap(); - let case = parse_case("bless_kind_case", &text).unwrap(); - let err = execute_case(&case, &path, true).await.unwrap_err(); - assert!(err.contains("query failed"), "got: {err}"); - assert_eq!(std::fs::read_to_string(&path).unwrap(), text); -} - -#[tokio::test] -async fn params_refusal_satisfies_an_error_expect() { - let query = "--- query\nquery q($q: String) {\n match {\n $p: Person\n $p.name = $q\n }\n return { $p.name }\n}\n"; - let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect error: q\n"); - let case = parse_case("x", &text).unwrap(); - execute_case(&case, Path::new("unused.gqt"), false) - .await - .unwrap(); -} - -#[test] -fn bless_splice_preserves_the_trailing_blank_separator() { - let original = "--- expect unordered\nold\n\n--- restart\n"; - let span = BodySpan { - start_line: 1, - len: 2, - }; - let rows = vec!["{\"n\":1}".to_string()]; - let out = splice_lines(original, span, &rows); - assert_eq!(out, "--- expect unordered\n{\"n\":1}\n\n--- restart\n"); +/// [`run_bounded`] over [`run_case`] for the file at `path`. +pub async fn run_case_bounded(path: PathBuf, budget: Duration, bless: bool) -> CaseOutcome { + let stem = stem_of(&path); + run_bounded(&stem, budget, run_case(path, bless)).await } -#[test] -fn bless_splice_replaces_only_the_expect_body() { - let original = "--- query\nq\n--- expect unordered\nold row\nold row 2\n--- restart\n"; - let span = BodySpan { - start_line: 3, - len: 2, - }; - let rows = vec!["{\"n\":1}".to_string()]; - let out = splice_lines(original, span, &rows); - assert_eq!( - out, - "--- query\nq\n--- expect unordered\n{\"n\":1}\n--- restart\n" - ); -} - -#[test] -fn bless_splice_inserts_into_an_empty_expect_body() { - let original = "--- expect unordered\n--- restart\n"; - let span = BodySpan { - start_line: 1, - len: 0, - }; - let rows = vec!["{\"n\":1}".to_string()]; - let out = splice_lines(original, span, &rows); - assert_eq!(out, "--- expect unordered\n{\"n\":1}\n--- restart\n"); -} +#[cfg(test)] +mod tests; diff --git a/crates/omnigraph-gqt/src/tests.rs b/crates/omnigraph-gqt/src/tests.rs new file mode 100644 index 000000000..d2fb241e4 --- /dev/null +++ b/crates/omnigraph-gqt/src/tests.rs @@ -0,0 +1,929 @@ +use super::*; + +const HDR: &str = "# issue: none\n"; +const SCHEMA: &str = "--- schema\nnode Person {\n name: String @key\n}\n"; +const SEED: &str = "--- seed\n{\"type\":\"Person\",\"data\":{\"name\":\"alice\"}}\n"; +const QUERY: &str = + "--- query\nquery all() {\n match { $p: Person }\n return { $p.name }\n}\n"; +const EXPECT: &str = "--- expect unordered\n{\"p.name\": \"alice\"}\n"; +const MUTATE: &str = "--- mutate\nquery ins($n: String) {\n insert Person { name: $n }\n}\n"; +const PARAMS: &str = "--- params\n{\"n\": \"bob\"}\n"; +const EXPECT_OK: &str = "--- expect ok\n"; + +fn refusal(stem: &str, text: &str) -> String { + parse_case(stem, text).expect_err("expected the case to be refused") +} + +#[test] +fn parses_a_minimal_case() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}"); + let case = parse_case("minimal", &text).unwrap(); + assert_eq!(case.items.len(), 1); + assert!(!case.needs_indices); + assert_eq!(case.traversal, None); +} + +#[test] +fn header_notes_repeat_and_continuation_lines_are_refused() { + let text = format!( + "# issue: 7\n# red_on: 2026-01-01, the run\n# notes: returned 8,\n# notes: not 20.\n{SCHEMA}{SEED}{QUERY}{EXPECT}" + ); + parse_case("issue_7_notes", &text).unwrap(); + let text = format!( + "# issue: 7\n# red_on: 2026-01-01, the run\n# returned 8: not 20.\n{SCHEMA}{SEED}{QUERY}{EXPECT}" + ); + assert!(refusal("issue_7_x", &text).contains("unknown header key")); + // The three misspellings that the old prose branch swallowed silently. + for typo in [ + "# Traversal: indexed", + "# traversal : indexed", + "# traversal=csr", + ] { + let text = format!("{HDR}{typo}\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + let reason = refusal("x", &text); + assert!( + reason.contains("unknown header key") || reason.contains("not `# : `"), + "{typo}: {reason}" + ); + } +} + +/// Bounded exhaustive walk of the header-line typo space: key spelling, +/// separator, leading and trailing whitespace, and the gap before the value. +/// A line is accepted exactly when it equals the canonical +/// `# traversal: indexed`, so a future key inherits the same proof. +#[test] +fn header_lines_are_accepted_only_in_canonical_form() { + let keys = [ + "traversal", + "Traversal", + "TRAVERSAL", + "traversa1", + "traversal_", + " traversal", + ]; + let seps = [":", " :", "=", "", "::"]; + let leads = ["# ", "#", "# ", " # "]; + let gaps = [" ", "", " ", "\t"]; + let trails = ["", " ", "\t"]; + let canonical = canonical_header_line("traversal", "indexed"); + // Distinct lines: two lead/key pairs collide (`#` + ` traversal` = `# ` + + // `traversal`, and `# ` + ` traversal` = `# ` + `traversal`), 120 + // duplicates over the 1440 grid points, so the set is what gets walked. + let mut lines = std::collections::BTreeSet::new(); + for lead in leads { + for key in keys { + for sep in seps { + for gap in gaps { + for trail in trails { + lines.insert(format!("{lead}{key}{sep}{gap}indexed{trail}")); + } + } + } + } + } + assert_eq!(lines.len(), 1320, "the typo space is 1320 distinct lines"); + let mut accepted = 0; + for line in &lines { + let text = format!("{HDR}{line}\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + match parse_case("x", &text) { + Ok(case) => { + assert_eq!(line, &canonical, "accepted a non-canonical line"); + assert_eq!(case.traversal, Some("indexed")); + accepted += 1; + } + Err(e) => assert_ne!(line, &canonical, "refused the canonical line: {e}"), + } + } + assert_eq!(accepted, 1); +} + +#[test] +fn refuses_missing_issue_header() { + let text = format!("# notes: no anchor\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("# issue:")); +} + +#[test] +fn refuses_numbered_issue_without_red_on() { + let text = format!("# issue: 7\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("issue_7_x", &text).contains("red_on")); +} + +#[test] +fn refuses_header_line_without_a_key() { + let text = format!("# stray prose\n{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("not `# : `")); +} + +#[test] +fn refuses_unknown_header_key() { + let text = format!("{HDR}# owner: me\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("unknown header key")); +} + +#[test] +fn refuses_bad_traversal_mode() { + let text = format!("{HDR}# traversal: bogus\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("indexed")); +} + +#[test] +fn refuses_non_header_line_before_first_section() { + let text = format!("{HDR}stray\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("precede the first section")); +} + +#[test] +fn refuses_comment_line_in_seed() { + let text = format!("{HDR}{SCHEMA}--- seed\n# a comment\n{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("seed")); +} + +#[test] +fn refuses_comment_line_in_expect_body() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect unordered\n# nope\n"); + assert!(refusal("x", &text).contains("expect")); +} + +#[test] +fn refuses_seed_before_schema() { + let text = format!("{HDR}{SEED}{SCHEMA}{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("first section")); +} + +#[test] +fn refuses_missing_seed_section() { + let text = format!("{HDR}{SCHEMA}{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("second section")); +} + +#[test] +fn refuses_case_without_a_query_or_mutate_step() { + let text = format!("{HDR}{SCHEMA}{SEED}"); + assert!(refusal("x", &text).contains("at least one query or mutate step")); +} + +#[test] +fn refuses_restart_only_step_list() { + let text = format!("{HDR}{SCHEMA}{SEED}--- restart\n"); + assert!(refusal("x", &text).contains("at least one query or mutate step")); +} + +#[test] +fn refuses_second_declaration_in_one_section() { + let two = "--- query\nquery a() {\n match { $p: Person }\n return { $p.name }\n}\nquery b() {\n match { $p: Person }\n return { $p.name }\n}\n"; + let text = format!("{HDR}{SCHEMA}{SEED}{two}{EXPECT}"); + assert!(refusal("x", &text).contains("exactly one declaration")); +} + +#[test] +fn refuses_mutation_declaration_under_query() { + let text = format!( + "{HDR}{SCHEMA}{SEED}--- query\nquery ins($n: String) {{\n insert Person {{ name: $n }}\n}}\n{EXPECT}" + ); + assert!(refusal("x", &text).contains("use `--- mutate`")); +} + +#[test] +fn refuses_read_declaration_under_mutate() { + let text = format!( + "{HDR}{SCHEMA}{SEED}--- mutate\nquery all() {{\n match {{ $p: Person }}\n return {{ $p.name }}\n}}\n{EXPECT_OK}" + ); + assert!(refusal("x", &text).contains("use `--- query`")); +} + +#[test] +fn refuses_bare_expect() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect\n"); + assert!(refusal("x", &text).contains("mode word")); +} + +#[test] +fn refuses_error_expect_without_substring() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect error:\n"); + assert!(refusal("x", &text).contains("substring")); +} + +#[test] +fn refuses_error_expect_with_body() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect error: boom\nbody\n"); + assert!(refusal("x", &text).contains("carries no body")); +} + +#[test] +fn refuses_affected_expect_missing_a_count() { + let text = format!("{HDR}{SCHEMA}{SEED}{MUTATE}{PARAMS}--- expect affected: nodes=1\n"); + assert!(refusal("x", &text).contains("nodes= edges=")); +} + +#[test] +fn refuses_unknown_expect_mode() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect sorted\n"); + assert!(refusal("x", &text).contains("unknown expect mode")); +} + +#[test] +fn refuses_row_expect_on_a_mutate_step() { + let text = format!("{HDR}{SCHEMA}{SEED}{MUTATE}{PARAMS}{EXPECT}"); + assert!(refusal("x", &text).contains("carry no rows")); +} + +#[test] +fn refuses_ok_expect_on_a_query_step() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT_OK}"); + assert!(refusal("x", &text).contains("a query step takes")); +} + +#[test] +fn refuses_query_step_without_expect() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}"); + assert!(refusal("x", &text).contains("missing its `--- expect`")); +} + +#[test] +fn refuses_expect_with_no_step_to_bind_to() { + let text = format!("{HDR}{SCHEMA}{SEED}--- restart\n{EXPECT}{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("no query or mutate step to bind to")); +} + +#[test] +fn refuses_params_without_a_step() { + let text = format!("{HDR}{SCHEMA}{SEED}{PARAMS}{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("directly follow")); +} + +#[test] +fn refuses_second_params_for_one_step() { + let text = format!("{HDR}{SCHEMA}{SEED}{MUTATE}{PARAMS}{PARAMS}{EXPECT_OK}"); + assert!(refusal("x", &text).contains("second `--- params`")); +} + +#[test] +fn refuses_restart_with_a_body() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}--- restart\nstray\n"); + assert!(refusal("x", &text).contains("carries no body")); +} + +#[test] +fn refuses_unknown_section() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}--- teardown\n"); + assert!(refusal("x", &text).contains("unknown section")); +} + +#[test] +fn refuses_schema_out_of_position() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}{SCHEMA}"); + assert!(refusal("x", &text).contains("out of position")); +} + +#[test] +fn refuses_negative_loop_bound() { + let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i -1 2\n{QUERY}{EXPECT}--- endloop\n"); + assert!(refusal("x", &text).contains("non-negative")); +} + +#[test] +fn refuses_empty_loop_range() { + let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 3 3\n{QUERY}{EXPECT}--- endloop\n"); + assert!(refusal("x", &text).contains("empty loop range")); +} + +#[test] +fn refuses_foreach_without_values() { + let text = format!("{HDR}{SCHEMA}{SEED}--- foreach $x\n{QUERY}{EXPECT}--- endloop\n"); + assert!(refusal("x", &text).contains("no values")); +} + +#[test] +fn refuses_foreach_value_outside_charset() { + let text = format!("{HDR}{SCHEMA}{SEED}--- foreach $x a\"b\n{QUERY}{EXPECT}--- endloop\n"); + assert!(refusal("x", &text).contains("[A-Za-z0-9_.-]")); +} + +#[test] +fn refuses_bad_loop_variable_name() { + let text = format!("{HDR}{SCHEMA}{SEED}--- loop $I 0 2\n{QUERY}{EXPECT}--- endloop\n"); + assert!(refusal("x", &text).contains("$[a-z][a-z0-9_]*")); +} + +#[test] +fn refuses_nested_loops() { + let text = format!( + "{HDR}{SCHEMA}{SEED}--- loop $i 0 2\n--- loop $j 0 2\n{QUERY}{EXPECT}--- endloop\n--- endloop\n" + ); + assert!(refusal("x", &text).contains("may not nest")); +} + +#[test] +fn refuses_endloop_without_a_loop() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}--- endloop\n"); + assert!(refusal("x", &text).contains("without an open loop")); +} + +#[test] +fn refuses_unclosed_loop() { + let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 0 2\n{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("not closed")); +} + +#[test] +fn refuses_loop_enclosing_no_steps() { + let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 0 2\n--- endloop\n{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("enclosing no steps")); +} + +#[test] +fn refuses_substitution_marker_in_query_body() { + let query = "--- query\nquery all() {\n match { $p: Person }\n return { $p.name }\n}\n"; + let text = + format!("{HDR}{SCHEMA}{SEED}{query}{EXPECT}").replace("$p.name }", "$p.name } // ${i}"); + assert!(refusal("x", &text).contains("only inside a params or expect body")); +} + +#[test] +fn refuses_substitution_marker_in_seed() { + let text = format!( + "{HDR}{SCHEMA}--- seed\n{{\"type\":\"Person\",\"data\":{{\"name\":\"${{i}}\"}}}}\n{QUERY}{EXPECT}" + ); + assert!(refusal("x", &text).contains("only inside a params or expect body")); +} + +#[test] +fn refuses_substitution_outside_a_loop() { + let text = + format!("{HDR}{SCHEMA}{SEED}{MUTATE}--- params\n{{\"n\": \"${{who}}\"}}\n{EXPECT_OK}"); + assert!(refusal("x", &text).contains("outside a loop")); +} + +#[test] +fn refuses_substitution_naming_the_wrong_variable() { + let text = format!( + "{HDR}{SCHEMA}{SEED}--- foreach $who bob\n{MUTATE}--- params\n{{\"n\": \"${{other}}\"}}\n{EXPECT_OK}--- endloop\n" + ); + assert!(refusal("x", &text).contains("enclosing loop's variable")); +} + +#[test] +fn refuses_unterminated_substitution() { + let text = format!( + "{HDR}{SCHEMA}{SEED}--- foreach $who bob\n{MUTATE}--- params\n{{\"n\": \"${{who\n{EXPECT_OK}--- endloop\n" + ); + assert!(refusal("x", &text).contains("unterminated")); +} + +#[test] +fn refuses_file_name_disagreeing_with_issue_header() { + let text = format!("# issue: 7\n# red_on: 2026-01-01, red.\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("issue_8_wrong", &text).contains("disagrees")); +} + +#[test] +fn refuses_issue_prefix_without_number_or_short_name() { + let text = format!("# issue: 7\n# red_on: 2026-01-01, red.\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("issue_7", &text).contains("issue__")); + assert!(refusal("issue_x", &text).contains("issue__")); +} + +#[test] +fn refuses_feature_name_with_numbered_issue_header() { + let text = format!("# issue: 7\n# red_on: 2026-01-01, red.\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("feature_name", &text).contains("issue_7_")); +} + +#[test] +fn refuses_file_name_outside_charset() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("Bad-Name", &text).contains("[a-z0-9_]")); +} + +#[test] +fn refuses_ordered_expect_without_an_order_clause() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect ordered\n{{\"p.name\": \"alice\"}}\n"); + assert!(refusal("x", &text).contains("order` clause")); +} + +#[test] +fn refuses_embed_schema() { + let schema = "--- schema\nnode Doc {\n slug: String @key\n text: String\n vec: Vector(4) @embed(\"text\")\n}\n"; + let seed = "--- seed\n"; + let text = format!("{HDR}{schema}{seed}{QUERY}{EXPECT}") + .replace("$p: Person", "$p: Doc") + .replace("$p.name", "$p.slug"); + assert!(refusal("x", &text).contains("@embed")); +} + +#[test] +fn refuses_nearest_over_a_string_literal() { + let query = "--- query\nquery q() {\n match { $p: Person }\n return { $p.name }\n order { nearest($p.name, \"alpha\") }\n}\n"; + let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect unordered\n"); + assert!(refusal("x", &text).contains("string argument")); +} + +#[test] +fn refuses_nearest_over_a_string_param() { + let query = "--- query\nquery q($q: String) {\n match { $p: Person }\n return { $p.name }\n order { nearest($p.name, $q) }\n}\n"; + let text = format!( + "{HDR}{SCHEMA}{SEED}{query}--- params\n{{\"q\": \"alpha\"}}\n--- expect unordered\n" + ); + assert!(refusal("x", &text).contains("string argument")); +} + +#[test] +fn accepts_empty_expect_body_as_empty_result_assertion() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect unordered\n"); + parse_case("x", &text).unwrap(); +} + +#[test] +fn search_construct_sets_the_index_decision() { + let schema = "--- schema\nnode Doc {\n slug: String @key\n text: String @index\n}\n"; + let query = "--- query\nquery q($q: String) {\n match {\n $d: Doc\n search($d.text, $q)\n }\n return { $d.slug }\n}\n"; + let text = format!( + "{HDR}{schema}--- seed\n{query}--- params\n{{\"q\": \"needle\"}}\n--- expect unordered\n" + ); + let case = parse_case("x", &text).unwrap(); + assert!(case.needs_indices); +} + +#[test] +fn normalization_equates_integer_and_float_spellings() { + let a: Value = serde_json::from_str("{\"total\": 2}").unwrap(); + let b: Value = serde_json::from_str("{\"total\": 2.0}").unwrap(); + assert_eq!(canonical_json(&a), canonical_json(&b)); +} + +#[test] +fn normalization_does_not_collapse_large_integers() { + let a: Value = serde_json::from_str("{\"n\": 9007199254740993}").unwrap(); + let b: Value = serde_json::from_str("{\"n\": 9007199254740992}").unwrap(); + assert_ne!(canonical_json(&a), canonical_json(&b)); +} + +#[test] +fn normalization_ignores_noise_below_scale_12() { + let a: Value = serde_json::from_str("{\"x\": 0.1000000000000001}").unwrap(); + let b: Value = serde_json::from_str("{\"x\": 0.1}").unwrap(); + assert_eq!(canonical_json(&a), canonical_json(&b)); +} + +#[test] +fn canonical_form_sorts_object_keys_and_recurses() { + let a: Value = serde_json::from_str("{\"b\": [{\"z\": 1, \"a\": 2}], \"a\": null}").unwrap(); + assert_eq!(canonical_json(&a), "{\"a\":null,\"b\":[{\"a\":2,\"z\":1}]}"); +} + +#[test] +fn unordered_comparison_is_multiset_equality() { + let rows = |s: &str| -> Vec { + s.lines() + .map(|l| serde_json::from_str(l).unwrap()) + .collect() + }; + let expected = rows("{\"n\": 1}\n{\"n\": 1}\n{\"n\": 2}"); + let actual = rows("{\"n\": 2}\n{\"n\": 1}\n{\"n\": 1}"); + compare_rows(&expected, &actual, false).unwrap(); + let missing_dup = rows("{\"n\": 1}\n{\"n\": 2}"); + assert!(compare_rows(&expected, &missing_dup, false).is_err()); +} + +#[test] +fn ordered_comparison_is_positional() { + let rows = |s: &str| -> Vec { + s.lines() + .map(|l| serde_json::from_str(l).unwrap()) + .collect() + }; + let expected = rows("{\"n\": 1}\n{\"n\": 2}"); + let swapped = rows("{\"n\": 2}\n{\"n\": 1}"); + assert!(compare_rows(&expected, &swapped, true).is_err()); + compare_rows(&expected, &expected.clone(), true).unwrap(); +} + +#[tokio::test] +async fn execution_reports_a_row_mismatch() { + let text = + format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect unordered\n{{\"p.name\": \"nobody\"}}\n"); + let case = parse_case("mismatch", &text).unwrap(); + let err = execute_case(&case, Path::new("unused.gqt"), false) + .await + .unwrap_err(); + assert!(err.contains("row mismatch"), "got: {err}"); + assert!(err.contains("step 1 (query)"), "got: {err}"); +} + +#[tokio::test] +async fn bless_rewrites_the_failing_expect_and_converges() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("bless_case.gqt"); + let text = + format!("{HDR}{SCHEMA}{SEED}{QUERY}--- expect unordered\n{{\"p.name\": \"nobody\"}}\n"); + std::fs::write(&path, &text).unwrap(); + let case = parse_case("bless_case", &text).unwrap(); + let err = execute_case(&case, &path, true).await.unwrap_err(); + assert!(err.contains("expect rewritten"), "got: {err}"); + + let blessed = std::fs::read_to_string(&path).unwrap(); + assert!(blessed.contains("{\"p.name\":\"alice\"}"), "got: {blessed}"); + let case = parse_case("bless_case", &blessed).unwrap(); + execute_case(&case, &path, false).await.unwrap(); +} + +#[test] +fn refuses_duplicate_issue_header() { + let text = format!("{HDR}# issue: none\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("x", &text).contains("duplicate")); +} + +#[test] +fn refuses_empty_red_on_value() { + let text = format!("# issue: 7\n# red_on: \n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("issue_7_x", &text).contains("needs a value")); + let text = format!("# issue: 7\n# red_on:\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("issue_7_x", &text).contains("not `# : `")); +} + +#[test] +fn refuses_noncanonical_issue_header_number() { + let text = format!("# issue: 0563\n# red_on: 2026-01-01, red.\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("issue_563_x", &text).contains("no sign or leading zeros")); + let text = format!("# issue: +563\n# red_on: 2026-01-01, red.\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("issue_563_x", &text).contains("no sign or leading zeros")); +} + +#[test] +fn refuses_leading_zero_issue_digits() { + let text = format!("# issue: 7\n# red_on: 2026-01-01, red.\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + assert!(refusal("issue_007_x", &text).contains("leading zeros")); +} + +#[test] +fn refuses_arguments_on_bare_sections() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}--- restart now\n"); + assert!(refusal("x", &text).contains("takes no arguments")); + let junk_query = + "--- query fast\nquery all() {\n match { $p: Person }\n return { $p.name }\n}\n"; + let text = format!("{HDR}{SCHEMA}{SEED}{junk_query}{EXPECT}"); + assert!(refusal("x", &text).contains("takes no arguments")); +} + +#[test] +fn refuses_crlf_line_endings() { + let text = format!("{HDR}{SCHEMA}{SEED}{QUERY}{EXPECT}").replace('\n', "\r\n"); + assert!(refusal("x", &text).contains("line endings")); +} + +#[test] +fn refuses_loop_range_over_the_cap() { + let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 0 10001\n{QUERY}{EXPECT}--- endloop\n"); + assert!(refusal("x", &text).contains("10000 cap")); +} + +#[test] +fn refuses_signed_or_padded_numeric_tokens() { + let text = + format!("{HDR}{SCHEMA}{SEED}{MUTATE}{PARAMS}--- expect affected: nodes=+1 edges=0\n"); + assert!(refusal("x", &text).contains("nodes= edges=")); + let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 00 2\n{QUERY}{EXPECT}--- endloop\n"); + assert!(refusal("x", &text).contains("plain decimal")); +} + +#[test] +fn refuses_ok_expect_with_body() { + let text = format!("{HDR}{SCHEMA}{SEED}{MUTATE}{PARAMS}--- expect ok\nstray\n"); + assert!(refusal("x", &text).contains("carries no body")); +} + +#[test] +fn refuses_affected_expect_with_body() { + let text = + format!("{HDR}{SCHEMA}{SEED}{MUTATE}{PARAMS}--- expect affected: nodes=1 edges=0\nstray\n"); + assert!(refusal("x", &text).contains("carries no body")); +} + +#[test] +fn refuses_schema_inside_a_loop() { + let text = format!("{HDR}{SCHEMA}{SEED}--- foreach $x a\n{SCHEMA}{QUERY}{EXPECT}--- endloop\n"); + assert!(refusal("x", &text).contains("out of position")); +} + +#[test] +fn refuses_loop_headers_with_a_body() { + let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 0 2\nstray\n{QUERY}{EXPECT}--- endloop\n"); + assert!(refusal("x", &text).contains("carries no body")); + let text = format!("{HDR}{SCHEMA}{SEED}--- loop $i 0 2\n{QUERY}{EXPECT}--- endloop\nstray\n"); + assert!(refusal("x", &text).contains("carries no body")); +} + +#[test] +fn refuses_nearest_over_a_string_property() { + let query = "--- query\nquery q() {\n match { $p: Person }\n return { $p.name }\n order { nearest($p.name, $p.name) }\n}\n"; + let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect unordered\n"); + assert!(refusal("x", &text).contains("vector parameter")); +} + +#[test] +fn traversal_header_forces_index_builds() { + let text = format!("{HDR}# traversal: indexed\n{SCHEMA}{SEED}{QUERY}{EXPECT}"); + let case = parse_case("x", &text).unwrap(); + assert!(case.needs_indices); + assert_eq!(case.traversal, Some("indexed")); +} + +#[test] +fn pin_violation_names_the_path_that_ran() { + assert_eq!(pin_violation("indexed", 3, 0, true), None); + assert_eq!(pin_violation("csr", 0, 2, true), None); + assert_eq!(pin_violation("indexed", 0, 0, false), None); + let v = pin_violation("indexed", 1, 2, false).unwrap(); + assert!( + v.contains("pinned `indexed`, ran `csr` on 2 expand(s)"), + "{v}" + ); + let v = pin_violation("csr", 4, 0, false).unwrap(); + assert!( + v.contains("pinned `csr`, ran `indexed` on 4 expand(s)"), + "{v}" + ); + // A step that must expand and shows nothing on the pinned path: the + // pin and the probes were dropped together. + let v = pin_violation("indexed", 0, 0, true).unwrap(); + assert!(v.contains("no expand ran on it"), "{v}"); + let v = pin_violation("csr", 0, 0, true).unwrap(); + assert!(v.contains("no expand ran on it"), "{v}"); +} + +#[test] +fn expects_expand_ignores_bound_edges_and_plain_bindings() { + let unbound = parse_query(TRAVERSAL_QUERY.trim_start_matches("--- query\n")).unwrap(); + assert!(expects_expand(&unbound.queries[0].match_clause)); + let bound = "query f($n: String) {\n match {\n $a: Person\n $a.name = $n\n \ + $a $k:knows $b\n }\n return { $b.name }\n}\n"; + let bound = parse_query(bound).unwrap(); + assert!(!expects_expand(&bound.queries[0].match_clause)); + let plain = parse_query(QUERY.trim_start_matches("--- query\n")).unwrap(); + assert!(!expects_expand(&plain.queries[0].match_clause)); + let negated = "query f() {\n match {\n $a: Person\n not { $a knows $x }\n }\n \ + return { $a.name }\n}\n"; + let negated = parse_query(negated).unwrap(); + assert!(!expects_expand(&negated.queries[0].match_clause)); +} + +/// A two-node, one-edge graph with a one-hop traversal, for the pin tests. +const TRAVERSAL_SCHEMA: &str = "--- schema\nnode Person {\n name: String @key\n}\n\n\ + edge Knows: Person -> Person {\n since: I64\n}\n"; +const TRAVERSAL_SEED: &str = "--- seed\n{\"type\":\"Person\",\"data\":{\"name\":\"alice\"}}\n\ + {\"type\":\"Person\",\"data\":{\"name\":\"bob\"}}\n\ + {\"edge\":\"Knows\",\"from\":\"alice\",\"to\":\"bob\",\"data\":{\"id\":\"k-1\",\"since\":2020}}\n"; +const TRAVERSAL_QUERY: &str = "--- query\nquery friends($n: String) {\n match {\n $a: Person\n \ + $a.name = $n\n $a knows $b\n }\n return { $b.name }\n}\n"; +const TRAVERSAL_PARAMS: &str = "--- params\n{\"n\": \"alice\"}\n"; +const TRAVERSAL_EXPECT: &str = "--- expect unordered\n{\"b.name\": \"bob\"}\n"; + +/// The pin reaches the executor on both paths: a pinned step runs its +/// expands on the pinned path only, and the probes see them (a zero count +/// on both paths would make the check vacuous). +#[tokio::test] +async fn pinned_step_runs_only_its_pinned_path() { + for (mode, expect_indexed) in [("indexed", true), ("csr", false)] { + let text = format!( + "{HDR}# traversal: {mode}\n{TRAVERSAL_SCHEMA}{TRAVERSAL_SEED}{TRAVERSAL_QUERY}{TRAVERSAL_PARAMS}{TRAVERSAL_EXPECT}" + ); + let case = parse_case("pinned", &text).unwrap(); + execute_case(&case, Path::new("unused.gqt"), false) + .await + .unwrap_or_else(|e| panic!("{mode}: {e}")); + + let (db, _uri, _dir) = open_case_store(&case).await.unwrap(); + let Some(Item::Step(Step::Query(step))) = case.items.first() else { + panic!("first item is the query step"); + }; + let params = build_params(step.params_raw.as_ref(), &step.ast_params, None).unwrap(); + let (outcome, counts) = under_traversal( + Some(mode), + db.query( + ReadTarget::branch("main"), + &step.source, + &step.name, + ¶ms, + ), + ) + .await; + outcome.unwrap(); + let counts = counts.unwrap(); + let (indexed, csr) = ( + counts.indexed.load(Ordering::Relaxed), + counts.csr.load(Ordering::Relaxed), + ); + if expect_indexed { + assert!( + indexed >= 1 && csr == 0, + "{mode}: indexed={indexed} csr={csr}" + ); + } else { + assert!( + csr >= 1 && indexed == 0, + "{mode}: indexed={indexed} csr={csr}" + ); + } + } +} + +#[test] +fn refuses_ordered_expect_on_an_rrf_led_order() { + let query = "--- query\nquery q($v: Vector(4), $t: String) {\n match { $p: Person }\n \ + return { $p.name }\n order { rrf(nearest($p.vec, $v), bm25($p.name, $t)) }\n}\n"; + let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect ordered\n"); + let message = refusal("x", &text); + assert!(message.contains("led by `rrf()`"), "{message}"); + let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect unordered\n"); + parse_case("x", &text).unwrap(); +} + +#[test] +fn refuses_ordered_expect_with_an_aggregate_in_return() { + let query = "--- query\nquery q($t: String) {\n match { $p: Person\n search($p.name, $t) }\n \ + return { count($p) as total }\n order { bm25($p.name, $t) }\n}\n"; + let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect ordered\n"); + assert!(refusal("x", &text).contains("aggregate in its `return` list")); + let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect unordered\n"); + parse_case("x", &text).unwrap(); +} + +#[tokio::test(flavor = "multi_thread")] +async fn bounded_fails_a_case_over_its_budget() { + let slow = run_bounded("slow", Duration::from_millis(50), async { + tokio::time::sleep(Duration::from_secs(30)).await; + Ok(()) + }) + .await; + assert_eq!(slow.stem, "slow"); + assert!( + slow.elapsed < Duration::from_secs(5), + "timeout did not cut the case short" + ); + let err = slow.result.as_ref().unwrap_err(); + assert!(err.contains("budget of 0.05s"), "{err}"); + assert!(err.contains(CASE_TIMEOUT_ENV), "{err}"); +} + +#[tokio::test(flavor = "multi_thread")] +async fn bounded_records_a_panicking_case() { + let out = run_bounded("p", Duration::from_secs(10), async { + if std::hint::black_box(true) { + panic!("boom"); + } + Ok(()) + }) + .await; + assert_eq!(out.stem, "p"); + let err = out.result.as_ref().unwrap_err(); + assert!(err.starts_with("case panicked: boom"), "{err}"); +} + +/// The checked-in corpus itself: at least one case, and no foreign entry (a +/// mis-renamed, nested, symlinked, or dot-prefixed case would otherwise +/// silently never run: the test target registers what its +/// `datatest_stable::harness!` pattern matches, and `list_cases` mirrors +/// that rule so this test refuses what the target would skip). +#[test] +fn corpus_layout() { + let root = corpus_root(); + let (files, foreign) = list_cases(&root); + assert!( + foreign.is_empty(), + "foreign entries under {}: {}", + root.display(), + foreign.join(", ") + ); + assert!( + !files.is_empty(), + "no .gqt cases found under {}; a broken checkout must never read as green", + root.display() + ); +} + +#[test] +fn runner_refuses_a_process_traversal_override() { + assert!(traversal_override_refusal(None).is_none()); + let reason = traversal_override_refusal(Some(OsStr::new("csr"))).unwrap(); + assert!(reason.contains("OMNIGRAPH_TRAVERSAL_MODE=csr"), "{reason}"); + assert!(reason.contains("# traversal:"), "{reason}"); +} + +/// Same name battery as `scripts/check-fix-regression.py --self-test` +/// (`corpus_case`): the runner and the gate must agree on what a case is. +/// The symlink and non-UTF-8 rows exist only here: the gate sees path +/// strings, and the test target's walk (`datatest-stable` over `walkdir`, +/// links not followed, names that are not UTF-8 dropped) would skip both. +#[test] +fn corpus_flags_foreign_entries() { + let dir = tempfile::tempdir().unwrap(); + std::fs::write(dir.path().join("a.gqt"), "x").unwrap(); + std::fs::write(dir.path().join("b.txt"), "x").unwrap(); + std::fs::write(dir.path().join(".hidden.gqt"), "x").unwrap(); + std::fs::write(dir.path().join(".DS_Store"), "x").unwrap(); + std::fs::write(dir.path().join("c.GQT"), "x").unwrap(); + std::fs::create_dir(dir.path().join("nested")).unwrap(); + std::fs::write(dir.path().join("nested").join("d.gqt"), "x").unwrap(); + let mut expected = vec![ + ".hidden.gqt".to_string(), + "b.txt".to_string(), + "c.GQT".to_string(), + "nested".to_string(), + ]; + #[cfg(unix)] + { + std::os::unix::fs::symlink("a.gqt", dir.path().join("link.gqt")).unwrap(); + expected.push("link.gqt".to_string()); + } + // APFS refuses a name that is not valid UTF-8 (EILSEQ), so this row runs + // where the file system takes it. + #[cfg(target_os = "linux")] + { + use std::os::unix::ffi::OsStrExt; + let bad = std::ffi::OsStr::from_bytes(b"bad\xff.gqt"); + std::fs::write(dir.path().join(bad), "x").unwrap(); + expected.push(bad.to_string_lossy().into_owned()); + } + expected.sort(); + let (files, foreign) = list_cases(dir.path()); + assert_eq!(files, vec![dir.path().join("a.gqt")]); + assert_eq!(foreign, expected); +} + +#[tokio::test] +async fn bless_refuses_cases_containing_loops() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("bless_loop_case.gqt"); + let text = format!( + "{HDR}{SCHEMA}{SEED}--- foreach $x a b\n{QUERY}--- expect unordered\n{{\"p.name\": \"nobody\"}}\n--- endloop\n" + ); + std::fs::write(&path, &text).unwrap(); + let case = parse_case("bless_loop_case", &text).unwrap(); + let err = execute_case(&case, &path, true).await.unwrap_err(); + assert!(err.contains("bless: refused"), "got: {err}"); + assert_eq!(std::fs::read_to_string(&path).unwrap(), text); +} + +#[tokio::test] +async fn bless_never_rewrites_on_a_kind_mismatch() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("bless_kind_case.gqt"); + let query = "--- query\nquery q() {\n match { $p: Person }\n return { $p.nope }\n}\n"; + let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect unordered\n{{\"p.nope\": \"x\"}}\n"); + std::fs::write(&path, &text).unwrap(); + let case = parse_case("bless_kind_case", &text).unwrap(); + let err = execute_case(&case, &path, true).await.unwrap_err(); + assert!(err.contains("query failed"), "got: {err}"); + assert_eq!(std::fs::read_to_string(&path).unwrap(), text); +} + +#[tokio::test] +async fn params_refusal_satisfies_an_error_expect() { + let query = "--- query\nquery q($q: String) {\n match {\n $p: Person\n $p.name = $q\n }\n return { $p.name }\n}\n"; + let text = format!("{HDR}{SCHEMA}{SEED}{query}--- expect error: q\n"); + let case = parse_case("x", &text).unwrap(); + execute_case(&case, Path::new("unused.gqt"), false) + .await + .unwrap(); +} + +#[test] +fn bless_splice_preserves_the_trailing_blank_separator() { + let original = "--- expect unordered\nold\n\n--- restart\n"; + let span = BodySpan { + start_line: 1, + len: 2, + }; + let rows = vec!["{\"n\":1}".to_string()]; + let out = splice_lines(original, span, &rows); + assert_eq!(out, "--- expect unordered\n{\"n\":1}\n\n--- restart\n"); +} + +#[test] +fn bless_splice_replaces_only_the_expect_body() { + let original = "--- query\nq\n--- expect unordered\nold row\nold row 2\n--- restart\n"; + let span = BodySpan { + start_line: 3, + len: 2, + }; + let rows = vec!["{\"n\":1}".to_string()]; + let out = splice_lines(original, span, &rows); + assert_eq!( + out, + "--- query\nq\n--- expect unordered\n{\"n\":1}\n--- restart\n" + ); +} + +#[test] +fn bless_splice_inserts_into_an_empty_expect_body() { + let original = "--- expect unordered\n--- restart\n"; + let span = BodySpan { + start_line: 1, + len: 0, + }; + let rows = vec!["{\"n\":1}".to_string()]; + let out = splice_lines(original, span, &rows); + assert_eq!(out, "--- expect unordered\n{\"n\":1}\n--- restart\n"); +} diff --git a/crates/omnigraph-gqt/tests/gq_logic_tests.rs b/crates/omnigraph-gqt/tests/gq_logic_tests.rs new file mode 100644 index 000000000..f483e2c16 --- /dev/null +++ b/crates/omnigraph-gqt/tests/gq_logic_tests.rs @@ -0,0 +1,78 @@ +//! One libtest test per `cases/*.gqt`, registered at run time by +//! `datatest-stable` (`harness = false` in `Cargo.toml`), so +//! `cargo test -p omnigraph-gqt ` runs the matching cases, +//! `-- --list` names them all, and `--test-threads` sets the concurrency. +//! The runner, format, and self-tests live in `src/lib.rs`; the corpus +//! layout check (no foreign entries, never empty) is the `corpus_layout` +//! unit test there. +// The `Send`/`Sync` walk of the spawned case future (engine query futures +// inside) overflows the default recursion limit; `src/lib.rs` raises it for +// the same reason. +#![recursion_limit = "512"] + +use std::path::Path; +use std::sync::OnceLock; + +use omnigraph_gqt::{ + bless_from_env, case_budget_from_env, run_case_bounded, traversal_override_refusal, +}; +use tokio::runtime::Runtime; + +/// Worker stack for the case runtime: the engine's query futures overflow +/// the 2 MiB default (CI sets `RUST_MIN_STACK` to this same value for the +/// engine test jobs; pinning it here keeps this target's cases independent +/// of it; the crate's unit tests run on libtest's own threads). +const WORKER_STACK_BYTES: usize = 16 * 1024 * 1024; + +/// One multi-thread runtime shared by every case; libtest calls `case` +/// from its own worker threads, each call spawns its case onto the runtime +/// and blocks on the join handle. +fn runtime() -> &'static Runtime { + static RT: OnceLock = OnceLock::new(); + RT.get_or_init(|| { + tokio::runtime::Builder::new_multi_thread() + .enable_all() + .thread_stack_size(WORKER_STACK_BYTES) + .build() + .expect("tokio runtime") + }) +} + +fn case(path: &Path) -> datatest_stable::Result<()> { + if let Some(reason) = + traversal_override_refusal(std::env::var_os("OMNIGRAPH_TRAVERSAL_MODE").as_deref()) + { + return Err(reason.into()); + } + let rt = runtime(); + let task = rt.spawn(run_case_bounded( + path.to_path_buf(), + case_budget_from_env(), + bless_from_env(), + )); + // `run_case_bounded` catches the case's own panics; the task around it + // holds nothing that can panic. + let outcome = rt.block_on(task).expect("a case task never panics"); + let secs = outcome.elapsed.as_secs_f64(); + match outcome.result { + Ok(()) => { + println!("ok {} {secs:.2}s", outcome.stem); + Ok(()) + } + Err(detail) => { + // The detail (row diff, refusal, panic text, budget overrun) goes + // to stdout under the FAIL line: the harness renders a returned + // error Debug-escaped on one line, which hides a multi-line diff. + println!( + "FAIL {} {secs:.2}s\n {}", + outcome.stem, + detail.replace('\n', "\n ") + ); + Err(format!("{}: see its FAIL block above", outcome.stem).into()) + } + } +} + +datatest_stable::harness! { + { test = case, root = "cases", pattern = r"^[^./][^/]*\.gqt$" }, +} diff --git a/docs/dev/testing.md b/docs/dev/testing.md index 5c24c5568..0c8f72ffc 100644 --- a/docs/dev/testing.md +++ b/docs/dev/testing.md @@ -25,6 +25,7 @@ The invariants behind these rules are in [invariants.md](invariants.md). Lance-d | `omnigraph-cli` | `crates/omnigraph-cli/tests/` | `tests/support/mod.rs` | | `omnigraph-dst` | `crates/omnigraph-dst/tests/` (`scenarios.rs`, `lane_b.rs`, `torn_init.rs`) plus in-source proofs | Crate-local fixtures. Deterministic simulation; needs `--cfg tokio_unstable` (the crate-local `.cargo/config.toml` sets it when cargo runs from the crate dir; every test file is `#![cfg(tokio_unstable)]`-gated and the crate compiles empty without it, so the default workspace gate is unaffected). `#[ignore]`d tests are fleet/hunt instruments driven by the DST workflows | | `omnigraph-bench` | In-source configuration tests and `crates/omnigraph-bench/tests/` | Checked-in cases and suites under `benchmarks/` | +| `omnigraph-gqt` | `tests/gq_logic_tests.rs`, one libtest test per `.gqt` case (`datatest-stable`, `harness = false`), plus in-source format self-tests and the corpus layout check | The `.gqt` corpus under `crates/omnigraph-gqt/cases/`; format in RFC 0045 | Do not copy server or CLI process setup into a new suite. Their support modules own hermetic configuration, binary startup, temporary roots, and common assertions. @@ -35,7 +36,7 @@ The engine integration suite is grouped by behavior, not implementation module: | Concern | Existing owners | |---|---| | Initialization and representative journeys | `lifecycle.rs`, `end_to_end.rs`, `composite_flow.rs`, `consistency.rs` | -| Query results and operators | `aggregation.rs`, `literal_filters.rs`, `ordering.rs`, `traversal.rs`, `traversal_indexed.rs`, `proptest_equivalence.rs`, `gq_logic_tests.rs` (walks the `.gqt` cases under `tests/gq_logic_tests/`) | +| Query results and operators | `aggregation.rs`, `literal_filters.rs`, `ordering.rs`, `traversal.rs`, `traversal_indexed.rs`, `proptest_equivalence.rs`; the `.gqt` corpus lives in `omnigraph-gqt` (`crates/omnigraph-gqt/cases/`) | | Search and physical indexes | `search.rs`, `scalar_indexes.rs`, `lance_surface_guards.rs`, `rrf_prefilter_gate.rs` (the rrf plan gate's differential oracle and fences), `repro_issue_563.rs` (`#[ignore]`d overflow-scale symptom tier) | | Writes, validation, schema, and policy | `writes.rs`, `validators.rs`, `schema_apply.rs`, `policy_engine_chassis.rs` | | Branches, snapshots, diffs, and merges | `branching.rs`, `point_in_time.rs`, `changes.rs`, `merge_truth_table.rs`, `merge_fast_forward.rs` | @@ -96,22 +97,29 @@ Focused iteration: ```bash cargo test -p omnigraph-engine --test traversal cargo test -p omnigraph-engine --test writes concurrent -cargo test -p omnigraph-engine --test gq_logic_tests +cargo test -p omnigraph-gqt # every .gqt case + the format self-tests +cargo test -p omnigraph-gqt --test gq_logic_tests issue_563 # the cases whose file name contains issue_563 +cargo test -p omnigraph-gqt --test gq_logic_tests -- --list # one line per case cargo test -p omnigraph-server --test data_routes cargo test -p omnigraph-cli --test cli_data cargo test -p omnigraph-cluster --test failpoints --features failpoints cargo test -p omnigraph-bench --locked ``` -`OMNIGRAPH_GQ_LOGIC_TESTS=[,]` restricts the gq logic-test run -to case files whose name contains a value; `OMNIGRAPH_GQ_BLESS=1` rewrites the -failing step's expect rows in place (local workflow only, never CI). The walker -keeps at most `OMNIGRAPH_GQ_JOBS=` cases in flight (default: the machine's -available parallelism) and fails a case that exceeds -`OMNIGRAPH_GQ_CASE_TIMEOUT_SECS=` seconds (default 10); every `ok`/`FAIL` -line carries the case's elapsed time, and a case over budget belongs in a -`heavy-repro:` `#[ignore]`d test under `tests/repro_issue_*.rs`, not the -corpus. +Every `.gqt` case is its own libtest test named `case::.gqt`, registered +at run time (`datatest-stable`), so the ordinary name filter selects cases, a +case-only pull request needs no Rust change, and `cargo-nextest` sees each +case (an IDE's test-results view lists cases from the +libtest-shaped output; no per-case gutter runnable exists, since no source +item does). `--test-threads=` bounds how many cases run +concurrently (default: the machine's available parallelism); each case fails +if it exceeds `OMNIGRAPH_GQ_CASE_TIMEOUT_SECS=` seconds (default 10); +`OMNIGRAPH_GQ_BLESS=1` rewrites the failing step's expect rows in place (local +workflow only, never CI). Every `ok`/`FAIL` line carries the case's elapsed +time, and a case over budget belongs in a `heavy-repro:` `#[ignore]`d test +under `crates/omnigraph/tests/repro_issue_*.rs`, not the corpus. A name filter +that matches no case is libtest's ordinary green zero-test run; read the +`filtered out` count. Canonical workspace graph: diff --git a/docs/rfcs/0045-gq-logic-tests.md b/docs/rfcs/0045-gq-logic-tests.md index 777ec5af1..496ea81ad 100644 --- a/docs/rfcs/0045-gq-logic-tests.md +++ b/docs/rfcs/0045-gq-logic-tests.md @@ -7,7 +7,7 @@ implementation: partial authors: - azimafroozeh created: 2026-08-29 -updated: 2026-09-02 +updated: 2026-09-03 discussion: https://github.com/ModernRelay/omnigraph/pull/584 supersedes: [] superseded_by: [] @@ -28,10 +28,16 @@ cases (cases with no issue anchor) for new or existing behavior alike. A holding a `.pg` schema, seed rows as JSONL, and one or more ***steps***: a read query or a mutation, each with its params and its expected outcome (rows, affected counts, or an error message); a case may also restart the -store between steps and repeat a step group over a value list. One test -target, `crates/omnigraph/tests/gq_logic_tests.rs`, walks -`tests/gq_logic_tests/*.gqt` and runs each case against a fresh temporary -store: init, load, index, then the steps in order. +store between steps and repeat a step group over a value list. A +dedicated workspace crate, `omnigraph-gqt` (`publish = false`, outside +`default-members`, and outside the explicit `-p` list `release.yml` +builds, so never in a release), holds the corpus and the runner: its one +integration-test target, `crates/omnigraph-gqt/tests/gq_logic_tests.rs`, +registers every top-level, non-dot-prefixed +`crates/omnigraph-gqt/cases/*.gqt` file (any other entry except an +extension-less dot-file fails `corpus_layout`, Runner mechanics) as its own libtest-compatible test and +runs each case against a fresh temporary store: init, load, index, then +the steps in order. Around the harness sits an enforcement ladder: AGENTS.md contract sentences making logic tests the default medium and holding every issue @@ -75,7 +81,7 @@ seed material for the DST generators. Authoring a regression for a fixed issue: -1. Write `tests/gq_logic_tests/issue_NNN_short_name.gqt`. +1. Write `crates/omnigraph-gqt/cases/issue_NNN_short_name.gqt`. 2. Run it on the unfixed build and watch it fail; record what failed in the `# red_on:` header line, mandatory for issue-anchored cases (a logic test nobody watched fail guards nothing). @@ -89,10 +95,12 @@ the case did witness a red state during development. Running: ```bash -cargo test -p omnigraph-engine --test gq_logic_tests +cargo test -p omnigraph-gqt ``` -The target prints one line per case with its elapsed time +The crate is not a workspace default member, so a bare `cargo test` at +the root skips it; `-p omnigraph-gqt` or `--workspace` reaches it. The +target prints one line per case with its elapsed time (`ok issue_563_aggregate_uncapped 0.12s` / `FAIL issue_563_underfill_retry 0.09s`) and fails at the end with the list of failing cases and, per failure, the failing step named by ordinal and kind @@ -103,12 +111,14 @@ would run against a store state the failed step no longer vouches for); across cases the target runs every case before failing, so one broken case never hides another. A file the harness refuses (any fail-closed check in the Design section) reports as a failing case carrying the -refusal message, and the walk continues. The lines reach the terminal -under `--nocapture`; a plain -`cargo test` shows them on failure. `OMNIGRAPH_GQ_LOGIC_TESTS=issue_563` -restricts the run to cases whose file name contains the value, and a -comma-separated list selects the union (the `OMNIGRAPH_DST_SEEDS` -precedent). +refusal message, and the remaining cases still run. The per-case lines +print on every run: the target's harness never captures output, so +`--nocapture` changes nothing for it (the crate's unit tests, under real +libtest, still honor it). Every case is its own libtest-compatible test, +named `case::.gqt`, so +`cargo test -p omnigraph-gqt --test gq_logic_tests issue_563` +restricts the run to cases whose file name contains the argument, and +`-- --list` names every registered case. ***Bless mode*** (the update-in-place workflow rustc calls `--bless` and expect-test drives with `UPDATE_EXPECT=1`): `OMNIGRAPH_GQ_BLESS=1` @@ -133,17 +143,20 @@ CI: the workspace `Test Workspace` job picks the target up automatically but does not run on pull requests (`ci.yml`); regressions must be exercised before merge, so a per-change job (`GQ Logic Tests`) in a new workflow file, `.github/workflows/gq-logic-tests.yml`, runs -`cargo test -p omnigraph-engine --test gq_logic_tests --locked -- --nocapture` +`cargo test -p omnigraph-gqt --locked -- --nocapture` on every PR. The workflow triggers on push to `main`, `workflow_dispatch`, -and `pull_request` with its types declared explicitly (`opened`, -`synchronize`, `reopened`, `edited`, `labeled`, `unlabeled`), because the -gate job below must re-run when a body edit or the waiver label changes -its answer. The types list is workflow-level, so label and body-edit -events also re-run the test job, and that job compiles the engine crate -and the test binary: minutes with a warm cache, tens of minutes cold. The -test job honors the docs-only classification that `ci.yml`'s -`Classify Changes` job defines, the way the other required Rust jobs do: -on a docs-only PR it skips its build and reports success (the +and `pull_request` with only its code-bearing types declared (`opened`, +`synchronize`, `reopened`); the gate below lives in its own workflow and +declares the body-edit and label types itself, so a body edit or a label +change never re-runs the Rust build. The test job compiles the engine +crate, the `omnigraph-gqt` library, and its two test binaries: minutes +with a warm cache, tens of minutes cold. The test job honors the +docs-only classification of `ci.yml`'s `Classify Changes` job, the way +the other required Rust jobs do, through a verbatim copy of that job +carried in its own workflow (`Classify Changes (GQ Logic Tests)`; GitHub +Actions cannot make a job depend on another workflow's job, and +`scripts/check-classify-copy.py` refuses drift from `ci.yml`): on a +docs-only PR it skips its build and reports success (the `Test omnigraph-server --features aws` job's pattern), so the required context never stays pending. The gate job always runs, at seconds-scale. Action references are pinned, per the @@ -151,39 +164,59 @@ repo's workflow-pin check; no failpoints features are needed. What a green `GQ Logic Tests` job promises, quotable: every `.gqt` case parsed, ran, and matched its expects, none refused. -Fix-PR gate: a required CI check (`Fix Regression Gate`, a second job in -the same workflow, running only on `pull_request` events since a push or -dispatch run has no PR body to read; a skipped context does not block) -reads the PR body for GitHub's closing keywords (close, closes, closed, +Fix-PR gate: a required CI check (`Fix Regression Gate`, a job in its +own workflow, `.github/workflows/fix-regression-gate.yml`, on +`pull_request_target`, which takes the workflow file and the gate script +from the base branch and fetches the head only as data for the diff +range, so a pull request cannot weaken the copy that runs; the workflow +has no push or dispatch trigger, since those runs have no PR body to +read) reads the PR body for GitHub's closing keywords (close, closes, closed, fix, fixes, fixed, resolve, resolves, resolved), matched the way GitHub's own parser matches them: case-insensitive, a word boundary before the keyword (so "hotfix #563" never fires on `fix`), an optional colon, then whitespace (optional when the colon is present) and `#N`, with leading zeros in `N` normalized away. Every issue so closed needs a matching -addition in the diff (the two owner locations `docs/dev/testing.md` -lists per package: `tests/` targets and in-source test modules), checked -independently per issue: an added `.gqt` case in the logic test corpus -whose file name carries `issue_N`, an added `# issue: N` header line in a -corpus case, or an added Rust line defining a function whose name carries -`issue_N`, where `N` must be followed by a non-digit or by the end of the -line or path (`issue_5630` never matches issue 563; `issue_563_underfill` -does). The Rust shape matches only in top-level test targets, -`crates/*/tests/.rs`, and in-source modules, `crates/*/src/**`; -helper and fixture modules under `tests/` never match (a helper named for -an issue is not a test). A Rust definition is skipped when its name starts -with `_`, when the line is a declaration ending in `;`, or when the same -name is removed elsewhere in the diff (a rename, not an addition). -A comment, string, or fixture line mentioning the issue never satisfies -the gate. The gate is a diff check, and what it guarantees about -execution differs by shape: a corpus shape executes, since the walker -runs every `.gqt` in the directory, refuses a malformed one, and the -`GQ Logic Tests` job is required; a Rust shape is a naming check only, -since no Rust test target other than `gq_logic_tests` runs on a pull -request (`Test Workspace` runs post-merge, CI above), and a defined -function can besides be `#[ignore]`d or cfg-gated, so whether that test -is registered, runs in the suite, and asserts the right thing stays with -review, which the first AGENTS.md sentence primes: a Rust test needs a -reason the format cannot express. The gate reads only that form in the +regression in the diff, added or strengthened (the two owner locations +`docs/dev/testing.md` lists per package: `tests/` targets and in-source +test modules), checked independently per issue: a `.gqt` case in the +logic test corpus named `issue_N_*`, new or modified with +at least one added body line (not a `#` header line or a `//` comment), +or a Rust function whose name carries `issue_N`, either added with an +added `#[test]` or `#[::test]` attribute line +(`#[tokio::test(...)]` included) directly above it in the same hunk, or +existing, test-attributed, and given an added line carrying an +alphanumeric character, not a comment or an attribute, inside its body; +`N` must be followed by a non-digit or by the end of the line (Rust +shape) (`issue_5630` never matches issue 563; +`issue_563_underfill` does). The Rust shape matches only in top-level +test targets, `crates/*/tests/.rs` and `tools/*/tests/.rs`, +and in-source modules, `crates/*/src/**` and `tools/*/src/**`; helper and +fixture modules under `tests/` never match (a helper named for an issue +is not a test). A plain function, however named, never satisfies the +gate; an added definition is skipped when its name starts with `_`, when +the line is a declaration ending in `;`, or when the same name is +removed elsewhere in the diff (a rename alone never counts, a rename +plus an added assertion does). A comment, string, or fixture line +mentioning the issue never satisfies the gate, and owners the gate does +not recognize (Python and shell scripts among them) satisfy it only +through `no-repro`. Adjacency and body-location rules, with their named +residues, are in the Decision log (2026-09-02). The gate is a diff check, and what it guarantees about +execution differs by shape: a corpus shape executes, since the target +registers and runs every `.gqt` in the corpus, refuses a malformed one, and the +`GQ Logic Tests` job is required; a Rust shape is a test-attributed +definition, not a run, since, among Rust test targets, only +`omnigraph-gqt`'s (the corpus target and its unit tests), +`Test omnigraph-server --features aws`, and `DST pinned suite` +(`cargo test -p omnigraph-dst`, `dst.yml`) run on a pull request +(`Test Workspace` runs post-merge, CI above); a test-attributed `issue_N` +function inside `crates/omnigraph-gqt/`, `crates/omnigraph-server/`, or +`crates/omnigraph-dst/` therefore does run, and the +Rust shape stays a naming check everywhere else, where a defined +function can besides be `#[ignore]`d or cfg-gated (workspace clippy on +the pull request refuses an unreferenced private function, not those), +so whether that test runs in the suite and asserts the right thing stays +with review, which the first AGENTS.md sentence primes: a Rust test needs +a reason the format cannot express. The gate reads only that form in the PR body: closings by full URL, `owner/repo#N` reference, a bare no-space `fixes#N`, commit-message keyword, or manual close after merge pass unexamined; that @@ -199,12 +232,12 @@ deleted. The gate's guarantee, quotable: exit 0 exactly when every issue the body closes by keyword has its matching addition or the PR carries `no-repro`, and the AGENTS.md contract sentence is present (the grep in the Enforcement ladder); a corpus match means the case ran green in the -required job, and a Rust match means only that a function of that name -was added. The guarantee holds over the PR's own tree and the labels -triage rights control: the check runs the PR's own copy of the script -and workflow file, and labels reach it comma-joined, so a label name -containing a comma could carry the waiver token; creating a label needs -the same triage rights as applying one, and that residue is accepted. +required job, and a Rust match means a test-attributed function of that +name was added or extended. The guarantee holds over the base branch's +copy of the script and workflow file and the labels triage rights +control: labels reach the check comma-joined, so a label name containing +a comma could carry the waiver token; creating a label needs the same +triage rights as applying one, and that residue is accepted. ## Design @@ -298,10 +331,14 @@ the header); `//` comments inside query and mutate sections are simply GQ text. Header lines are `#` lines before the first section, keys `# issue:`, `# red_on:`, `# notes:`, `# traversal:`. `# issue:` is always required; `# red_on:` is required when `# issue:` names a number and -optional under `# issue: none`; a `#` line not starting a key continues -the previous entry (a first header line starting no key is refused); any -other `# :` key is refused; a key given twice is refused; `# notes:` -and `# traversal:` are optional. `# issue:` takes a number in canonical +optional under `# issue: none`; a header line is accepted exactly when +it equals `# : ` byte for byte, for one of the four keys in +that spelling and a value with no leading or trailing whitespace, and +every other non-blank line is refused (a stray space is answered with the +canonical line, a bad shape with the grammar, an unknown key with the key +list; no line ever continues a previous entry); a key given twice is +refused, except `# notes:`, which repeats to carry a multi-line note; +`# notes:` and `# traversal:` are optional. `# issue:` takes a number in canonical spelling (no sign, no leading zeros) or `none`; any other spelling is refused. `# traversal:` takes `indexed` or `csr` and pins every step to that mode, for cases whose subject is one traversal path (Execution @@ -429,7 +466,8 @@ index step runs for it regardless of the constructs the steps use, so the pinned path runs covered rather than on a fallback. The trade-off in one sentence: the default corpus exercises the shipped path, and a pin reproduces a mode-specific defect. The `OMNIGRAPH_TRAVERSAL_MODE` process -variable never reaches a logic test either way. The seam is task-local +variable would reach an unpinned case, so every case fails with a +refusal naming it while it is set. The seam is task-local and scope-bound, so concurrent cases never interfere; it is public today, used by `tests/proptest_equivalence.rs` and `tests/traversal_indexed.rs`, which keep owning the engine's @@ -497,14 +535,18 @@ expect section is JSONL, one object per row, same keys. authoring rule, since no expect mode can absorb a varying set); an operation that later stops being deterministic is refused by name in the harness, the way `@embed` is today. Second, given that set, the - engine's order is total: an `order` clause qualifies when the source - batch carries `.id` columns, which is every non-aggregate query, - since `apply_ordering` appends each bound variable's `.id` column - as an ascending tie-break (plain orderings and `nearest`/`bm25`-led - orderings alike); an aggregate result batch carries no `.id` - column, so group rows tied on the sort key keep first-seen order. The - harness checks the parsed declaration and refuses `ordered` where the - second condition fails: no `order` clause; an `order` clause led by + engine's order is total, which is an authoring rule: the `order` keys + must be total over the rows the step returns. The `.id` tie-break + `apply_ordering` appends to every non-aggregate ordering is an + implementation detail no expect may depend on (it is the `@key` value + for keyed node types and a per-load ULID otherwise, so an unkeyed + type's order changes across runs), an aggregate result batch carries no + `.id` column at all, so group rows tied on the sort key have no + guaranteed order, and a tie on the sort keys surfaces as flakiness the + harness cannot see statically (`ordered_two_key_sort.gqt` is the corpus + example). The harness checks the parsed declaration and refuses + `ordered` where no total order is possible: no `order` clause; an + `order` clause led by `rrf()`, whose fusion sorts by score alone; and any aggregate in the `return` list (an `Aggregate` expression, the engine's own `projections_have_aggregates` definition). One authoring rule follows @@ -532,34 +574,52 @@ practice): assert the resulting row order, never the score values. ### Runner mechanics -The walker is a single `#[tokio::test(flavor = "multi_thread")]` entry -point (tokio is already every engine integration test's runtime) that -lists `tests/gq_logic_tests/*.gqt` rooted at `CARGO_MANIFEST_DIR` -(the `forbidden_apis.rs` walk precedent) and spawns each case as a task -into a `tokio::task::JoinSet`. Case concurrency comes from that task set, -not from libtest: to libtest the whole walker is one test. Case execution -is bounded: each case opens its own store and may build its indexes, so -the number of cases in flight at once is capped independently of corpus -size; the cap is a walker implementation detail, not format contract. +The test target is `harness = false` and hands discovery to +`datatest-stable`: every `cases/*.gqt` file, rooted at the crate, is +registered at run time as its own libtest-compatible test (a libtest-mimic +trial under `datatest-stable`) named `case::.gqt`. The runner it +calls (parser, execution, comparison, bless) is the crate's library, +`crates/omnigraph-gqt/src/lib.rs`, and the format self-tests are unit +tests beside it in `crates/omnigraph-gqt/src/tests.rs`; the crate is +`publish = false` and never built for release. Cases run on one shared +multi-thread tokio runtime whose worker stacks are 16 MiB (the engine's +query futures overflow the 2 MiB default; the value equals the CI jobs' +`RUST_MIN_STACK`, so the harness target does not depend on that +variable; tokio is already every engine integration test's runtime). +Case concurrency is libtest-mimic's, not the target's: each case opens its +own store and may build its indexes, so the number of cases in flight at +once is set by libtest's `--test-threads=` flag, which +`datatest-stable` honors, independently of corpus size; that flag is a +runner knob, not format contract. The per-PR corpus is the ***fast tier***: a case is expected to finish -in well under a second. The walker's per-case budget defaults to 10 +in well under a second. The runner's per-case budget defaults to 10 seconds, generous against that expectation so a slow CI runner never -trips it; an environment variable overrides the default, and the timeout -failure message prints the budget in force. The elapsed time on each +trips it; `OMNIGRAPH_GQ_CASE_TIMEOUT_SECS` overrides the default, and the +timeout failure message prints the budget in force. The elapsed time on each ok/FAIL line, not the timeout, is the drift signal a reviewer reads. A -case that trips the budget belongs to the nightly tier defined in the -Enforcement ladder below, so slowness fails the PR introducing it instead -of accumulating in the required job. Each case's -outcome, a panic included, is caught and recorded (the `JoinSet` surfaces -task panics as join errors), which lets the target run every case before -failing. The walker fails when its glob matches no files (a broken -checkout or a bad rename, never a green run) and when the corpus -directory holds any entry that is not a `.gqt` file (a mis-renamed or -nested case must never silently skip). Zero new dependencies and -an ordinary libtest harness, so the workspace invocation (including its -`-- --nocapture`) is -untouched; per-file test identity via libtest-mimic is the recorded -upgrade path (Alternatives). +case that trips the budget belongs to the heavy-repro tier, defined below +in the Enforcement ladder, so slowness fails the PR introducing it +instead of accumulating in the required job. Each case's +outcome, a panic included, is caught and reported as that case's own +failure, which lets the target run every case before +failing. A corpus directory holding no case file makes the target panic +at startup with `no test cases found for test 'case'`, +`datatest-stable`'s own refusal, before any name filter runs (a broken +checkout or a bad rename, never a green run; `--exact` excepted: it +resolves the one name without scanning); a name filter matching +nothing runs zero tests and exits green, libtest's own behavior, where +the merged selector failed on an unmatched value. The `corpus_layout` +unit test fails on an empty corpus and on any entry that is not a +top-level regular `.gqt` file with a UTF-8 name (a symlink is foreign), +dot-prefixed `.gqt` names included; dot-prefixed +entries without the extension (`.DS_Store`, `.gitkeep`) are skipped (a +mis-renamed, nested, or dot-prefixed case must never silently skip). One +new dev-dependency, `datatest-stable` (bringing `libtest-mimic`, +`fancy-regex`, `camino`, `escape8259` into the lockfile), which takes +libtest's own arguments, so the workspace's `-- --nocapture` is accepted +as before (inert for this target); per-file test identity, the upgrade +path the merged design recorded (Decision log, 2026-09-03), is thereby +taken. ### Enforcement ladder @@ -570,7 +630,7 @@ the same PR to defer to the second sentence, so the fix-carries-regression rule keeps one phrasing: > Query-behavior tests default to `.gqt` logic tests under -> `crates/omnigraph/tests/gq_logic_tests/`; a Rust test needs a reason the +> `crates/omnigraph-gqt/cases/`; a Rust test needs a reason the > logic test format cannot express (mechanism assertions, scale symptoms, > process environment, concurrency). @@ -617,11 +677,11 @@ today; they are renamed to `heavy-repro:` in the PR that lands the nightly job (Rollout). The CI gate check and the `no-repro` waiver close the ladder (behavior in -the previous section). The gate is `scripts/check-fix-regression.py`, a +User and operational behavior). The gate is `scripts/check-fix-regression.py`, a Python script beside `check-agents-md.sh`, run by the `Fix Regression Gate` job after its own self-test. The gate script also asserts the first contract sentence is still present in `AGENTS.md` (a literal grep for the -corpus path `crates/omnigraph/tests/gq_logic_tests/`), so deleting the +corpus path `crates/omnigraph-gqt/cases/`), so deleting the contract without deleting the gate fails closed. The ladder's recurring human costs are named and accepted: maintainers apply `no-repro` and adjudicate when an author believes no repro is possible; reviewers own the @@ -636,17 +696,20 @@ engine surface the harness calls is public and chokepoint-registered in the `forbidden_apis.rs` const registries: `query`, `query_with_head`, and `run_query_at` read-only, `load_jsonl` / `load_jsonl_file` under `LOAD_V9`, the `mutate` family under `MUTATION_V9`, and `open` / `open_with_storage` -under the `RecoveryExecutor` write protocol. That walker covers -`crates/omnigraph/src/**` only, so the test target itself adds no registry -entries; no deny-list item is affected, and no new public API is added. +under the `RecoveryExecutor` write protocol. That `forbidden_apis.rs` +walk covers `crates/omnigraph/src/**` only, so the `omnigraph-gqt` crate +adds no registry entries; no deny-list item is affected, and no new +public API is added. The target adds no shared state to the test suite (Execution semantics). ## Compatibility and reversibility No storage or wire surface changes; the RFC is purely additive to tests, -CI, and contributor docs. Reverting means deleting the logic test -directory, the test target, the two CI jobs, and the AGENTS.md sentences; -the logic test files remain readable, self-contained behavior records +CI, and contributor docs. Reverting means deleting the `omnigraph-gqt` +crate and its `members` entry in the workspace `Cargo.toml` (which drops +`datatest-stable` and its lockfile closure), the two workflow files with +their scripts, and the +AGENTS.md sentences; the logic test files remain readable, self-contained behavior records either way. Format evolution is fail-closed: unknown sections, unknown header keys, and missing required headers are refusals, never silent skips, so an older harness refuses a newer logic test rather than @@ -667,7 +730,6 @@ forces the question: bound and timeout. Out of v1; the heavy-repro tier stays Rust until a case that trips the fast-tier budget has a reason to stay in the format. -- Per-file test identity via libtest-mimic (Alternatives). - Run-twice verification: each query step re-run under a second execution configuration (a pinned traversal mode, index absence, or a forced canonical execution) with the row sets compared. Run-twice is @@ -697,10 +759,11 @@ permanently rather than deferred (DST's domain, per Execution semantics). filtered `cargo test --workspace -- issue_N` run beside the diff check, passing only when at least one matching test ran and passed): closes the Rust-shape gap above. Corpus shapes need no such gate: the required - job proves them, and `OMNIGRAPH_GQ_LOGIC_TESTS=issue_N` selects one + job proves them, and + `cargo test -p omnigraph-gqt --test gq_logic_tests issue_N` selects one locally. Two costs: the gate re-runs on PR-body edits and label events, and where the diff check costs seconds, a filtered run compiles every - crate's test targets (`--workspace`), not the one engine target the + crate's test targets (`--workspace`), not the `omnigraph-gqt` crate the test job already builds; and libtest's substring filter has no word boundary, so `issue_563` also selects `issue_5630`. Deferred as the upgrade path, taken if review ever finds a named regression that never @@ -721,13 +784,13 @@ permanently rather than deferred (DST's domain, per Execution semantics). - **insta snapshot testing:** splits the query and its expectation across files and moves review into a bespoke tool; in-place expectations with git diff as the review gate preserve red-first provenance better. -- **libtest-mimic per-file tests:** one `Trial` per logic test gives real - test identity (`cargo test -p omnigraph-engine issue_563` selects one - case) and cargo-nextest compatibility, but costs `harness = false` and - changes how the workspace's `-- --nocapture` flag lands (compatibility - unverified). Deferred, not rejected: the env-var filter covers - selection, and the walker can swap to Trials without touching the - format. +- **libtest-mimic per-file tests:** taken, in the `datatest-stable` form + (Decision log). One libtest test per logic test gives real + test identity + (`cargo test -p omnigraph-gqt --test gq_logic_tests issue_563` selects + one case) and cargo-nextest compatibility; it costs `harness = false`, + but `datatest-stable` accepts libtest's arguments, so the workspace's + `-- --nocapture` is accepted as before (inert for this target). ## Evidence and tests @@ -771,22 +834,23 @@ Harness self-tests pin every refusal this RFC specifies (the File format, Execution semantics, and Bless mode sections), one test per refusal. -Docs follow the testing map: the harness joins the "Query results and -operators" row of the engine ownership table in `docs/dev/testing.md`, plus a -focused-iteration command in its Commands section (`check-agents-md.sh` keeps +Docs follow the testing map: `omnigraph-gqt` has its own row in the crate +table of `docs/dev/testing.md`, the "Query results and operators" row of the +engine ownership table points at it, and its Commands section carries the +whole-corpus, one-case, and `--list` invocations (`check-agents-md.sh` keeps the docs indexes honest, so no new orphan doc file). ## Rollout -1. **The implementation PR** (one PR, merged only after this RFC is - accepted): the `gq_logic_tests` test target with the full format - (steps, loops, restart), the bounded walker with its per-case budget - and elapsed-time lines, the five cases above, the self-tests, the +1. **The implementation PR** (one PR, shipped as #596): the + `gq_logic_tests` test target with the full format + (steps, loops, restart), the bounded runner with its per-case budget + and elapsed-time lines, the cases above, the self-tests, the three AGENTS.md sentences (logic-tests-by-default, regression per fix, - `#[ignore]` species-in-message), the gate script, the workflow with - both jobs (the test job honoring the docs-only classification, the - gate job on `pull_request` events), and the docs (`docs/dev/testing.md` - rows, `docs/dev/ci.md`). Requiredness is wired the way this repo wires + `#[ignore]` species-in-message), the gate script, the two workflows + (the test workflow carrying its classification copy, the gate workflow + on `pull_request_target`; Decision log 2026-09-02), and the docs + (`docs/dev/testing.md` rows, `docs/dev/ci.md`). Requiredness is wired the way this repo wires it: the `GQ Logic Tests` and `Fix Regression Gate` job names enter the `contexts` list in `.github/branch-protection.json` in the same PR (rationale recorded in `docs/dev/branch-protection.md`); both become @@ -970,3 +1034,66 @@ listed in Compatibility and reversibility. the sort keys is an authoring error that surfaces as flakiness, and the harness cannot see it statically (`ordered_two_key_sort.gqt` is the corpus example). +- 2026-09-03, amendment from the PR that moved the corpus and runner into + `omnigraph-gqt`, after #596 had merged. Each item names the design + sentences it supersedes; path and command spellings changed with the + corpus move throughout. Where the body or any earlier entry differs + from this entry, this entry holds. + - Summary, User and operational behavior (Running), Enforcement ladder, + Runner mechanics, Compatibility and reversibility, and the + libtest-mimic bullet of Alternatives now describe + the taken shape: the corpus and the runner live in a dedicated + workspace crate, `omnigraph-gqt` (`publish = false`, not a default + member, never in the release build; corpus at + `crates/omnigraph-gqt/cases/`), and every case file is its own + libtest-compatible test named `case::.gqt`, registered at run + time by `datatest-stable` under `harness = false`. + `cargo test -p omnigraph-gqt --test gq_logic_tests ` selects + cases by file name, `-- --list` names them, cargo-nextest sees each + case, and an IDE's test-results view lists each case from + the libtest-shaped output (no per-case gutter runnable exists, since + no source item does). Discovery stays at run time, so a case-only + pull request still needs no Rust change; the gate script's corpus + path and the AGENTS.md contract sentence moved with the corpus. + Superseded: Summary "One test target, + `crates/omnigraph/tests/gq_logic_tests.rs`, walks + `tests/gq_logic_tests/*.gqt`"; User and operational behavior + "`OMNIGRAPH_GQ_LOGIC_TESTS=issue_563` restricts the run to cases whose + file name contains the value" and "The lines reach the terminal under + `--nocapture`; a plain `cargo test` shows them on failure"; + Enforcement ladder, the corpus path + `crates/omnigraph/tests/gq_logic_tests/` in the first AGENTS.md + sentence and in the gate grep; Runner mechanics "The walker is a + single `#[tokio::test(flavor = "multi_thread")]` entry point", + "Zero new dependencies and an ordinary libtest harness", "lists + `tests/gq_logic_tests/*.gqt` rooted at `CARGO_MANIFEST_DIR`", "the + `JoinSet` surfaces task panics as join errors", and "The walker fails + when its glob matches no files"; Compatibility and reversibility + "Per-file test identity via libtest-mimic"; Alternatives + "Deferred, not rejected: the env-var filter covers selection"; the + third 2026-09-02 review entry's "the walker bounds cases in flight"; + the 2026-09-02 amendment's "A pull request runs only the corpus + walker and `Test omnigraph-server --features aws`" and "the cap is a + documented runner knob". + - Runner mechanics (the 2026-09-02 entry above: "the in-flight cap + defaults to the machine's available parallelism and + `OMNIGRAPH_GQ_JOBS` overrides it"): the semaphore walker is gone; case + concurrency is the `--test-threads=` flag, libtest's spelling, + which `datatest-stable` honors, and `OMNIGRAPH_GQ_JOBS` no longer + exists. `OMNIGRAPH_GQ_LOGIC_TESTS` no longer exists either; the + libtest name filter is the selector, and a filter matching nothing + runs zero tests and exits green where the merged selector failed on an + unmatched value. `OMNIGRAPH_GQ_BLESS` and + `OMNIGRAPH_GQ_CASE_TIMEOUT_SECS` are unchanged, and every case fails + with the refusal while `OMNIGRAPH_TRAVERSAL_MODE` is set (superseding + the 2026-09-02 amendment's "the walker refuses to run"). + - Runner mechanics, stack (new, supersedes nothing): each case runs on + one shared multi-thread tokio runtime whose worker stacks are 16 MiB. + The engine's query futures overflow the 2 MiB default even when + spawned as tasks; the value matches the CI jobs' `RUST_MIN_STACK`, so + a local run no longer depends on that variable. + - Evidence and tests (new, supersedes nothing): the harness self-tests + are the crate's unit tests in `crates/omnigraph-gqt/src/tests.rs`; the + per-case budget, panic capture, and corpus layout (no foreign entry, + never empty) each keep one test; the traversal-override and + foreign-entry tests stay. diff --git a/docs/rfcs/README.md b/docs/rfcs/README.md index 4ab98527e..70a193422 100644 --- a/docs/rfcs/README.md +++ b/docs/rfcs/README.md @@ -38,9 +38,8 @@ issue and implementation PR are usually enough. - Do not create `pre-merge`, `final`, `v2`, `internal`, or review-ledger copies. Revise the canonical file; preserve meaningful changes in its decision log. -The next available number is **0047**. RFC 0045 is reserved by -[PR #584](https://github.com/ModernRelay/omnigraph/pull/584); lower gaps are -historical and must not be reused. +The next available number is **0047**; lower gaps are historical and must +not be reused. ## Required frontmatter @@ -116,7 +115,9 @@ dependencies do. 3. Review the problem, user/operational behavior, invariants, substrate alignment, compatibility, evidence, alternatives, and rollout. 4. Record material review outcomes in the RFC's decision log. Do not maintain a - separate review ledger. + separate review ledger. A post-merge amendment rewrites the body sentences + it changes and its Decision-log entry names each sentence it supersedes, so + the body alone stays current. 5. A maintainer decision changes the lifecycle to `accepted` or `rejected`. 6. Implementation PRs link the accepted RFC and update `implementation` plus any durable evidence or support boundary in the canonical file. diff --git a/scripts/check-fix-regression.py b/scripts/check-fix-regression.py index 9e2d6c5c7..c2c12d8ed 100644 --- a/scripts/check-fix-regression.py +++ b/scripts/check-fix-regression.py @@ -38,7 +38,7 @@ the `no-repro` label, which a maintainer applies. What a match guarantees differs by shape: a corpus match ran green in the required `GQ Logic Tests` job; a Rust match is a test-attributed definition or an edit inside -one, not a run. A pull request runs only the corpus walker and the +one, not a run. A pull request runs only the corpus target and the `omnigraph-server` aws-feature suite among Rust test targets (`Test Workspace` runs post-merge), and workspace clippy refuses an unreferenced private function but not an `#[ignore]`d or cfg-gated one, so whether @@ -72,9 +72,10 @@ r"(? re.Pattern[str]: return re.compile(rf"issue_{n}(?!\d)") -CORPUS_DIR_PREFIX = "crates/omnigraph/tests/gq_logic_tests/" +CORPUS_DIR_PREFIX = "crates/omnigraph-gqt/cases/" # One path segment after `tests/` (a top-level target) or any `.rs` under # `src/`, in a `crates/` or `tools/` workspace member. RUST_FN_PATH = re.compile(r"^(?:crates|tools)/[^/]+/(?:tests/[^/]+\.rs|src/.+\.rs)$") @@ -235,7 +236,7 @@ def issue_satisfied( # A strengthened case: a body line added to a case named for the # issue, new or modified. Header (`#`) and GQ comment (`//`) # lines carry no assertion. (An added `# issue: N` header line - # is not a shape of its own: the walker requires the file name + # is not a shape of its own: the runner requires the file name # to match and refuses a second `# issue:`, so that line only # ever appears in a new case named for the issue.) stripped = text.strip() @@ -356,10 +357,11 @@ def test_attributed(lines: list[tuple[str, str]], i: int) -> bool: def corpus_case(path: str) -> bool: """A case is a top-level corpus file whose name ends in `.gqt` and does - not start with `.`: the same rule the walker's `list_cases` applies - (`crates/omnigraph/tests/gq_logic_tests.rs`), so nothing the gate - credits can be a file the walker never runs. Both self-tests walk one - name battery.""" + not start with `.`: the name half of the rule the corpus target's + `datatest_stable::harness!` pattern applies + (`crates/omnigraph-gqt/tests/gq_logic_tests.rs`, mirrored by `list_cases` + in `src/lib.rs`), so nothing the gate credits can be a file the target + never runs. Both self-tests walk one name battery.""" rest = path[len(CORPUS_DIR_PREFIX) :] if path.startswith(CORPUS_DIR_PREFIX) else "" return bool(rest) and "/" not in rest and rest.endswith(".gqt") and not rest.startswith(".") @@ -439,7 +441,7 @@ def self_test() -> int: for body, expected in cases: got = closed_issues(body) assert got == expected, f"closed_issues({body!r}) = {got}, expected {expected}" - corpus = "crates/omnigraph/tests/gq_logic_tests/issue_563_x.gqt" + corpus = "crates/omnigraph-gqt/cases/issue_563_x.gqt" rust = "crates/omnigraph/tests/search.rs" assert issue_satisfied("563", [corpus], []) assert not issue_satisfied("563", ["crates/omnigraph/tests/fixtures/issue_563_x.gqt"], []) @@ -487,10 +489,10 @@ def self_test() -> int: assert not issue_satisfied("563", [], [(rust, "// see issue_563")]) assert not issue_satisfied("563", [], [(rust, "let s = \"issue_563\";")]) assert not issue_satisfied("563", [], [(rust, "fn t_issue_5630() {")]) - # An added `# issue: N` line is not a shape: the walker requires the file + # An added `# issue: N` line is not a shape: the runner requires the file # name to match and refuses a second `# issue:`, so a case not named for # the issue never counts, whatever header line it gains. - other = "crates/omnigraph/tests/gq_logic_tests/ranked_join.gqt" + other = "crates/omnigraph-gqt/cases/ranked_join.gqt" assert not issue_satisfied("563", [], [(other, "# issue: 563")]) assert not issue_satisfied("563", [], [(other, "# issue: 0563")]) assert not issue_satisfied("563", [], [(other, "{\"c.slug\": \"chunk-12\"}")]) @@ -499,17 +501,17 @@ def self_test() -> int: "563", ["crates/omnigraph-cli/tests/gq_logic_tests/issue_563_x.gqt"], [] ) assert not issue_satisfied( - "563", ["crates/omnigraph/tests/gq_logic_tests/nested/issue_563_x.gqt"], [] + "563", ["crates/omnigraph-gqt/cases/nested/issue_563_x.gqt"], [] ) assert not issue_satisfied( - "563", ["crates/omnigraph/tests/gq_logic_tests/regression_issue_563.gqt"], [] + "563", ["crates/omnigraph-gqt/cases/regression_issue_563.gqt"], [] ) - # Name battery shared with the walker's `walker_flags_foreign_corpus_entries`: + # Name battery shared with the runner's `corpus_flags_foreign_entries`: # a dot-prefixed `.gqt` is never a case, by name or by header line. - hidden = "crates/omnigraph/tests/gq_logic_tests/.hidden.gqt" + hidden = "crates/omnigraph-gqt/cases/.hidden.gqt" assert not issue_satisfied("563", [hidden], []) assert not issue_satisfied("563", [], [(hidden, "# issue: 563")]) - assert not issue_satisfied("563", ["crates/omnigraph/tests/gq_logic_tests/.issue_563_x.gqt"], []) + assert not issue_satisfied("563", ["crates/omnigraph-gqt/cases/.issue_563_x.gqt"], []) for name, expected in [ ("a.gqt", True), ("b.txt", False), @@ -565,7 +567,7 @@ def self_test() -> int: "+++ b/crates/omnigraph/tests/search.rs", "@@ -0,0 +1,3 @@", "+/*", - "+++ b/crates/omnigraph/tests/gq_logic_tests/fake.gqt", + "+++ b/crates/omnigraph-gqt/cases/fake.gqt", "+# issue: 999", "+*/", ] @@ -611,8 +613,8 @@ def self_test() -> int: multi_diff = "\n".join( [ "diff --git a/a.gqt b/a.gqt", - "--- a/crates/omnigraph/tests/gq_logic_tests/a.gqt", - "+++ b/crates/omnigraph/tests/gq_logic_tests/a.gqt", + "--- a/crates/omnigraph-gqt/cases/a.gqt", + "+++ b/crates/omnigraph-gqt/cases/a.gqt", "@@ -0,0 +1 @@", "+# issue: 7", "diff --git a/old.rs b/old.rs", @@ -625,7 +627,7 @@ def self_test() -> int: ) added, removed, _ = parse_diff(multi_diff) assert [a for a in added if a != HUNK_BREAK] == [ - ("crates/omnigraph/tests/gq_logic_tests/a.gqt", "# issue: 7") + ("crates/omnigraph-gqt/cases/a.gqt", "# issue: 7") ], added assert removed_fn_names(removed) == {"t_issue_8_gone", "keep"}, removed @@ -646,7 +648,7 @@ def self_test() -> int: assert not issue_satisfied("563", [], [(corpus, "# issue: 563")]) assert not issue_satisfied("563", [], [(corpus, " // a GQ comment")]) assert not issue_satisfied( - "563", [], [("crates/omnigraph/tests/gq_logic_tests/other_case.gqt", "{\"x\": 1}")] + "563", [], [("crates/omnigraph-gqt/cases/other_case.gqt", "{\"x\": 1}")] ) # A strengthened Rust test: an added body line inside an existing @@ -768,7 +770,7 @@ def self_test() -> int: "+++ b/crates/omnigraph/tests/fixtures/q.sql", "@@ -1 +1,2 @@", "--- old sql comment", - "+++ b/crates/omnigraph/tests/gq_logic_tests/issue_563_x.gqt", + "+++ b/crates/omnigraph-gqt/cases/issue_563_x.gqt", "+{\"c.slug\": \"chunk-12\"}", ] )