From 56f564ac4eba39e086bb557393886b52e5327dda Mon Sep 17 00:00:00 2001 From: Leo Meyerovich Date: Mon, 27 Jul 2026 12:52:33 -0700 Subject: [PATCH 1/2] fix(gfql): decline three silently-wrong bounded var-length polars shapes (#1787) Pandas is the oracle and the contract is parity-or-NotImplementedError. Three BOUNDED var-length shapes in the native polars `rows(binding_ops=...)` path returned a DIFFERENT count with no error. Same root-cause family as the unbounded case #1781 declined: pandas' step_pairs come from the var-length `edge_op.execute` hop, whose hop-window pruning -- and, when seeded, its per-seed BFS -- changes the edge multiplicity in a way a rebuild from the raw matching edge table does not reproduce. Declined (diverging graphs out of 60 random ones per shape, differential fuzz against the pandas oracle): * directed `-[*k..m]->` with min_hops >= 3 (53/60; pandas 0 vs polars 35), and min_hops >= 2 off a FILTERED seed (27-30/60). Plain min_hops <= 2 is fuzz-clean, including as a non-first segment -- that is the graph-bench q3 `-[*1..k]->` shape, still served. * undirected `-[*1..1]-` / `-[*1]-`, the degenerate window: 60/60, halves the count (pandas 36 vs polars 18). `-[*1..2]-` and wider agree, which is why the existing tests missed it. * undirected `-[*1..k]-` that does not start from the full node set: filtered seed 51/60, non-first segment 36/60. Every directed equivalent agrees, so this is specific to the undirected doubled-pair expansion. The gate keys on an EXPLICIT var-length window rather than on `EdgeSemantics.is_multihop`, because `-[*1..1]-` resolves to min == max == 1 and so is NOT multihop -- yet pandas still routes it through the var-length hop (`-[]-` gives 18 on a graph where `-[*1..1]-` gives 36). Plain `-[]-` is untouched. Tests: explicit decline + still-served neighbours for each of the three, plus a seeded differential fuzz over random small graphs (cyclic, parallel edges, self-loops) that asserts parity-or-raise AND that enough shapes are still served for the check to mean anything. Bounded windows only: undirected unbounded shapes through the pandas oracle can exhaust the box. The new test file is REGISTERED in bin/test-polars.sh. That list is an explicit allowlist, and the polars lane is the only CI lane with polars installed -- so the only lane that can execute this file at all. Without the registration the three `return None` decline statements were the only changed lines no lane ever ran (10/13 = 76.92%, exactly the changed-line-coverage gate failure); with it the block is 13/13. Mutation-checked: dropping the min_hops clause fails 4 tests, the degenerate undirected window 3, the undirected-seed clause 3, the whole gate 8. Full graphistry/tests/compute is unchanged by the declines (9 pre-existing failures, same set). Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_015YsqAZQLbqjSDrYSFz2GoB --- bin/test-polars.sh | 1 + .../gfql/lazy/engine/polars/row_pipeline.py | 46 ++++++ ...st_engine_polars_varlen_divergence_1787.py | 145 ++++++++++++++++++ 3 files changed, 192 insertions(+) create mode 100644 graphistry/tests/compute/gfql/test_engine_polars_varlen_divergence_1787.py diff --git a/bin/test-polars.sh b/bin/test-polars.sh index 3bfc16a4dd..8c45f811f7 100755 --- a/bin/test-polars.sh +++ b/bin/test-polars.sh @@ -24,6 +24,7 @@ POLARS_TEST_FILES=( graphistry/tests/compute/gfql/test_engine_polars_chain.py graphistry/tests/compute/gfql/test_engine_polars_row_pipeline.py graphistry/tests/compute/gfql/test_engine_polars_binding_rows.py + graphistry/tests/compute/gfql/test_engine_polars_varlen_divergence_1787.py graphistry/tests/compute/gfql/test_engine_polars_with_match_reentry.py graphistry/tests/compute/gfql/test_engine_polars_cypher_conformance.py graphistry/tests/compute/gfql/test_engine_polars_conformance_matrix.py diff --git a/graphistry/compute/gfql/lazy/engine/polars/row_pipeline.py b/graphistry/compute/gfql/lazy/engine/polars/row_pipeline.py index 173842eef6..6a3035a923 100644 --- a/graphistry/compute/gfql/lazy/engine/polars/row_pipeline.py +++ b/graphistry/compute/gfql/lazy/engine/polars/row_pipeline.py @@ -1312,6 +1312,52 @@ def _names(lf: pl.LazyFrame) -> List[str]: ) if _resolved_min != 1: return None + # #1787: three more BOUNDED var-length shapes this raw-edge reconstruction gets + # SILENTLY wrong, all found by differential fuzzing against the pandas oracle + # over random small graphs (the `n/60` counts below are diverging graphs out of + # 60 random ones per shape). Same root-cause family as the unbounded case #1781 + # declined: pandas' step_pairs come from the var-length `edge_op.execute` hop, + # whose hop-window pruning -- and, when seeded, its per-seed BFS -- changes the + # edge multiplicity in a way a rebuild from the raw matching edge table does not + # reproduce. Pandas is the oracle and the contract is parity-or-NotImplementedError, + # so decline until the multiplicity is reconstructible. + # + # Gated on an EXPLICIT var-length window, not on `sem.is_multihop`: the degenerate + # `-[*1..1]-` resolves to min == max == 1 and so is NOT multihop, yet pandas still + # routes it through the var-length hop (`-[]-` gives 18 where `-[*1..1]-` gives 36 + # on the same graph). That is exactly why the existing tests miss it. + if op.min_hops is not None or op.max_hops is not None: + _vl_max = op.max_hops if op.max_hops is not None else op.hops + _vl_min = op.min_hops if op.min_hops is not None else ( + op.hops if op.hops is not None else 1 + ) + # A seed is anything that starts the segment from less than the whole node + # set: a filtered start alias, or a re-entry / `WITH` seed frame. + _prev_op = ops[idx - 1] if idx >= 1 else None + _seeded_start = start_nodes is not None or ( + isinstance(_prev_op, ASTNode) and bool(_prev_op.filter_dict) + ) + if _vl_max is not None: + if op.direction == "undirected": + # - `-[*1..1]-` / `-[*1]-`: the DEGENERATE window halves the count + # (60/60; pandas 36 vs polars 18). `-[*1..2]-` and wider agree. + if _vl_max == 1: + return None + # - undirected `-[*1..k]-` OFF THE FULL NODE SET: the doubled-pair + # expansion over-counts when the segment does not start from every + # node -- filtered seed 51/60, non-first segment 36/60. Every + # DIRECTED equivalent agrees, so this is specific to the doubling. + if _seeded_start or idx > 1: + return None + # - directed `-[*k..m]->` with min_hops >= 3 (53/60; pandas 0 vs polars + # 35), and min_hops >= 2 WITH a filtered seed (27-30/60). + # `max_reached_hop` in compute/hop.py is a dedup-by-node BFS + # eccentricity, not a longest-walk length, so pandas returns empty (or + # prunes) where this rebuild expands a different edge multiset. + # min_hops <= 2 off the full node set is fuzz-clean (0/60, including as + # a non-first segment) -- that is the graph-bench q3 `-[*1..k]->` shape. + elif _vl_min >= 3 or (_vl_min >= 2 and _seeded_start): + return None if op.direction not in ("forward", "reverse", "undirected"): return None if any( diff --git a/graphistry/tests/compute/gfql/test_engine_polars_varlen_divergence_1787.py b/graphistry/tests/compute/gfql/test_engine_polars_varlen_divergence_1787.py new file mode 100644 index 0000000000..338ff96d37 --- /dev/null +++ b/graphistry/tests/compute/gfql/test_engine_polars_varlen_divergence_1787.py @@ -0,0 +1,145 @@ +"""Native polars var-length binding rows: parity with the pandas oracle, or decline (#1787). + +Three bounded shapes returned a DIFFERENT count from pandas with no error -- the contract is +parity-or-``NotImplementedError``, so they are now declined, mirroring what #1781 did for the +unbounded case: + + 1. directed ``-[*k..m]->`` with ``min_hops >= 3`` (and ``min_hops >= 2`` off a filtered seed); + 2. undirected ``-[*1..1]-`` / ``-[*1]-``, the degenerate window, which HALVED the count; + 3. undirected ``-[*1..k]-`` that does not start from the full node set (filtered seed, or a + non-first segment), which OVER-counted. + +The differential fuzz at the bottom is what found all three (and what proves the gate is not +merely declining everything): it asserts served shapes still equal pandas AND that the +neighbouring shapes are still served, so a lazily over-broad gate fails it. +""" +import random + +import pandas as pd +import pytest + +import graphistry + +pl = pytest.importorskip("polars") + + +def _pair(nodes, edges): + return ( + graphistry.nodes(nodes, "id").edges(edges, "s", "d"), + graphistry.nodes(pl.from_pandas(nodes), "id").edges(pl.from_pandas(edges), "s", "d"), + ) + + +def _count(g, query, engine): + return int(g.gfql(query, engine=engine)._nodes["c"].to_list()[0]) + + +# The fixtures from the issue, one per divergence. +BOUNDED_NODES = pd.DataFrame({"id": list(range(7))}) +BOUNDED_EDGES = pd.DataFrame( + [(4, 6), (0, 4), (1, 2), (4, 6), (5, 6), (4, 5), (4, 6)], columns=["s", "d"] +) +UNDIR_NODES = pd.DataFrame({"id": list(range(5))}) +UNDIR_EDGES = pd.DataFrame( + [(1, 2), (1, 2), (0, 2), (1, 0), (4, 2), (3, 2), (3, 4), (0, 3), (1, 2)], columns=["s", "d"] +) +SEEDED_NODES = pd.DataFrame({"id": list(range(7)), "kind": ["a", "b"] * 3 + ["a"]}) +SEEDED_EDGES = pd.DataFrame( + [(0, 3), (6, 3), (4, 2), (1, 3), (6, 3), (3, 1), (3, 5), (4, 1), (3, 3), (0, 3)], + columns=["s", "d"], +) + +DECLINED = [ + # (nodes, edges, query) -- each returned a wrong count before the gate. + (BOUNDED_NODES, BOUNDED_EDGES, "MATCH (a)-[*3..3]->(b) RETURN count(*) AS c"), + (BOUNDED_NODES, BOUNDED_EDGES, "MATCH (a)-[*3..4]->(b) RETURN count(*) AS c"), + (SEEDED_NODES, SEEDED_EDGES, "MATCH (a {kind:'a'})-[*2..3]->(b) RETURN count(*) AS c"), + (UNDIR_NODES, UNDIR_EDGES, "MATCH (a)-[*1..1]-(b) RETURN count(*) AS c"), + (UNDIR_NODES, UNDIR_EDGES, "MATCH (a)-[*1]-(b) RETURN count(*) AS c"), + (SEEDED_NODES, SEEDED_EDGES, "MATCH (a {kind:'a'})-[*1..2]-(b) RETURN count(*) AS c"), + (SEEDED_NODES, SEEDED_EDGES, "MATCH (a)-[]->(b)-[*1..2]-(c) RETURN count(*) AS c"), +] + +STILL_SERVED = [ + # Neighbours of each declined shape: the gate must not swallow these. + (BOUNDED_NODES, BOUNDED_EDGES, "MATCH (a)-[*1..2]->(b) RETURN count(*) AS c"), + (BOUNDED_NODES, BOUNDED_EDGES, "MATCH (a)-[*2..3]->(b) RETURN count(*) AS c"), + (UNDIR_NODES, UNDIR_EDGES, "MATCH (a)-[]-(b) RETURN count(*) AS c"), + (UNDIR_NODES, UNDIR_EDGES, "MATCH (a)-[*1..2]-(b) RETURN count(*) AS c"), + (UNDIR_NODES, UNDIR_EDGES, "MATCH (a)-[*1..3]-(b) RETURN count(*) AS c"), + (SEEDED_NODES, SEEDED_EDGES, "MATCH (a {kind:'a'})-[*1..2]->(b) RETURN count(*) AS c"), + (SEEDED_NODES, SEEDED_EDGES, "MATCH (a)-[]->(b)-[*2..3]->(c) RETURN count(*) AS c"), +] + + +@pytest.mark.parametrize("nodes,edges,query", DECLINED) +def test_divergent_varlen_shapes_decline(nodes, edges, query): + """A shape that cannot be reproduced must RAISE, never answer differently.""" + _, g_pl = _pair(nodes, edges) + with pytest.raises(NotImplementedError): + g_pl.gfql(query, engine="polars") + + +@pytest.mark.parametrize("nodes,edges,query", STILL_SERVED) +def test_neighbouring_varlen_shapes_still_served_and_match_pandas(nodes, edges, query): + """The gate is a scalpel: every neighbouring shape still runs natively AND matches pandas.""" + g_pd, g_pl = _pair(nodes, edges) + assert _count(g_pl, query, "polars") == _count(g_pd, query, "pandas") + + +@pytest.mark.parametrize("nodes,edges,query", DECLINED) +def test_declined_shapes_still_answerable_on_pandas(nodes, edges, query): + """Declining is an ENGINE limit, not a query rejection: pandas still answers all of them.""" + g_pd, _ = _pair(nodes, edges) + assert _count(g_pd, query, "pandas") >= 0 + + +FUZZ_SHAPES = [ + "MATCH (a)-[*1..2]->(b) RETURN count(*) AS c", + "MATCH (a)-[*2..2]->(b) RETURN count(*) AS c", + "MATCH (a)-[*2..3]->(b) RETURN count(*) AS c", + "MATCH (a)-[*3..3]->(b) RETURN count(*) AS c", + "MATCH (a)-[*1..1]-(b) RETURN count(*) AS c", + "MATCH (a)-[*1..2]-(b) RETURN count(*) AS c", + "MATCH (a)-[*1..3]-(b) RETURN count(*) AS c", + "MATCH (a {kind:'a'})-[*1..2]-(b) RETURN count(*) AS c", + "MATCH (a {kind:'a'})-[*1..2]->(b) RETURN count(*) AS c", + "MATCH (a {kind:'a'})-[*2..3]->(b) RETURN count(*) AS c", + "MATCH (a)-[]->(b)-[*1..2]-(c) RETURN count(*) AS c", + "MATCH (a)-[]->(b)-[*2..3]->(c) RETURN count(*) AS c", +] + + +def test_bounded_varlen_differential_fuzz_against_pandas_oracle(): + """Random small graphs (cyclic, parallel edges, self-loops): match pandas or raise. + + Bounded windows only -- undirected UNBOUNDED shapes through the pandas oracle can blow the + box up (75 GiB RSS observed), and they are already declined elsewhere. Seeded so a failure + is reproducible; the shape list spans both sides of every gate boundary, so a gate that + declines too much shows up as the served-count assertion below going to zero. + """ + rnd = random.Random(1787) + served = 0 + for _ in range(25): + n_nodes = rnd.randint(3, 7) + nodes = pd.DataFrame({ + "id": list(range(n_nodes)), + "kind": [rnd.choice("ab") for _ in range(n_nodes)], + }) + edges = pd.DataFrame( + [(rnd.randrange(n_nodes), rnd.randrange(n_nodes)) for _ in range(rnd.randint(3, 10))], + columns=["s", "d"], + ) + g_pd, g_pl = _pair(nodes, edges) + for query in FUZZ_SHAPES: + expected = _count(g_pd, query, "pandas") + try: + got = _count(g_pl, query, "polars") + except NotImplementedError: + continue # an honest decline is allowed; a different answer is not + served += 1 + assert got == expected, ( + f"polars diverged from the pandas oracle on {query!r}: " + f"{got} != {expected}\nnodes={nodes.to_dict('list')}\nedges={edges.values.tolist()}" + ) + assert served > 100, f"gate declines too much to be a useful parity check ({served})" From a41b2c362ae40c67949baba04babd67486427cdb Mon Sep 17 00:00:00 2001 From: Leo Meyerovich Date: Mon, 27 Jul 2026 15:52:32 -0700 Subject: [PATCH 2/2] test(gfql): engine-parametrize the #1787 varlen contract (pandas/polars/cuDF/polars-gpu) Review follow-ups on #1794. Comments -> tests: the gate in row_pipeline.py carried per-shape prose asserting which shapes diverge and by how much. A comment asserting behaviour is unverified and rots. Every numeric claim is now a named test; the comment keeps only the WHY of the deliberate divergence from master's serving decision plus a pointer to the tests. 29 -> 18 comment lines, executable logic byte-identical (the only non-comment hunks are two trailing comments on existing lines). Engine-parametrized: renamed to test_varlen_bounded_engine_parity_1787.py and parametrized over pandas / polars / cuDF / polars-gpu. It encodes the INTENDED per-engine behaviour rather than assuming identity -- the polars engines must decline exactly the shapes the pandas-API engines must answer -- with oracle counts pinned as literals so a regression that makes every engine equally wrong still fails. GPU params gate on a runtime probe and report a reasoned SKIP, never a silent pass; the coverage boundary (no CI lane runs cuDF or polars-gpu) is stated in the module docstring. Dropping the module-level polars importorskip also means the pandas params now execute in test-gfql-core, where the old file was skipped in its entirety (re: #1795). Found while doing so: cuDF silently returns 1 where pandas returns 9 on seeded undirected degenerate `(a {p:v})-[*1..1]-(b)` (23/40 random graphs; drop any one of seeded/undirected/max==1 and it agrees). Separate defect, filed as #1798 and pinned here with xfail(strict=True). Mutation-checked both directions: reverting the gate to 233b64c8 fails 13 tests; widening it to decline all bounded var-length fails 12. Verified: dgx-spark GB10 / cuDF 26.02.01 / polars 1.35.2 via graphistry/test-rapids-official:26.02-gfql-polars with --gpus all -- 80 passed, 1 xfailed, 0 skipped. CI-equivalent local polars lane (cuDF import-blocked) 2395 passed, 54 skipped, 0 failed. bin/lint.sh and bin/typecheck.sh clean. Note: the accompanying CI fix for the test-polars py3.12 timeout (#1797) touches .github/workflows/ci.yml and could not be pushed with the available token, which lacks the `workflow` OAuth scope. The patch is attached to the PR. Co-Authored-By: Claude Opus 5 Claude-Session: https://claude.ai/code/session_015YsqAZQLbqjSDrYSFz2GoB --- CHANGELOG.md | 1 + bin/test-polars.sh | 4 +- .../gfql/lazy/engine/polars/row_pipeline.py | 49 +- ...st_engine_polars_varlen_divergence_1787.py | 145 ------ .../test_varlen_bounded_engine_parity_1787.py | 427 ++++++++++++++++++ 5 files changed, 450 insertions(+), 176 deletions(-) delete mode 100644 graphistry/tests/compute/gfql/test_engine_polars_varlen_divergence_1787.py create mode 100644 graphistry/tests/compute/gfql/test_varlen_bounded_engine_parity_1787.py diff --git a/CHANGELOG.md b/CHANGELOG.md index ac88700713..aee5884cd6 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -28,6 +28,7 @@ This project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.htm - **GFQL execution context is declared rather than attached at runtime**: the private `_gfql_*` per-execution fields (index policy and registry, row-pipeline base graph, carried seed nodes, edge aliases, shortest-path backend, and the indexed-bindings handoff) are now declared on `Plottable` with defaults on `PlotterBase`, instead of being set with `setattr` and read back with `getattr(..., default)`. No public API or behaviour change; internal call sites are typed, and hand-rolled `Plottable` stand-ins must now construct the full context. ### Fixed +- **Three BOUNDED variable-length shapes returned a silently different count on `engine='polars'` (#1787)**: the native polars `rows(binding_ops=...)` builder rebuilds a variable-length segment from the raw matching edge table, and for three bounded shapes that rebuild produces a different edge multiplicity than the pandas oracle — with no error, just a different number. They now raise `NotImplementedError` like every other shape this lowering cannot reproduce, which is a deliberate behaviour change: these were *served* before, so a silent wrong answer becomes a loud, actionable error, and `engine='pandas'` still answers all of them. The three: (1) directed `-[*k..m]->` with `min_hops >= 3`, and `min_hops >= 2` when the segment starts from a filtered seed — `max_reached_hop` in `compute/hop.py` is a dedup-by-node BFS eccentricity rather than a longest-walk length, so the oracle prunes to empty where the rebuild expands a different edge multiset; (2) the DEGENERATE undirected window `-[*1..1]-` / `-[*1]-`, which halved the count — it resolves to `min == max == 1` and so is not "multihop", yet pandas still routes it through the variable-length hop, which is exactly why it slipped past the existing gates (the gate therefore keys on an explicit variable-length window, not on `is_multihop`); (3) undirected `-[*1..k]-` that does not start from the full node set (filtered seed, or a non-first segment), where the doubled-pair expansion over-counts. Every neighbouring shape stays native and is pinned as such — unseeded `min_hops <= 2`, the directed twin of each undirected decline, the directed degenerate window, and the plain `-[]-` edge — so the gate is a scalpel rather than a blanket refusal of variable-length segments. Same root-cause family as the unbounded shapes #1781 declined; the gate should shrink again once the multiplicity is reconstructible. Found by differential fuzzing against the pandas oracle, and pinned by an engine-parametrized suite (pandas / polars / cuDF / polars-gpu) that encodes the intended per-engine behaviour — the polars engines must decline exactly where the pandas-API engines must answer — plus a seeded fuzz that fails if the gate declines too much. - **`rows(table='edges')` silently returned the wrong table after a NAMED traversal**: the named-middle rewrite — which turns `[...named ops..., rows()]` into `rows(binding_ops=...)` so a Cypher multi-alias `RETURN` lowers to a bindings table — skipped itself when the call already carried `binding_ops`, `source` or `alias_endpoints`, but not when it carried a non-default `table`. So naming any op in the middle changed which table came back: an odd-length middle silently produced the BINDINGS table instead of the requested edges (no error, `table=` simply ignored), and an even-length middle — a path ending on an edge, e.g. `(person)-[r:KNOWS]-` — produced a non-alternating op list and raised "require ... a single connected alternating node/edge path". An unnamed middle was unaffected, which is what made it look shape-specific rather than naming-specific. Both chain surfaces are fixed (the generic chain and the native polars chain carry the rewrite independently, so fixing one left the other wrong). Note the guard keys on a NON-DEFAULT table: `rows()` declares `table='nodes'` and always emits it, so an explicit `rows(table='nodes')` is indistinguishable from a bare `rows()` and still rewrites — pinned as a known limitation rather than left to be rediscovered. - **Indexed bindings bypass returned the WHOLE edge table for `rows(table='edges')`**: the bypass exists to SKIP the canonical traversal, so whatever it hands the suffix is the pre-traversal graph. That is sound for a bindings table — the indexed path bag already is the answer — but the gate declined only for `source` / `alias_endpoints` / `alias_prefilters`, not for a non-default `table`, so a `rows(table='edges')` after a named middle read the full edge frame instead of the traversal-narrowed one (measured on a 12-edge fixture: 12 rows instead of 3, on pandas, cuDF and native polars, and the wrong values survive a following `select`). Both gates now decline a non-default `table` (`chain._plan_indexed_middle` and the native polars `_try_indexed_middle_polars`), matching the named-middle rewrite's guard. This became reachable only once that rewrite stopped firing for a non-default `table` — before it, the rewrite converted the call on BOTH the indexed and the scan path, so the two agreed on the wrong table. Same class as the earlier "indexed bypass could return every node for `rows()` over an unnamed pattern". - **The named-middle rewrite discarded the caller's other `rows()` params**: both rewrites built a FRESH `rows(binding_ops=...)` instead of adding `binding_ops` to the call the caller wrote, so every param the rewrite has no opinion about was thrown away. For `attach_prop_aliases` — the projection pushdown — that is observable: spelling a pattern as a NAMED middle rather than as explicit `binding_ops` silently attached every alias's properties instead of the requested ones, i.e. the same query returned a wider schema depending only on how it was written. `alias_prefilters` was dropped the same way. Both are now carried through, on both chain surfaces. Known remaining gap, pinned by a strict-xfail test rather than left to be rediscovered: the native polars bindings builder never receives `alias_prefilters` at all, so hand-written GFQL passing that hint without an equivalent post-filter still gets engine-dependent row counts (Cypher-generated plans always keep the post-filter, so they agree across engines and only lose the pushdown). diff --git a/bin/test-polars.sh b/bin/test-polars.sh index 8c45f811f7..adc4002108 100755 --- a/bin/test-polars.sh +++ b/bin/test-polars.sh @@ -24,7 +24,9 @@ POLARS_TEST_FILES=( graphistry/tests/compute/gfql/test_engine_polars_chain.py graphistry/tests/compute/gfql/test_engine_polars_row_pipeline.py graphistry/tests/compute/gfql/test_engine_polars_binding_rows.py - graphistry/tests/compute/gfql/test_engine_polars_varlen_divergence_1787.py + # engine-parametrized (pandas/polars/cudf/polars-gpu); the pandas params also run in + # test-gfql-core, but only this lane has polars installed + graphistry/tests/compute/gfql/test_varlen_bounded_engine_parity_1787.py graphistry/tests/compute/gfql/test_engine_polars_with_match_reentry.py graphistry/tests/compute/gfql/test_engine_polars_cypher_conformance.py graphistry/tests/compute/gfql/test_engine_polars_conformance_matrix.py diff --git a/graphistry/compute/gfql/lazy/engine/polars/row_pipeline.py b/graphistry/compute/gfql/lazy/engine/polars/row_pipeline.py index 6a3035a923..e4e45f0910 100644 --- a/graphistry/compute/gfql/lazy/engine/polars/row_pipeline.py +++ b/graphistry/compute/gfql/lazy/engine/polars/row_pipeline.py @@ -1312,50 +1312,39 @@ def _names(lf: pl.LazyFrame) -> List[str]: ) if _resolved_min != 1: return None - # #1787: three more BOUNDED var-length shapes this raw-edge reconstruction gets - # SILENTLY wrong, all found by differential fuzzing against the pandas oracle - # over random small graphs (the `n/60` counts below are diverging graphs out of - # 60 random ones per shape). Same root-cause family as the unbounded case #1781 - # declined: pandas' step_pairs come from the var-length `edge_op.execute` hop, - # whose hop-window pruning -- and, when seeded, its per-seed BFS -- changes the - # edge multiplicity in a way a rebuild from the raw matching edge table does not - # reproduce. Pandas is the oracle and the contract is parity-or-NotImplementedError, - # so decline until the multiplicity is reconstructible. + # #1787, same root-cause family as the unbounded shapes #1781 declined: pandas' + # step_pairs come from the var-length `edge_op.execute` hop, whose hop-window + # pruning -- and, when seeded, its per-seed BFS -- changes an edge multiplicity + # this raw-edge rebuild cannot reproduce. Declining is a DELIBERATE divergence + # from master, which served these: parity-or-NIE means a loud error, never a + # different number. Shrink the gate again once the multiplicity is reconstructible. + # WHICH shapes, why each boundary sits where it does, and the counts that prove + # each one are executable rather than prose -- every claim that used to be written + # out here is now a named test in: + # graphistry/tests/compute/gfql/test_varlen_bounded_engine_parity_1787.py # - # Gated on an EXPLICIT var-length window, not on `sem.is_multihop`: the degenerate - # `-[*1..1]-` resolves to min == max == 1 and so is NOT multihop, yet pandas still - # routes it through the var-length hop (`-[]-` gives 18 where `-[*1..1]-` gives 36 - # on the same graph). That is exactly why the existing tests miss it. + # Keyed on an EXPLICIT window, NOT on `sem.is_multihop`: `-[*1..1]-` resolves to + # min == max == 1, is therefore not multihop, and pandas still routes it here. if op.min_hops is not None or op.max_hops is not None: _vl_max = op.max_hops if op.max_hops is not None else op.hops _vl_min = op.min_hops if op.min_hops is not None else ( op.hops if op.hops is not None else 1 ) - # A seed is anything that starts the segment from less than the whole node - # set: a filtered start alias, or a re-entry / `WITH` seed frame. + # a seed is anything that starts the segment from less than the whole node + # set: a filtered start alias, or a re-entry / `WITH` seed frame _prev_op = ops[idx - 1] if idx >= 1 else None _seeded_start = start_nodes is not None or ( isinstance(_prev_op, ASTNode) and bool(_prev_op.filter_dict) ) if _vl_max is not None: if op.direction == "undirected": - # - `-[*1..1]-` / `-[*1]-`: the DEGENERATE window halves the count - # (60/60; pandas 36 vs polars 18). `-[*1..2]-` and wider agree. - if _vl_max == 1: + if _vl_max == 1: # the degenerate window `-[*1..1]-` / `-[*1]-` return None - # - undirected `-[*1..k]-` OFF THE FULL NODE SET: the doubled-pair - # expansion over-counts when the segment does not start from every - # node -- filtered seed 51/60, non-first segment 36/60. Every - # DIRECTED equivalent agrees, so this is specific to the doubling. - if _seeded_start or idx > 1: + if _seeded_start or idx > 1: # doubled-pair expansion over-counts return None - # - directed `-[*k..m]->` with min_hops >= 3 (53/60; pandas 0 vs polars - # 35), and min_hops >= 2 WITH a filtered seed (27-30/60). - # `max_reached_hop` in compute/hop.py is a dedup-by-node BFS - # eccentricity, not a longest-walk length, so pandas returns empty (or - # prunes) where this rebuild expands a different edge multiset. - # min_hops <= 2 off the full node set is fuzz-clean (0/60, including as - # a non-first segment) -- that is the graph-bench q3 `-[*1..k]->` shape. + # `max_reached_hop` (compute/hop.py) is a dedup-by-node BFS eccentricity, + # not a longest-walk length, so pandas prunes where this rebuild expands + # a different edge multiset elif _vl_min >= 3 or (_vl_min >= 2 and _seeded_start): return None if op.direction not in ("forward", "reverse", "undirected"): diff --git a/graphistry/tests/compute/gfql/test_engine_polars_varlen_divergence_1787.py b/graphistry/tests/compute/gfql/test_engine_polars_varlen_divergence_1787.py deleted file mode 100644 index 338ff96d37..0000000000 --- a/graphistry/tests/compute/gfql/test_engine_polars_varlen_divergence_1787.py +++ /dev/null @@ -1,145 +0,0 @@ -"""Native polars var-length binding rows: parity with the pandas oracle, or decline (#1787). - -Three bounded shapes returned a DIFFERENT count from pandas with no error -- the contract is -parity-or-``NotImplementedError``, so they are now declined, mirroring what #1781 did for the -unbounded case: - - 1. directed ``-[*k..m]->`` with ``min_hops >= 3`` (and ``min_hops >= 2`` off a filtered seed); - 2. undirected ``-[*1..1]-`` / ``-[*1]-``, the degenerate window, which HALVED the count; - 3. undirected ``-[*1..k]-`` that does not start from the full node set (filtered seed, or a - non-first segment), which OVER-counted. - -The differential fuzz at the bottom is what found all three (and what proves the gate is not -merely declining everything): it asserts served shapes still equal pandas AND that the -neighbouring shapes are still served, so a lazily over-broad gate fails it. -""" -import random - -import pandas as pd -import pytest - -import graphistry - -pl = pytest.importorskip("polars") - - -def _pair(nodes, edges): - return ( - graphistry.nodes(nodes, "id").edges(edges, "s", "d"), - graphistry.nodes(pl.from_pandas(nodes), "id").edges(pl.from_pandas(edges), "s", "d"), - ) - - -def _count(g, query, engine): - return int(g.gfql(query, engine=engine)._nodes["c"].to_list()[0]) - - -# The fixtures from the issue, one per divergence. -BOUNDED_NODES = pd.DataFrame({"id": list(range(7))}) -BOUNDED_EDGES = pd.DataFrame( - [(4, 6), (0, 4), (1, 2), (4, 6), (5, 6), (4, 5), (4, 6)], columns=["s", "d"] -) -UNDIR_NODES = pd.DataFrame({"id": list(range(5))}) -UNDIR_EDGES = pd.DataFrame( - [(1, 2), (1, 2), (0, 2), (1, 0), (4, 2), (3, 2), (3, 4), (0, 3), (1, 2)], columns=["s", "d"] -) -SEEDED_NODES = pd.DataFrame({"id": list(range(7)), "kind": ["a", "b"] * 3 + ["a"]}) -SEEDED_EDGES = pd.DataFrame( - [(0, 3), (6, 3), (4, 2), (1, 3), (6, 3), (3, 1), (3, 5), (4, 1), (3, 3), (0, 3)], - columns=["s", "d"], -) - -DECLINED = [ - # (nodes, edges, query) -- each returned a wrong count before the gate. - (BOUNDED_NODES, BOUNDED_EDGES, "MATCH (a)-[*3..3]->(b) RETURN count(*) AS c"), - (BOUNDED_NODES, BOUNDED_EDGES, "MATCH (a)-[*3..4]->(b) RETURN count(*) AS c"), - (SEEDED_NODES, SEEDED_EDGES, "MATCH (a {kind:'a'})-[*2..3]->(b) RETURN count(*) AS c"), - (UNDIR_NODES, UNDIR_EDGES, "MATCH (a)-[*1..1]-(b) RETURN count(*) AS c"), - (UNDIR_NODES, UNDIR_EDGES, "MATCH (a)-[*1]-(b) RETURN count(*) AS c"), - (SEEDED_NODES, SEEDED_EDGES, "MATCH (a {kind:'a'})-[*1..2]-(b) RETURN count(*) AS c"), - (SEEDED_NODES, SEEDED_EDGES, "MATCH (a)-[]->(b)-[*1..2]-(c) RETURN count(*) AS c"), -] - -STILL_SERVED = [ - # Neighbours of each declined shape: the gate must not swallow these. - (BOUNDED_NODES, BOUNDED_EDGES, "MATCH (a)-[*1..2]->(b) RETURN count(*) AS c"), - (BOUNDED_NODES, BOUNDED_EDGES, "MATCH (a)-[*2..3]->(b) RETURN count(*) AS c"), - (UNDIR_NODES, UNDIR_EDGES, "MATCH (a)-[]-(b) RETURN count(*) AS c"), - (UNDIR_NODES, UNDIR_EDGES, "MATCH (a)-[*1..2]-(b) RETURN count(*) AS c"), - (UNDIR_NODES, UNDIR_EDGES, "MATCH (a)-[*1..3]-(b) RETURN count(*) AS c"), - (SEEDED_NODES, SEEDED_EDGES, "MATCH (a {kind:'a'})-[*1..2]->(b) RETURN count(*) AS c"), - (SEEDED_NODES, SEEDED_EDGES, "MATCH (a)-[]->(b)-[*2..3]->(c) RETURN count(*) AS c"), -] - - -@pytest.mark.parametrize("nodes,edges,query", DECLINED) -def test_divergent_varlen_shapes_decline(nodes, edges, query): - """A shape that cannot be reproduced must RAISE, never answer differently.""" - _, g_pl = _pair(nodes, edges) - with pytest.raises(NotImplementedError): - g_pl.gfql(query, engine="polars") - - -@pytest.mark.parametrize("nodes,edges,query", STILL_SERVED) -def test_neighbouring_varlen_shapes_still_served_and_match_pandas(nodes, edges, query): - """The gate is a scalpel: every neighbouring shape still runs natively AND matches pandas.""" - g_pd, g_pl = _pair(nodes, edges) - assert _count(g_pl, query, "polars") == _count(g_pd, query, "pandas") - - -@pytest.mark.parametrize("nodes,edges,query", DECLINED) -def test_declined_shapes_still_answerable_on_pandas(nodes, edges, query): - """Declining is an ENGINE limit, not a query rejection: pandas still answers all of them.""" - g_pd, _ = _pair(nodes, edges) - assert _count(g_pd, query, "pandas") >= 0 - - -FUZZ_SHAPES = [ - "MATCH (a)-[*1..2]->(b) RETURN count(*) AS c", - "MATCH (a)-[*2..2]->(b) RETURN count(*) AS c", - "MATCH (a)-[*2..3]->(b) RETURN count(*) AS c", - "MATCH (a)-[*3..3]->(b) RETURN count(*) AS c", - "MATCH (a)-[*1..1]-(b) RETURN count(*) AS c", - "MATCH (a)-[*1..2]-(b) RETURN count(*) AS c", - "MATCH (a)-[*1..3]-(b) RETURN count(*) AS c", - "MATCH (a {kind:'a'})-[*1..2]-(b) RETURN count(*) AS c", - "MATCH (a {kind:'a'})-[*1..2]->(b) RETURN count(*) AS c", - "MATCH (a {kind:'a'})-[*2..3]->(b) RETURN count(*) AS c", - "MATCH (a)-[]->(b)-[*1..2]-(c) RETURN count(*) AS c", - "MATCH (a)-[]->(b)-[*2..3]->(c) RETURN count(*) AS c", -] - - -def test_bounded_varlen_differential_fuzz_against_pandas_oracle(): - """Random small graphs (cyclic, parallel edges, self-loops): match pandas or raise. - - Bounded windows only -- undirected UNBOUNDED shapes through the pandas oracle can blow the - box up (75 GiB RSS observed), and they are already declined elsewhere. Seeded so a failure - is reproducible; the shape list spans both sides of every gate boundary, so a gate that - declines too much shows up as the served-count assertion below going to zero. - """ - rnd = random.Random(1787) - served = 0 - for _ in range(25): - n_nodes = rnd.randint(3, 7) - nodes = pd.DataFrame({ - "id": list(range(n_nodes)), - "kind": [rnd.choice("ab") for _ in range(n_nodes)], - }) - edges = pd.DataFrame( - [(rnd.randrange(n_nodes), rnd.randrange(n_nodes)) for _ in range(rnd.randint(3, 10))], - columns=["s", "d"], - ) - g_pd, g_pl = _pair(nodes, edges) - for query in FUZZ_SHAPES: - expected = _count(g_pd, query, "pandas") - try: - got = _count(g_pl, query, "polars") - except NotImplementedError: - continue # an honest decline is allowed; a different answer is not - served += 1 - assert got == expected, ( - f"polars diverged from the pandas oracle on {query!r}: " - f"{got} != {expected}\nnodes={nodes.to_dict('list')}\nedges={edges.values.tolist()}" - ) - assert served > 100, f"gate declines too much to be a useful parity check ({served})" diff --git a/graphistry/tests/compute/gfql/test_varlen_bounded_engine_parity_1787.py b/graphistry/tests/compute/gfql/test_varlen_bounded_engine_parity_1787.py new file mode 100644 index 0000000000..0b25033303 --- /dev/null +++ b/graphistry/tests/compute/gfql/test_varlen_bounded_engine_parity_1787.py @@ -0,0 +1,427 @@ +"""Bounded variable-length segments: ONE contract, four engines (#1787). + +The contract is **parity with the pandas oracle, or ``NotImplementedError``** — never a +different answer with no error. Three bounded shapes broke it on the native polars row +pipeline, so they are now declined, mirroring what #1781 did for the unbounded case: + + 1. directed ``-[*k..m]->`` with ``min_hops >= 3`` (and ``min_hops >= 2`` off a filtered seed); + 2. undirected ``-[*1..1]-`` / ``-[*1]-``, the degenerate window, which HALVED the count; + 3. undirected ``-[*1..k]-`` that does not start from the full node set (filtered seed, or a + non-first segment), which OVER-counted. + +WHY THE DIVERGENCE FROM MASTER IS DELIBERATE, and the one thing no test below can state: on +master these shapes were *served*, so declining them is a visible behaviour change that turns +a silent wrong answer into a loud error. That is the intended direction — pandas is the +oracle, and an engine that cannot reproduce the oracle's edge multiplicity must say so. The +underlying limitation is a reconstruction gap, not a policy: pandas' ``step_pairs`` come from +the variable-length ``edge_op.execute`` hop, whose hop-window pruning (and, when seeded, its +per-seed BFS) changes edge multiplicity in a way a rebuild from the raw matching edge table +does not reproduce. When it becomes reconstructible, the gate should shrink and the +``*_declines_*`` tests below should be rewritten to parity tests. + +WHY THIS FILE IS ENGINE-PARAMETRIZED RATHER THAN polars-ONLY: "which shapes are answerable, +and with what value" is engine-agnostic semantics. It is parametrized over the four engines, +and it encodes the INTENDED PER-ENGINE behaviour rather than assuming identity — the polars +engines must DECLINE exactly where the pandas-API engines must ANSWER. Asserting agreement +everywhere would be false; asserting only polars would miss that cuDF is the other half of +the same contract. It already paid for itself: the cuDF divergence pinned at the bottom of +this file was found by adding the cuDF parameter. + +COVERAGE BOUNDARY, stated rather than hidden behind a skip: no CI lane runs cuDF or +polars-gpu (``ci-gpu.yml`` is hard-disabled and does not install the ``polars`` extra), so on +CI those two parameters report SKIPPED — visibly, via a runtime probe, never as a silent pass. +They are exercised out of band on the dgx GPU box against +``graphistry/test-rapids-official:26.02-gfql-polars`` (``docker run --gpus all`` — omitting +``--gpus all`` FABRICATES failures rather than skipping). Treat a green CI run as evidence for +pandas + polars only. +""" +from __future__ import annotations + +from functools import lru_cache +from typing import List, NamedTuple, Tuple + +import pandas as pd +import pytest + +import graphistry +from graphistry.Plottable import Plottable + +# Two families, not four independent engines. cuDF runs the pandas-API traversal, so it must +# ANSWER; polars-gpu is the GPU collect target of the SAME lazy polars engine, so it must +# decline wherever polars declines. Kept as fixed tuples rather than an importability probe: +# an engine that cannot run here must show up as a SKIPPED parameter, not vanish from the +# report as though it had never been in scope. +PANDAS_API_ENGINES: Tuple[str, ...] = ("pandas", "cudf") +POLARS_API_ENGINES: Tuple[str, ...] = ("polars", "polars-gpu") +ALL_ENGINES: Tuple[str, ...] = PANDAS_API_ENGINES + POLARS_API_ENGINES + + +class Shape(NamedTuple): + """One (graph, query) case plus the pandas-oracle answer, pinned as a literal. + + The oracle count is pinned rather than recomputed so a regression that makes every engine + equally wrong still fails — cross-engine agreement alone would pass it. + """ + + id: str + nodes: pd.DataFrame + edges: pd.DataFrame + query: str + oracle: int + + +# --- fixtures (from the #1787 differential fuzz) ------------------------------------------- + +BOUNDED_NODES = pd.DataFrame({"id": list(range(7))}) +BOUNDED_EDGES = pd.DataFrame( + [(4, 6), (0, 4), (1, 2), (4, 6), (5, 6), (4, 5), (4, 6)], columns=["s", "d"] +) +UNDIR_NODES = pd.DataFrame({"id": list(range(5))}) +UNDIR_EDGES = pd.DataFrame( + [(1, 2), (1, 2), (0, 2), (1, 0), (4, 2), (3, 2), (3, 4), (0, 3), (1, 2)], columns=["s", "d"] +) +SEEDED_NODES = pd.DataFrame({"id": list(range(7)), "kind": ["a", "b"] * 3 + ["a"]}) +SEEDED_EDGES = pd.DataFrame( + [(0, 3), (6, 3), (4, 2), (1, 3), (6, 3), (3, 1), (3, 5), (4, 1), (3, 3), (0, 3)], + columns=["s", "d"], +) + + +def _bounded(id_: str, query: str, oracle: int) -> Shape: + return Shape(id_, BOUNDED_NODES, BOUNDED_EDGES, query, oracle) + + +def _undir(id_: str, query: str, oracle: int) -> Shape: + return Shape(id_, UNDIR_NODES, UNDIR_EDGES, query, oracle) + + +def _seeded(id_: str, query: str, oracle: int) -> Shape: + return Shape(id_, SEEDED_NODES, SEEDED_EDGES, query, oracle) + + +# --- the contract, as data ----------------------------------------------------------------- + +#: Shapes the polars engines must DECLINE and the pandas-API engines must ANSWER correctly. +DECLINED_BY_POLARS: List[Shape] = [ + _bounded("dir-min3-exact", "MATCH (a)-[*3..3]->(b) RETURN count(*) AS c", 0), + _bounded("dir-min3-window", "MATCH (a)-[*3..4]->(b) RETURN count(*) AS c", 0), + _seeded("dir-min2-seeded", "MATCH (a {kind:'a'})-[*2..3]->(b) RETURN count(*) AS c", 32), + _undir("undir-degenerate-window", "MATCH (a)-[*1..1]-(b) RETURN count(*) AS c", 36), + _undir("undir-degenerate-exact", "MATCH (a)-[*1]-(b) RETURN count(*) AS c", 36), + _seeded("undir-seeded", "MATCH (a {kind:'a'})-[*1..2]-(b) RETURN count(*) AS c", 38), + _seeded("undir-non-first-segment", "MATCH (a)-[]->(b)-[*1..2]-(c) RETURN count(*) AS c", 316), +] + +#: Neighbours of every declined shape, on BOTH sides of each gate boundary. These are what +#: prove the gate is a scalpel: a lazily over-broad gate that declined "anything variable +#: length" would fail every one of them. +SERVED_EVERYWHERE: List[Shape] = [ + _bounded("dir-min1", "MATCH (a)-[*1..2]->(b) RETURN count(*) AS c", 12), + _bounded("dir-min2-unseeded", "MATCH (a)-[*2..3]->(b) RETURN count(*) AS c", 6), + _undir("undir-plain-edge", "MATCH (a)-[]-(b) RETURN count(*) AS c", 18), + _undir("undir-min1-max2", "MATCH (a)-[*1..2]-(b) RETURN count(*) AS c", 212), + _undir("undir-min1-max3", "MATCH (a)-[*1..3]-(b) RETURN count(*) AS c", 948), + _undir("dir-degenerate-window", "MATCH (a)-[*1..1]->(b) RETURN count(*) AS c", 9), + _seeded("dir-seeded-min1", "MATCH (a {kind:'a'})-[*1..2]->(b) RETURN count(*) AS c", 19), + _seeded("dir-non-first-segment", "MATCH (a)-[]->(b)-[*2..3]->(c) RETURN count(*) AS c", 80), +] + + +# --- engine plumbing ----------------------------------------------------------------------- + +_PROBE_NODES = pd.DataFrame({"id": [0, 1, 2]}) +_PROBE_EDGES = pd.DataFrame([(0, 1), (1, 2)], columns=["s", "d"]) + + +def _graph(engine: str, nodes: pd.DataFrame, edges: pd.DataFrame) -> Plottable: + """Build the same logical graph in the engine's native frame type.""" + if engine in POLARS_API_ENGINES: + pl = pytest.importorskip("polars") + return graphistry.nodes(pl.from_pandas(nodes), "id").edges(pl.from_pandas(edges), "s", "d") + if engine == "cudf": + cudf = pytest.importorskip("cudf") + return graphistry.nodes(cudf.from_pandas(nodes), "id").edges( + cudf.from_pandas(edges), "s", "d" + ) + return graphistry.nodes(nodes, "id").edges(edges, "s", "d") + + +@lru_cache(maxsize=None) +def _engine_runnable(engine: str) -> bool: + """Probe by RUNNING the smallest version of what these tests do. + + Cheaper probes do not discriminate on a box with cudf/cudf_polars importable but no + working CUDA runtime — frame construction and simple ops all succeed there and the suite + then dies inside the first real kernel. So the probe is an actual traversal. + """ + try: + g = _graph(engine, _PROBE_NODES, _PROBE_EDGES) + g.gfql("MATCH (a)-[]->(b) RETURN count(*) AS c", engine=engine) + return True + except Exception: # noqa: BLE001 — any failure here means "cannot run", never "test fails" + return False + + +def _require(engine: str) -> None: + if not _engine_runnable(engine): + pytest.skip( + f"engine {engine!r} is not runnable in this environment — NOT evidence that it " + "passes; see the COVERAGE BOUNDARY note in this module's docstring" + ) + + +def _count(g: Plottable, query: str, engine: str) -> int: + """Scalar ``count(*)`` out of any engine's result frame. + + Dispatched on the engine rather than on ``hasattr(col, "to_pandas")``: polars Series + ALSO expose ``to_pandas()``, but it needs pyarrow, which the polars test lane does not + install — so an attribute probe would silently route polars down a path that can fail + on a dependency this file has no reason to require. + """ + col = g.gfql(query, engine=engine)._nodes["c"] + return int(col.to_pandas().iloc[0]) if engine == "cudf" else int(col.to_list()[0]) + + +def _answer(engine: str, shape: Shape) -> int: + return _count(_graph(engine, shape.nodes, shape.edges), shape.query, engine) + + +_DECLINED_IDS = [s.id for s in DECLINED_BY_POLARS] +_SERVED_IDS = [s.id for s in SERVED_EVERYWHERE] + + +# --- the contract, as tests ---------------------------------------------------------------- + + +@pytest.mark.parametrize("engine", POLARS_API_ENGINES) +@pytest.mark.parametrize("shape", DECLINED_BY_POLARS, ids=_DECLINED_IDS) +def test_unreconstructible_shape_declines_on_the_polars_engines(shape: Shape, engine: str) -> None: + """A shape whose multiplicity is not reconstructible must RAISE, never answer differently.""" + _require(engine) + with pytest.raises(NotImplementedError): + _answer(engine, shape) + + +@pytest.mark.parametrize("engine", PANDAS_API_ENGINES) +@pytest.mark.parametrize("shape", DECLINED_BY_POLARS, ids=_DECLINED_IDS) +def test_unreconstructible_shape_is_still_answered_on_the_pandas_api_engines( + shape: Shape, engine: str +) -> None: + """Declining is an ENGINE limit, not a query rejection: pandas and cuDF still answer.""" + _require(engine) + assert _answer(engine, shape) == shape.oracle + + +@pytest.mark.parametrize("engine", ALL_ENGINES) +@pytest.mark.parametrize("shape", SERVED_EVERYWHERE, ids=_SERVED_IDS) +def test_neighbouring_shape_is_answered_identically_on_every_engine( + shape: Shape, engine: str +) -> None: + """The gate is a scalpel: every neighbour of a declined shape still runs, and agrees.""" + _require(engine) + assert _answer(engine, shape) == shape.oracle + + +# --- the boundary claims the gate is built on ---------------------------------------------- +# Each of these was a prose comment in row_pipeline.py. A comment asserting behaviour is +# unverified and rots; these do not. + + +@pytest.mark.parametrize("engine", ALL_ENGINES) +def test_degenerate_window_is_not_the_same_query_as_a_plain_edge(engine: str) -> None: + """WHY the gate keys on an EXPLICIT window instead of ``sem.is_multihop``. + + ``-[*1..1]-`` resolves to min == max == 1 and is therefore NOT multihop, yet pandas still + routes it through the variable-length hop and gets a different answer from ``-[]-`` on the + very same graph (36 vs 18 — the undirected doubling). Gating on ``is_multihop`` would have + let the degenerate window straight through, which is exactly why it went unnoticed. + """ + _require(engine) + plain = _count(_graph(engine, UNDIR_NODES, UNDIR_EDGES), + "MATCH (a)-[]-(b) RETURN count(*) AS c", engine) + assert plain == 18, "the plain undirected edge is served by every engine, unchanged" + if engine in POLARS_API_ENGINES: + with pytest.raises(NotImplementedError): + _count(_graph(engine, UNDIR_NODES, UNDIR_EDGES), + "MATCH (a)-[*1..1]-(b) RETURN count(*) AS c", engine) + else: + degenerate = _count(_graph(engine, UNDIR_NODES, UNDIR_EDGES), + "MATCH (a)-[*1..1]-(b) RETURN count(*) AS c", engine) + assert degenerate == 36 != plain + + +@pytest.mark.parametrize("engine", PANDAS_API_ENGINES) +def test_directed_min_hops_3_collapses_to_empty_on_the_oracle(engine: str) -> None: + """The observable consequence of ``max_reached_hop`` being a BFS eccentricity. + + ``compute/hop.py`` prunes against a dedup-by-node eccentricity, not a longest-walk length, + so on this graph the oracle returns NOTHING at ``min_hops >= 3`` while still returning rows + at ``min_hops == 2``. A raw-edge rebuild expands a different edge multiset and cannot land + on empty, which is why ``min_hops >= 3`` is unreconstructible rather than merely narrower. + """ + _require(engine) + g = _graph(engine, BOUNDED_NODES, BOUNDED_EDGES) + assert _count(g, "MATCH (a)-[*2..3]->(b) RETURN count(*) AS c", engine) == 6 + assert _count(g, "MATCH (a)-[*3..3]->(b) RETURN count(*) AS c", engine) == 0 + assert _count(g, "MATCH (a)-[*3..4]->(b) RETURN count(*) AS c", engine) == 0 + + +@pytest.mark.parametrize("engine", POLARS_API_ENGINES) +def test_directed_min_hops_2_declines_only_when_the_segment_is_seeded(engine: str) -> None: + """``min_hops == 2`` is fuzz-clean off the FULL node set and only diverges under a seed. + + The seeded hop runs a per-seed BFS, which is what changes the multiplicity. Pinning both + sides keeps the gate from being widened to all of ``min_hops >= 2`` on a hunch — the + unseeded form is the graph-bench ``-[*1..k]->`` shape and must stay native. + """ + _require(engine) + g = _graph(engine, SEEDED_NODES, SEEDED_EDGES) + assert _count(g, "MATCH (a)-[*2..3]->(b) RETURN count(*) AS c", engine) == 50 + with pytest.raises(NotImplementedError): + _count(g, "MATCH (a {kind:'a'})-[*2..3]->(b) RETURN count(*) AS c", engine) + + +@pytest.mark.parametrize("engine", POLARS_API_ENGINES) +def test_undirected_seed_declines_while_its_directed_twin_is_served(engine: str) -> None: + """The over-count is specific to the undirected DOUBLING, not to seeding. + + Same seed, same window, same graph: the directed spelling agrees with pandas and stays + native, so the gate is correctly restricted to ``direction == "undirected"``. + """ + _require(engine) + g = _graph(engine, SEEDED_NODES, SEEDED_EDGES) + assert _count(g, "MATCH (a {kind:'a'})-[*1..2]->(b) RETURN count(*) AS c", engine) == 19 + with pytest.raises(NotImplementedError): + _count(g, "MATCH (a {kind:'a'})-[*1..2]-(b) RETURN count(*) AS c", engine) + + +@pytest.mark.parametrize("engine", POLARS_API_ENGINES) +def test_undirected_non_first_segment_declines_while_its_directed_twin_is_served( + engine: str, +) -> None: + """A non-first segment starts from less than the full node set, exactly like a seed.""" + _require(engine) + g = _graph(engine, SEEDED_NODES, SEEDED_EDGES) + assert _count(g, "MATCH (a)-[]->(b)-[*1..2]->(c) RETURN count(*) AS c", engine) == 50 + with pytest.raises(NotImplementedError): + _count(g, "MATCH (a)-[]->(b)-[*1..2]-(c) RETURN count(*) AS c", engine) + + +@pytest.mark.parametrize("engine", POLARS_API_ENGINES) +def test_degenerate_window_decline_is_undirected_only(engine: str) -> None: + """``-[*1..1]->`` (DIRECTED) has no doubling to get wrong and must stay native.""" + _require(engine) + g = _graph(engine, UNDIR_NODES, UNDIR_EDGES) + assert _count(g, "MATCH (a)-[*1..1]->(b) RETURN count(*) AS c", engine) == 9 + + +# --- the fuzz that found all three ---------------------------------------------------------- + +FUZZ_SHAPES: Tuple[str, ...] = ( + "MATCH (a)-[*1..2]->(b) RETURN count(*) AS c", + "MATCH (a)-[*2..2]->(b) RETURN count(*) AS c", + "MATCH (a)-[*2..3]->(b) RETURN count(*) AS c", + "MATCH (a)-[*3..3]->(b) RETURN count(*) AS c", + "MATCH (a)-[*1..1]-(b) RETURN count(*) AS c", + "MATCH (a)-[*1..2]-(b) RETURN count(*) AS c", + "MATCH (a)-[*1..3]-(b) RETURN count(*) AS c", + "MATCH (a {kind:'a'})-[*1..2]-(b) RETURN count(*) AS c", + "MATCH (a {kind:'a'})-[*1..2]->(b) RETURN count(*) AS c", + "MATCH (a {kind:'a'})-[*2..3]->(b) RETURN count(*) AS c", + "MATCH (a)-[]->(b)-[*1..2]-(c) RETURN count(*) AS c", + "MATCH (a)-[]->(b)-[*2..3]->(c) RETURN count(*) AS c", +) + + +def _random_graph(seed: int) -> Tuple[pd.DataFrame, pd.DataFrame]: + import random + + rnd = random.Random(seed) + n_nodes = rnd.randint(3, 7) + nodes = pd.DataFrame({ + "id": list(range(n_nodes)), + "kind": [rnd.choice("ab") for _ in range(n_nodes)], + }) + edges = pd.DataFrame( + [(rnd.randrange(n_nodes), rnd.randrange(n_nodes)) for _ in range(rnd.randint(3, 10))], + columns=["s", "d"], + ) + return nodes, edges + + +@pytest.mark.parametrize("engine", [e for e in ALL_ENGINES if e != "pandas"]) +def test_bounded_varlen_differential_fuzz_against_pandas_oracle(engine: str) -> None: + """Random small graphs (cyclic, parallel edges, self-loops): match pandas, or raise. + + Bounded windows only — undirected UNBOUNDED shapes through the pandas oracle can blow the + box up (75 GiB RSS observed) and are already declined elsewhere. Seeded, so a failure is + reproducible. The shape list spans both sides of every gate boundary, so a gate that + declines too much shows up as the served-count assertion at the end going to zero. + """ + _require(engine) + served = 0 + for seed in range(25): + nodes, edges = _random_graph(1787 + seed) + g_pd = _graph("pandas", nodes, edges) + g_engine = _graph(engine, nodes, edges) + for query in FUZZ_SHAPES: + expected = _count(g_pd, query, "pandas") + try: + got = _count(g_engine, query, engine) + except NotImplementedError: + continue # an honest decline is allowed; a different answer is not + served += 1 + assert got == expected, ( + f"{engine} diverged from the pandas oracle on {query!r}: " + f"{got} != {expected}\nnodes={nodes.to_dict('list')}" + f"\nedges={edges.values.tolist()}" + ) + assert served > 100, f"{engine} declines too much to be a useful parity check ({served})" + + +# --- a divergence this cross-engine parametrization FOUND ------------------------------------ + +# Minimal repro reduced from the fuzz: `{kind:'a'}` selects nodes 1 and 2. +CUDF_DIVERGENCE_NODES = pd.DataFrame({"id": [0, 1, 2, 3], "kind": ["b", "a", "a", "b"]}) +CUDF_DIVERGENCE_EDGES = pd.DataFrame( + [(2, 2), (0, 3), (2, 2), (3, 1), (0, 3), (2, 2), (1, 1)], columns=["s", "d"] +) + + +# The cuDF parameter carries the xfail at COLLECTION time (a marker added inside the test body +# is applied too late to be reliable), and STRICT so the fix cannot land unnoticed. +_DEGENERATE_SEED_ENGINES = [ + pytest.param( + engine, + marks=pytest.mark.xfail( + strict=True, + reason="#1798: cuDF answers 1 where the pandas oracle answers 9", + ), + ) + if engine == "cudf" + else engine + for engine in ALL_ENGINES +] + + +@pytest.mark.parametrize("engine", _DEGENERATE_SEED_ENGINES) +def test_seeded_undirected_degenerate_window_agrees_with_the_oracle(engine: str) -> None: + """``(a {..})-[*1..1]-(b)``: a SEPARATE cuDF defect, in the pandas-API family (#1798). + + Not the #1787 gap and not fixed by this PR — the polars engines decline this shape (it is + the ``undir-degenerate-window`` case above with a seed), so the polars fix cannot mask it. + cuDF answers 1 where pandas answers 9, silently: 23 of 40 random graphs diverge on this + shape alone, while the UNSEEDED ``-[*1..1]-``, the seeded ``-[]-``, the seeded + ``-[*1..2]-`` and the directed ``-[*1..1]->`` are all clean (0/40 each). + + Recorded as a STRICT xfail rather than left out: when #1798 lands, this test fails loudly + and must be un-xfailed, whereas a comment or a skip would rot silently. + """ + _require(engine) + g = _graph(engine, CUDF_DIVERGENCE_NODES, CUDF_DIVERGENCE_EDGES) + query = "MATCH (a {kind:'a'})-[*1..1]-(b) RETURN count(*) AS c" + if engine in POLARS_API_ENGINES: + with pytest.raises(NotImplementedError): + _count(g, query, engine) + return + assert _count(g, query, engine) == 9