diff --git a/.github/workflows/ci.yml b/.github/workflows/ci.yml index f54c9a439..9af248e00 100644 --- a/.github/workflows/ci.yml +++ b/.github/workflows/ci.yml @@ -287,6 +287,16 @@ jobs: matrix: os: [ubuntu-latest, windows-latest, macos-latest] steps: + # Every test binary of the debug build links the whole engine, which + # nearly fills an ubuntu runner's disk: remove preinstalled SDKs that + # this job does not use. + - name: Free disk space + if: runner.os == 'Linux' + run: | + df -h / + sudo rm -rf /usr/share/dotnet /usr/local/lib/android /opt/ghc /opt/hostedtoolcache/CodeQL + df -h / + - uses: actions/checkout@v7 - name: Install Rust toolchain @@ -298,31 +308,38 @@ jobs: - name: Configure sccache uses: mozilla-actions/sccache-action@v0.0.11 + # Crate sources only: archiving this job's target directory filled the + # runner's disk, and sccache keeps the compiled crates between runs. - name: Cache cargo registry uses: actions/cache@v6 with: path: | ~/.cargo/registry ~/.cargo/git - target - key: ${{ runner.os }}-cargo-test-${{ hashFiles('**/Cargo.lock') }} + key: ${{ runner.os }}-cargo-registry-${{ hashFiles('**/Cargo.lock') }} restore-keys: | - ${{ runner.os }}-cargo-test- + ${{ runner.os }}-cargo-registry- # Full profile: --all-features tests every language parser, AI index, # RDF store, algorithms, CDC, and other optional modules together. # This catches integration issues across feature combinations. # Spec tests are excluded here as they run in the dedicated rust-spec job. + # Line tables keep file and line in backtraces at a fraction of the size + # of full debug info. - name: Run tests run: cargo nextest run --all-features --workspace --exclude grafeo-spec-tests env: RUSTC_WRAPPER: sccache + SCCACHE_GHA_ENABLED: "true" + CARGO_PROFILE_DEV_DEBUG: line-tables-only # nextest does not run doc-tests, so run them separately - name: Run doc-tests run: cargo test --all-features --workspace --doc env: RUSTC_WRAPPER: sccache + SCCACHE_GHA_ENABLED: "true" + CARGO_PROFILE_DEV_DEBUG: line-tables-only # Rust tests (release, ubuntu only) # Release-mode catches issues from inlining and overflow checks. @@ -386,6 +403,8 @@ jobs: - profile: rdf features: "--features gql,sparql,graphql,triple-store,spill,mmap,regex" engine_features: "--features gql,sparql,graphql,triple-store,shacl,wal,grafeo-file,spill,mmap,regex" + # The facade's `rdf` profile on its own keeps its data across a reopen (#544). + facade_test: "--features rdf --test rdf_profile" # Analytics persona: Data Scientist (algorithms + search + bulk import) - profile: analytics features: "--features algos,vector-index,text-index,hybrid-search,jsonl-import,parquet-import" @@ -456,6 +475,17 @@ jobs: env: RUSTC_WRAPPER: sccache + # A facade test built with only the profile's features, for profiles + # whose behavior depends on what the profile itself enables. + - name: Test facade profile (${{ matrix.profile }}) + if: matrix.facade_test + run: > + cargo test + -p grafeo + --no-default-features ${{ matrix.facade_test }} + env: + RUSTC_WRAPPER: sccache + # Miri undefined behavior check miri: name: Miri UB Check diff --git a/.github/workflows/release.yml b/.github/workflows/release.yml index 6b2333cb1..2c54e82b2 100644 --- a/.github/workflows/release.yml +++ b/.github/workflows/release.yml @@ -194,7 +194,6 @@ jobs: uses: softprops/action-gh-release@v3 with: files: ./checksums.txt - continue-on-error: true # Publish crates to crates.io (in dependency order) publish-crates: @@ -350,6 +349,10 @@ jobs: name: Publish to npm runs-on: ubuntu-latest needs: [create-release, build-node-native] + environment: npm + permissions: + contents: read + id-token: write steps: - uses: actions/checkout@v7 @@ -381,30 +384,31 @@ jobs: echo "=== Platform packages ===" find npm -name '*.node' -ls + - name: Update npm (trusted publishing needs 11.5.1 or later) + run: npm install -g npm@^11.5.1 + - name: Publish platform packages working-directory: crates/bindings/node - env: - NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }} run: | for dir in npm/*/; do if ls "$dir"*.node 1>/dev/null 2>&1; then - echo "Publishing $(basename "$dir")..." - (cd "$dir" && npm publish --access public) || true + (cd "$dir" && bash "$GITHUB_WORKSPACE/scripts/npm-publish.sh") fi done - name: Publish @grafeo-db/js working-directory: crates/bindings/node - run: npm publish --access public --ignore-scripts - env: - NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }} - continue-on-error: true + run: bash "$GITHUB_WORKSPACE/scripts/npm-publish.sh" --ignore-scripts # Publish WASM package to npm publish-wasm: name: Publish WASM to npm runs-on: ubuntu-latest needs: create-release + environment: npm + permissions: + contents: read + id-token: write steps: - uses: actions/checkout@v7 @@ -431,18 +435,22 @@ jobs: # Copy our maintained package.json over the generated one. cp crates/bindings/wasm/package.json crates/bindings/wasm/pkg/package.json + - name: Update npm (trusted publishing needs 11.5.1 or later) + run: npm install -g npm@^11.5.1 + - name: Publish @grafeo-db/wasm working-directory: crates/bindings/wasm/pkg - run: npm publish --access public - env: - NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }} - continue-on-error: true + run: bash "$GITHUB_WORKSPACE/scripts/npm-publish.sh" # Publish WASM lite package to npm publish-wasm-lite: name: Publish WASM Lite to npm runs-on: ubuntu-latest needs: create-release + environment: npm + permissions: + contents: read + id-token: write steps: - uses: actions/checkout@v7 @@ -466,18 +474,22 @@ jobs: - name: Fix WASM lite package metadata run: cp crates/bindings/wasm/package-lite.json crates/bindings/wasm/pkg-lite/package.json + - name: Update npm (trusted publishing needs 11.5.1 or later) + run: npm install -g npm@^11.5.1 + - name: Publish @grafeo-db/wasm-lite working-directory: crates/bindings/wasm/pkg-lite - run: npm publish --access public - env: - NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }} - continue-on-error: true + run: bash "$GITHUB_WORKSPACE/scripts/npm-publish.sh" # Publish CLI npm wrapper packages publish-cli-npm: name: Publish CLI to npm runs-on: ubuntu-latest needs: build-binaries + environment: npm + permissions: + contents: read + id-token: write strategy: fail-fast: false matrix: @@ -544,18 +556,22 @@ jobs: chmod +x ${{ matrix.binary }} 2>/dev/null || true ls -la + - name: Update npm (trusted publishing needs 11.5.1 or later) + run: npm install -g npm@^11.5.1 + - name: Publish platform package working-directory: platform-pkg - run: npm publish --access public - env: - NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }} - continue-on-error: true + run: bash "$GITHUB_WORKSPACE/scripts/npm-publish.sh" # Publish main CLI npm launcher (after platform packages) publish-cli-npm-launcher: name: Publish @grafeo-db/cli runs-on: ubuntu-latest needs: publish-cli-npm + environment: npm + permissions: + contents: read + id-token: write steps: - uses: actions/checkout@v7 @@ -565,12 +581,12 @@ jobs: node-version: '24' registry-url: 'https://registry.npmjs.org' + - name: Update npm (trusted publishing needs 11.5.1 or later) + run: npm install -g npm@^11.5.1 + - name: Publish launcher package working-directory: packages/grafeo-cli-npm - run: npm publish --access public - env: - NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }} - continue-on-error: true + run: bash "$GITHUB_WORKSPACE/scripts/npm-publish.sh" # Publish C# bindings to NuGet publish-nuget: @@ -608,8 +624,7 @@ jobs: run: dotnet pack -c Release -o ../../nupkg - name: Publish to NuGet - run: dotnet nuget push "crates/bindings/csharp/nupkg/*.nupkg" --api-key ${{ secrets.NUGET_API_KEY }} --source https://api.nuget.org/v3/index.json - continue-on-error: true + run: dotnet nuget push "crates/bindings/csharp/nupkg/*.nupkg" --api-key ${{ secrets.NUGET_API_KEY }} --source https://api.nuget.org/v3/index.json --skip-duplicate # Build CLI Python wheels using workflow artifacts (no release asset race) build-cli-wheels: @@ -700,7 +715,7 @@ jobs: uses: pypa/gh-action-pypi-publish@release/v1 with: packages-dir: dist/ - continue-on-error: true + skip-existing: true # Notify Discord about the new release notify-discord: diff --git a/CHANGELOG.md b/CHANGELOG.md index 390d51c13..bae998f3a 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -2,103 +2,111 @@ All notable changes to Grafeo, for future reference (and enjoyment). -## [0.5.44] - Unreleased +## [0.5.44] - 2026-10-04 -> **Heads-up: 0.5.45 changes the on-disk format.** The `.grafeo` container, its catalog and the WAL move to new formats in one step. 0.5.45 migrates a database automatically the first time it opens it, and WAL-directory databases (paths without the `.grafeo` extension) become a single `.grafeo` file. After that, 0.5.44 and older can no longer open the database, so keep a backup if you may need to go back. +Durability and consistency release. Crash-safe checkpoints and WAL recovery, indexes and constraints that survive a reopen, schema checks on every write path, per-graph conflicts and grants, commit and rollback in O(changes), and fixes for shortest paths, variable-length edges, list comprehensions, subqueries, `OPTIONAL MATCH` and `MERGE`. Plus graph handles, upserts by key and write counters. + +> **Heads-up: 0.5.45 changes the on-disk format.** 0.5.45 migrates a database automatically on first open (WAL-directory databases become a single `.grafeo` file); after that, 0.5.44 and older can no longer open it, so keep a backup if you may need to go back. ### Changed -- **Breaking (Rust API, `grafeo-engine`): `RdfPlanner::with_wal` is no longer public.** The planner now records into the session's WAL buffer instead of the WAL. -- **Breaking (Rust API, `grafeo-core`): the mutation operators write through a `GraphWriter`.** `CreateNodeOperator`, `CreateEdgeOperator`, `DeleteNodeOperator`, `DeleteEdgeOperator`, `AddLabelOperator`, `RemoveLabelOperator`, `SetPropertyOperator`, `MergeOperator` and `MergeRelationshipOperator` take the transaction context, constraint validator and write tracker through a `GraphWriter` passed to their constructor (a plain store still works) instead of their own `with_transaction_context`, `with_validator` and `with_write_tracker`. `SetPropertyOperator::with_labels` and `with_edge_type` are gone: the checks use the entity's own labels and type. -- **Breaking (Rust API, `grafeo-engine`): the direct write methods return `Result`.** `create_node`, `create_node_with_props`, `create_edge`, `create_edge_with_props`, `set_node_property`, `set_edge_property`, `remove_node_property`, `remove_edge_property`, `add_node_label`, `remove_node_label`, `delete_node`, `delete_edge`, `batch_create_nodes` and `batch_create_nodes_with_props` on `GrafeoDB` and `Session` now return a `Result`. Each call commits on its own, with the checks of a query, and it fails instead of writing something invalid: `set_node_property` and `set_edge_property` on an entity that does not exist (silently ignored before), `create_edge` with an endpoint that does not exist (it created an edge to nowhere) and `delete_node` on a node that still has edges (it deleted the node and left its edges behind) are now errors. The Python, Node.js, C and WASM bindings raise or return these errors. -- **Breaking (Rust API, `grafeo-engine`): the transaction manager tracks writes per graph.** `TransactionManager::get_write_set`, `reset_write_set` and the `write_set` and `read_set` of `TransactionInfo` hold the new `GraphEntity` (the graph and the node or edge); `record_write` and `record_read` still take a `NodeId`, `EdgeId` or `EntityId` for the default graph. -- **CDC label change events list the labels after the change**: removing a label reported the labels before the removal in `labels`, while adding one reported the labels after it. `labels` is now always the labels after the change, and the new `before_labels` (a new field of the Rust `ChangeEvent`) holds the labels before it. +- **Breaking (Rust, `grafeo-engine`): the direct write methods return `Result`** (`create_node`, `create_edge`, `set_node_property`, `delete_node`, the batch calls and the rest, on `GrafeoDB` and `Session`). Each call commits on its own with the checks of a query, so writing to a missing entity or endpoint, or deleting a node that still has edges, now fails. The bindings raise these errors. +- **Breaking (Rust, `grafeo-core`): the mutation operators take a `GraphWriter`** instead of `with_transaction_context`, `with_validator` and `with_write_tracker`; `SetPropertyOperator::with_labels` and `with_edge_type` are gone. +- **Breaking (Rust, `grafeo-core`): operators that only order, cut, deduplicate or compare rows take no output schema** (`LimitOperator`, `SkipOperator`, `LimitSkipOperator`, `SortOperator`, `TopKOperator`, `DistinctOperator`, `ShuffleOperator`, `ExceptOperator`, `IntersectOperator`). +- **Breaking (Rust, `grafeo-core`): a sort key's `NullOrder` no longer flips for a descending key** (`SortKey::descending` sets `NullsFirst`), and `value_utils::compare_values_with_nulls` is replaced by `compare_sort_values`. +- **Breaking (Rust, `grafeo-core`): `FilterExpression::ExistsSubquery` and `CountSubquery` have `end_var` and `edge_var`**, and `ExistsSubquery` reads `min_hops` and `max_hops`. +- **Breaking (Rust, `grafeo-engine`)**: the transaction manager's write and read sets hold `GraphEntity` (graph plus node or edge), the logical `ExpandOp` has a `quantified` field, and `RdfPlanner::with_wal` is no longer public. +- **Breaking (Rust, `grafeo-engine`): `ChangeEvent` has a `graph` field** (`None` for the default graph), and `CdcLog` keys its events by graph and entity, with `history_in` and `history_since_in` for a named graph. +- **Breaking (Rust, `grafeo-adapters`): the GQL `QueryClause::InlineCall` has `scope` and `combined` fields, and the Cypher `Clause::CallSubquery` is a struct variant** with `query`, `scope`, `unions` and `union_all`. +- **CDC label change events**: `labels` is now always the labels after the change, and the new `before_labels` holds the previous ones. +- **Some queries that ran now fail with an error, as in openCypher**: an expression in `WITH` without a name; a `CALL` subquery that returns an outer variable or reads one it does not import (`CALL () { ... }`, a Cypher `CALL { ... }` without an importing `WITH`); an importing `WITH` with a `WHERE`, alias, `ORDER BY`, `SKIP` or `LIMIT`; `UNION` mixed with `UNION ALL`; and a Cypher query that ends with a `CALL` subquery that returns rows (add a `RETURN` after it). Before, they gave wrong or silently null results, or failed with an internal error (see Fixed). ### Fixed -- **Some checkpoints left parts of the database out of the `.grafeo` file**: checkpoints of a compacted database (after `compact()`), including the one at `close()`, left out the schema, indexes and RDF data, so they were gone after reopening. The periodic checkpoint timer (`checkpoint_interval`) kept writing the store from before `compact()`, so a crash after its checkpoint lost the compacted data, and `async_write_snapshot()` (the `async-storage` feature, used by grafeo-server) wrote nothing on a normal database and only the deletions on a compacted one. Every checkpoint now writes the whole database. -- **A session whose graph was dropped kept running on the default graph**: after another session dropped the graph a session had selected with `USE GRAPH` (or its schema), that session's queries read and wrote the default graph instead. They now fail with `Graph '...' does not exist` until the session selects another graph, and so do the direct calls after `set_current_graph` selected a graph that was dropped since. -- **Per-graph grants did not hold for a graph selected with `use_graph()`**: the `USE GRAPH` statement checks the identity's grants, but `Session::use_graph()` did not, so the session's statements and direct writes then ran on a graph it had no grant for; and a read-only grant did not stop writes to its graph. Every statement and direct write now needs a grant for the selected graph, a read-write one to write. -- **Read-only transactions, roles and databases did not stop every write** ([#413](https://github.com/GrafeoDB/grafeo/issues/413)): inside `START TRANSACTION READ ONLY`, Cypher, Gremlin, GraphQL and SPARQL writes succeeded (only GQL was rejected), and the direct write API (`create_node`, `set_node_property`, `delete_node`, the batch calls) wrote through a read-only transaction, a session with the `ReadOnly` role and a database opened with `open_read_only()`. These writes now fail in every language and through the direct API. -- **Sessions after `compact()` ignored parts of the database**: SPARQL writes through a session went to a store of their own and were lost at once, queries were not reported to CDC, and the graph selected with `set_current_graph` was ignored. Sessions of a compacted database now use its RDF store, CDC and selected graph. Queries after `compact()` (and after reopening a compacted `.grafeo` file) are still not written to the WAL, so on a persistent database a crash loses them, and a WAL-directory database loses them even on a clean close; the direct API is logged, and a checkpoint or clean close of a `.grafeo` file keeps everything. After a crash, a compacted file also replays only what direct calls created since its last checkpoint, not their updates or deletes of data from before `compact()`. -- **Direct reads missed a compacted database's data and panicked on an external store**: after `compact()`, `get_node`, `get_edge`, `get_node_labels`, `find_nodes_by_property`, `node_count` and `edge_count` did not see the nodes and edges from before the compaction, which queries still found. `info()` and `detailed_stats()` counted only what was written since, and `validate()` reported an edge from a compacted node as dangling. On a database built with `with_store` or `with_read_store`, these reads, `current_epoch()`, `create_graph()` and `list_graphs()` panicked. The reads now see what queries see; `create_graph()` returns an error there and `list_graphs()` an empty list. -- **Reopening a `.grafeo` database or copying one with `to_memory()` lost its indexes**: property indexes (of every graph), vector indexes and text indexes had to be created again after each open, although the file held their definitions and the saved vector and text index data. `to_memory()` and `open_in_memory()` also dropped the schema, the constraints, the index names and (with the `temporal` feature) the property history, so the copy accepted duplicate `UNIQUE` values, and the copy of a compacted database (after `compact()`) held only what was written after the compaction. Opening a `.grafeo` file now brings back the indexes of every graph, loading vector and text indexes from their saved data instead of rebuilding them, and `to_memory()` loads its copy the same way, so the copy holds what a reopen would. Indexes created with `CREATE INDEX` keep their names for `DROP INDEX`. WAL-directory databases (paths without the `.grafeo` extension) still reopen without their indexes until 0.5.45 ([#401](https://github.com/GrafeoDB/grafeo/issues/401)). -- **An embedding changed while its vector index was spilled to disk reverted at reload** ([#522](https://github.com/GrafeoDB/grafeo/issues/522)): after the embeddings were spilled (under memory pressure or with `TierOverride::ForceDisk`), bringing them back with `reload_eligible()`, also after a close and reopen, overwrote embeddings set in the meantime with their spilled values. The newer values now stay. -- **`create_graph()` and `drop_graph()` were not written to the WAL**: after the WAL was replayed (a crash, or reopening a WAL-directory database), an empty graph created with `create_graph()` was gone, and a graph dropped with `drop_graph()` came back with its data. Both are now logged like `CREATE GRAPH` and `DROP GRAPH`. -- **Backups missed writes made while they ran**: records written while `backup_full()` or `backup_incremental()` was running (`/admin/backup` in grafeo-server) ended up in neither that backup nor the next one, so a restore silently lacked them; with writes going on, a restore could miss a sizeable share of them. A full backup now covers exactly what its copied file holds, and an incremental backup starts a new WAL file before it reads, so later records go to the next backup. -- **`DurabilityMode::Adaptive` never synced the WAL**: the background thread that syncs it at the configured interval was never started, so the mode behaved like `NoSync`, and a power loss or operating system crash could lose any commit. The thread now runs from open until `close()`. -- **A checkpoint that failed midway could leave a `.grafeo` file that no longer opened** ([#418](https://github.com/GrafeoDB/grafeo/issues/418)): a full disk or a crash while `close()`, `wal_checkpoint()` or a periodic checkpoint was writing the file overwrote the previous state in place, and the next open failed with `section ... CRC mismatch`. A checkpoint now writes the new state to a separate `.checkpoint` first and only then copies it over the database file, so a failed checkpoint leaves the last good state readable, and the next open finishes one that was interrupted. A checkpoint needs free disk space for a second copy of the file while it runs. -- **Edges appeared twice after reopening a `.grafeo` database that was checkpointed and then not closed cleanly** ([#417](https://github.com/GrafeoDB/grafeo/issues/417)): after `wal_checkpoint()`, a full backup (`backup_full()`, `/admin/backup` in grafeo-server) or a periodic checkpoint, followed by a crash or a stop without `close()`, reopening replayed WAL records the file already contained, and `MATCH` returned those edges twice. Replay now skips nodes and edges that already exist, and a normal reopen repairs databases affected by this. Checkpoints also update the WAL only after the file is safely written, then remove the WAL files the file covers (keeping those an incremental backup still needs), so the sidecar WAL no longer grows until `close()`. -- **Direct writes were not durable until `close()`** ([#395](https://github.com/GrafeoDB/grafeo/issues/395)): nodes, edges, properties and labels written with the direct API outside a transaction (for example `createNode()` in Node.js or `create_node()` in Python) were lost after a crash, even after `wal_checkpoint()`. Each call is now durable when it returns, and the batch calls (`batch_create_nodes`, `batch_create_nodes_with_props`) are recovered completely or not at all. -- **Concurrent transactions could corrupt WAL recovery** ([#411](https://github.com/GrafeoDB/grafeo/issues/411)): WAL records of different sessions interleaved, so after a crash one session's rollback could erase another session's committed writes, and one session's commit could bring back another's rolled-back writes. Transactions still open at `close()` were committed on reopen, and writes undone by a rollback to a savepoint came back. Each transaction's changes are now written to the WAL as one group when it commits, a rollback writes nothing, and a transaction cut off by a crash is discarded on the next open. Writes outside a transaction through a session, schema changes and database-level SPARQL updates are written as soon as they are applied, and direct session writes to a named graph now replay into that graph. -- **Two processes could open the same directory database and silently overwrite each other's writes** ([#405](https://github.com/GrafeoDB/grafeo/issues/405)): databases on a path without the `.grafeo` extension had no lock, so the process that closed last discarded the other's commits. They are now locked like `.grafeo` files: a second open, from any process, fails with `database is locked by another process` until the first one closes. -- **A statement that failed inside a transaction kept what it wrote before failing**: in `UNWIND [5, 'x'] AS v INSERT (:Doc {id: v})` on an integer `id`, the node for 5 stayed after the second row failed and `commit()` made it durable, and a `MERGE` whose `ON CREATE SET` failed left its node. A failed statement or direct call inside a transaction is now undone completely, and the transaction goes on. -- **A commit that failed with a write-write conflict left the transaction active** ([#409](https://github.com/GrafeoDB/grafeo/issues/409)): every later write to the same nodes or edges failed with a conflict until the process restarted, and on a persistent database the failed transaction's writes could reappear after reopen. A failed commit now aborts the transaction completely. -- **Commit and rollback got slower as the database grew** ([#410](https://github.com/GrafeoDB/grafeo/issues/410)): beginning, committing and rolling back a transaction scanned every node and edge, so a one-node insert took 0.08 ms at 10,000 nodes and 36 ms at 1,000,000, holding the write lock. Each transaction now keeps a list of what it changed, and all three cost only as much as the change. -- **Concurrent writes could deadlock with the `tiered-storage` feature** (grafeo-core): creating a node or edge held the epoch arena's lock while waiting for the version index, the reverse of the order readers use, so a few threads writing and committing at once could stop for good. -- **Setting a property got slower the more properties a node had**: every set collected all of the node's properties and took the lock on all nodes to update a count that nothing read. It no longer does; `NodeRecord::props_count` (grafeo-core) is now always 0, like `EdgeRecord::props_count` always was. -- **Every hundredth commit paused longer the bigger the database got**: after every 100 commits, the cleanup of old versions went through every node, edge and property while holding the write lock. It now visits only the nodes, edges and properties that have older versions; without the `temporal` feature there are none. -- **Lookups by `id()` or by an indexed property scanned every node** ([#454](https://github.com/GrafeoDB/grafeo/issues/454), [#356](https://github.com/GrafeoDB/grafeo/issues/356)): `MATCH (n) WHERE id(n) = $id` scanned all nodes, and a lookup whose key comes from the row, such as `UNWIND $rows AS row MATCH (n {id: row.id})` or a variable bound by an earlier `MATCH`, scanned them once per row even with a property index on `id`, so a batch of updates by key took time proportional to rows times nodes. These now look each node up directly, also for `IN` lists and for both ends of an edge (`MATCH (s)-[r]->(d) WHERE id(s) = $s AND id(d) = $d`). `EXPLAIN` shows `[seek: id]` or `[index: ...]` only where such a lookup happens; it claimed an index lookup for the row-keyed shapes before. -- **A transaction could not delete what it had created**: `DELETE` and `DETACH DELETE` of a node or edge inserted earlier in the same transaction did nothing, so it survived the commit. -- **Rolled-back nodes still counted in the planner's statistics**: after a rollback, label counts included the nodes the transaction had created, which skewed query plans. `PreparedCommit::info()` also reported 0 nodes and edges written before the commit; it now counts the entities the transaction wrote. -- **`wal_checkpoint()` lost data in WAL-directory databases** ([#419](https://github.com/GrafeoDB/grafeo/issues/419)): on a path without the `.grafeo` extension, the WAL is the only copy of the data, but a checkpoint made the next open skip older WAL files and deleted them once the WAL had rotated (64 MiB). `wal_checkpoint()` now only syncs the WAL there, and opening such a database replays every WAL file, so databases checkpointed by older versions recover completely as long as their WAL files still exist. -- **Schema changes were lost when the WAL was replayed** ([#422](https://github.com/GrafeoDB/grafeo/issues/422)): WAL-directory databases on every reopen, and single-file databases after a crash before the next checkpoint, lost every `ALTER GRAPH TYPE` and every `CREATE OR REPLACE` of a node type, edge type, graph type or procedure. Replay now rebuilds the same schema the statements built. A WAL record with a kind this version does not know now fails the open with an error naming the record, instead of being skipped (or, for type constraints, turning into a UNIQUE constraint). -- **`DROP CONSTRAINT` did nothing** ([#420](https://github.com/GrafeoDB/grafeo/issues/420)): it reported success, also for names that did not exist, and the constraint kept rejecting writes. Constraints are now stored by name. `DROP CONSTRAINT` removes the named constraint and fails for an unknown name unless `IF EXISTS` is given; `CREATE CONSTRAINT` fails for a name in use unless `IF NOT EXISTS` is given; `SHOW CONSTRAINTS` lists them (it always returned nothing). A constraint created without a name is named after its label, properties and kind, such as `City_name_not_null`. Constraints that a `.grafeo` file got from 0.5.43 keep working but have no name, so they cannot be dropped by name. -- **`CREATE CONSTRAINT` was lost on reopen** ([#421](https://github.com/GrafeoDB/grafeo/issues/421)): in WAL-directory databases after every reopen, and in `.grafeo` databases after a crash before the next checkpoint. Replaying the WAL now restores constraints and their drops. -- **`SET` and `REMOVE` skipped the schema checks on nodes**: they could give a typed property a value of another type, remove a property a `NOT NULL` or `NODE KEY` constraint requires, or copy a value a `UNIQUE` constraint forbids. They are now checked like `INSERT`, against the node's labels. -- **`SET n:Label` skipped the schema checks**: adding a label did not check the node against that label's constraints and node type, so `MATCH (n:Guest) SET n:Person` could give a second `Person` the same unique email. It is now checked like `INSERT` with that label. -- **`SET` on an edge never checked the edge type**: `MATCH ()-[r:RATES]->() SET r.stars = 'five'` was accepted although `RATES` declares `stars INTEGER`. Edge property writes are now checked against the edge's type. -- **Removed properties stayed visible as null**: without the `temporal` feature (as in the Python and Node.js packages), `REMOVE n.p` and `SET n.p = NULL` left `p` in `keys(n)` and `properties(n)` with the value null, and a node created with a null property value kept it. A property set to null is now removed, as in GQL, and creating with a null value stores nothing. -- **MERGE skipped checks that INSERT and SET make**: `ON MATCH SET` checked only property types and not `UNIQUE` or `NODE KEY`, nodes created by `MERGE` lacked their node type's `DEFAULT` values, relationships created by `MERGE` were not checked against the edge type's endpoint labels, and concurrent `MERGE` updates of the same node were not detected as a write conflict. -- **Parameterized queries skipped the schema and constraint checks** ([#526](https://github.com/GrafeoDB/grafeo/issues/526)): a write through `execute_with_params`, Cypher with parameters or SQL/PGQ with parameters (in Python and Node.js, every `execute(query, params)`) was not checked against `UNIQUE`, `NOT NULL`, `NODE KEY`, property types, edge endpoint types or the property size limit, so `INSERT (:Person {email: $email})` accepted a duplicate email that the same statement with a literal rejected. In a named graph such writes were also recorded for conflict detection as writes to the default graph. They now run exactly like the same statement with literal values, and a repeated parameterized query reuses its parsed plan. -- **`PROFILE` was ignored when parameters were passed** ([#460](https://github.com/GrafeoDB/grafeo/issues/460)): it ran the query and returned its rows instead of the profile. -- **Transactions in different named graphs failed with write conflicts**: every graph numbers its nodes and edges from 0, and conflict detection compared only the numbers, so while one transaction had written node 0 of one graph, another transaction writing node 0 of another graph failed with `Write-write conflict`. Conflicts are now detected per graph. -- **`UNIQUE` did not hold within a transaction**: two nodes with the same value, created in one statement (`UNWIND ['a', 'a'] AS x INSERT (:Doc {id: x})`) or in one transaction, were both accepted. A value the transaction wrote earlier now counts as taken. -- **Writes to nodes created earlier in the same transaction skipped the checks**: inside a transaction, `SET`, `REMOVE` and `SET n:Label` on a node the transaction had created were not checked against its constraints. -- **`UNIQUE` and `NODE KEY` constraints on several properties were checked one property at a time**: `ON (n.name, n.city) UNIQUE` rejected a second `Alix` in another city. They now reject only a node with the same values for all the properties. -- **A failed `ALTER NODE TYPE` or `ALTER EDGE TYPE` kept part of its changes**: in `ALTER NODE TYPE Sensor ADD PROPERTY location STRING DROP PROPERTY missing`, the statement failed but `location` stayed added. A statement with several alterations now applies all of them or none. -- **`EXCEPT`, `INTERSECT` and `OTHERWISE` did not check their branches' columns** ([#481](https://github.com/GrafeoDB/grafeo/issues/481)): unlike `UNION`, they accepted branches with a different number of columns or differently named columns, so `MATCH (a:Person) RETURN a.name AS x EXCEPT MATCH (b:Person) RETURN b.age AS y` compared names with ages. SQL/PGQ checked none of its set operations, `UNION` included. They now fail with the same error as `UNION`, also in chains. -- **Cypher `=~` matched any part of the string**: openCypher matches the whole string, but `n.name =~ 'Alpha'` was true for `'AlphaService'` and `'.*Service'` matched `'AlphaServiceImpl'`. The pattern now has to match the whole string (inline flags such as `(?i)` still apply); Gremlin's `regex()` still finds its pattern anywhere, as TinkerPop defines it. -- **Regular expressions and `LIKE` patterns were compiled again for every row** ([#458](https://github.com/GrafeoDB/grafeo/issues/458)): a filter such as `n.fileName =~ '(?i).*(route|api).*'` spent most of its time compiling (235 ms over 1,500 nodes, against 2 ms for the same filter with `CONTAINS`). Each pattern is now compiled once and reused. -- **Cypher and SQL/PGQ checked only the first label of some node patterns** ([#513](https://github.com/GrafeoDB/grafeo/issues/513)): in Cypher, `MATCH ()-[r]->(n:A:B)` returned every `A` node the edge reached, with or without `B`, and the endpoints of `shortestPath` and `allShortestPaths` lost their extra labels the same way. In SQL/PGQ, `MATCH (n:A:B)` inside `GRAPH_TABLE` ignored `B` both at the start of a pattern and on an edge target. A node pattern now requires all of its labels wherever it appears. -- **Shortest-path searches returned pairs without a path and ignored hop bounds** ([#514](https://github.com/GrafeoDB/grafeo/issues/514)): GQL `ANY SHORTEST` / `ALL SHORTEST` and Cypher `shortestPath` / `allShortestPaths` returned a row with a null length for a pair of nodes without a path, paired a node with itself at length 0 even with `->+` or `[*]`, and ignored the edge's quantifier. Such a pair now has no row (`OPTIONAL MATCH` keeps it with a null path), a minimum of one hop finds the shortest cycle back to the node, and the path must fit the quantifier: `->{1,3}` or `[*..3]` allows at most three hops, and an edge without a quantifier is a single hop, as in the GQL standard (previously any length). A query with many pairs no longer loses rows after a batch of pairs without a path. -- **`EXISTS`, `NOT EXISTS` and `COUNT` subqueries in `WHERE` could run before their variables were bound**: when a pattern's `WHERE` was only such a subquery, as in `MATCH (a)-[:KNOWS]->(b) WHERE EXISTS { MATCH (b)-[:LIVES_IN]->(:City) }`, or when the subquery read a variable a `WITH` introduced, the planner moved it below the step that binds that variable, so the query returned no rows (with `COUNT`, rows without values). A subquery, and a `rand()` condition, now run where they are written. -- **The variable of a variable-length edge pattern held only the last edge** (GQL and Cypher): in `MATCH (a)-[r*1..3]->(b)` and `MATCH (a)-[r]->{1,3}(b)`, `r` was a single relationship, so `size(r)` was null and `all(e IN r WHERE ...)` matched nothing. `r` is now the list of the path's relationships, one per hop. -- **List comprehensions, list predicates and `reduce()` dropped items whose expression called a function** ([#538](https://github.com/GrafeoDB/grafeo/issues/538)): `[x IN ['a', 'b'] | toUpper(x)]` returned `[]`, `[e IN relationships(p) | type(e)]` too, and `all(x IN ['a'] WHERE toUpper(x) = 'A')` was false. Every expression that works in `RETURN` now works on the items, including index access, nested comprehensions (a reused name shadows the outer one) and the edges of `reverse(relationships(p))`, `tail(...)` and slices. -- **`RETURN relationships(p)` and `RETURN nodes(p)` returned internal ids**: they now return the relationships and nodes, as `RETURN r` does for a single relationship; so do `head`, `last`, `reverse`, `tail` and index access on such lists. -- **Property maps on variable-length edges were checked only against the last hop**: `-[*1..3 {w: 1}]->` in Cypher and `-[{w: 1}]->{1,3}` in GQL, named or anonymous, matched paths whose earlier edges did not have `w = 1`. The map now has to hold for every hop. -- **`all(e IN relationships(p) WHERE e.w = 1)` matched nothing**: `e.w` on the items of `relationships(p)` and `edges(p)` was NULL inside list predicates and comprehensions, so `[e IN relationships(p) | e.w]` returned `[]`. -- **List comprehensions and list predicates ignored the other variables of the row**: in `[v IN list WHERE v >= lim | v * lim]` or `any(v IN list WHERE v = n.x)`, every variable other than the loop variable evaluated to nothing, so every element was dropped and the predicate was false. They now read the row's variables. -- **GQL `INSERT (a)-[:T]->(b)` failed with `Undefined variable`** when an endpoint had no label, e.g. `INSERT ({id: 1})-[:T]->({id: 2})`. Unlabeled endpoints are now created. -- **GQL `INSERT` after `MATCH` built paths wrong**: `MATCH (a) INSERT (a)<-[:T]-(:B)` failed with `Undefined variable`, and in `MATCH (a) INSERT (a)-[:R]->(:C)-[:S]->(:D)` the `:S` edge started at `a` instead of the new `:C` node. `INSERT` can no longer give labels or properties to a variable that is already bound, such as `MATCH (a) INSERT (a:X)-[:T]->(b)`, which created a second node named `a`: it fails with an error that points to `SET`. -- **Times with UTC offsets compared inconsistently**: `time('14:00+01:00') = time('13:00Z')` was false while `<=` and `>=` were true. Offset times now compare by instant, consistently for `=`, ordering, `DISTINCT` and grouping. A time without an offset compares as if it were UTC (it used to compare by clock time against offset times), and never equals an offset time. -- **`toString()` returned an internal form for dates, times, durations, lists and maps**: `toString(datetime('2024-01-15T14:30:00'))` gave `Timestamp(Timestamp(1705329000000000μs))`, and `CAST(... AS STRING)` the same. Temporal values now become ISO 8601 text (`2024-01-15T14:30:00.000000Z`), lists and maps their literal form (`[1, 2]`). -- **Very long `^` chains or runs of `-`/`+` signs or `NOT`s could crash the process** with a stack overflow in the GQL, Cypher and SQL/PGQ parsers; they now fail with the usual nesting-depth error. -- **Docs**: the Discord invite on the docs site pointed to an expired link, and the constraint examples of the GQL schema guide used Cypher's `REQUIRE ... IS UNIQUE` form, which GQL rejects; they now use `ON (p.email) UNIQUE` ([#344](https://github.com/GrafeoDB/grafeo/issues/344)). +Every query result in the differential test corpus that differs from 0.5.43 is listed, with its reason, in `scripts/difftest/reviewed/0.5.44.txt`. + +- **Checkpoints after `compact()` left data out of the `.grafeo` file**: the schema, indexes and RDF data were dropped (also at `close()`), the periodic checkpoint wrote the pre-compaction store, and `async_write_snapshot()` wrote little or nothing. Every checkpoint now writes the whole database. +- **A failed checkpoint could leave a `.grafeo` file that no longer opened** ([#418](https://github.com/GrafeoDB/grafeo/issues/418)). Checkpoints now write to `.checkpoint` first, so the last good state stays readable; this needs free disk space for a second copy of the file. +- **Edges appeared twice after a checkpoint followed by a crash** ([#417](https://github.com/GrafeoDB/grafeo/issues/417)). Replay now skips entities the file already holds and repairs affected databases, and checkpoints trim the sidecar WAL. +- **Direct writes were not durable until `close()`** ([#395](https://github.com/GrafeoDB/grafeo/issues/395)). Each call is now durable when it returns, and batch calls are recovered completely or not at all. +- **Concurrent transactions could corrupt WAL recovery** ([#411](https://github.com/GrafeoDB/grafeo/issues/411)): one session's rollback could erase another's commit, and open or rolled-back writes came back on reopen. Each transaction is now logged as one group at commit. +- **Two processes could open the same WAL-directory database** ([#405](https://github.com/GrafeoDB/grafeo/issues/405)); it is now locked like a `.grafeo` file. +- **`wal_checkpoint()` lost data in WAL-directory databases** ([#419](https://github.com/GrafeoDB/grafeo/issues/419)). It now only syncs the WAL there, and open replays every WAL file. +- **`DurabilityMode::Adaptive` never synced the WAL**, so it behaved like `NoSync`, and backups missed writes made while they ran; those now land in this backup or the next. +- **Schema changes, constraints and `create_graph()` / `drop_graph()` were lost on WAL replay** ([#421](https://github.com/GrafeoDB/grafeo/issues/421), [#422](https://github.com/GrafeoDB/grafeo/issues/422)). An unknown WAL record now fails the open instead of being skipped. +- **`DROP CONSTRAINT` did nothing** ([#420](https://github.com/GrafeoDB/grafeo/issues/420)). Constraints are now stored by name: `CREATE` and `DROP CONSTRAINT` support `IF NOT EXISTS` / `IF EXISTS`, `SHOW CONSTRAINTS` lists them, and unnamed ones get a name such as `City_name_not_null`. Constraints from 0.5.43 files have no name and cannot be dropped by name. +- **Reopening a `.grafeo` database or `to_memory()` lost its indexes** (lookups on a reopened file scanned every node, [#459](https://github.com/GrafeoDB/grafeo/issues/459)), and `to_memory()` also dropped the schema, constraints and property history. WAL-directory databases still lose their indexes until 0.5.45 ([#401](https://github.com/GrafeoDB/grafeo/issues/401)). +- **Rust builds with only the `rdf` profile kept no data across a reopen** ([#544](https://github.com/GrafeoDB/grafeo/issues/544)); the facade's `rdf` profile now includes the LPG store. +- **Embeddings changed while their vector index was spilled reverted on reload** ([#522](https://github.com/GrafeoDB/grafeo/issues/522)). +- **Sessions and direct reads after `compact()` missed parts of the database**: SPARQL writes were lost, CDC missed queries, the selected graph was ignored, and `get_node`, `node_count`, `info()`, `schema()` and similar did not see pre-compaction data (or panicked on a `with_store` database). Queries after `compact()` are still not written to the WAL, so a crash loses them; direct calls are logged. +- **Grants and read-only modes did not stop every write** ([#413](https://github.com/GrafeoDB/grafeo/issues/413)): `use_graph()` skipped the grant check, read-only grants allowed writes, and read-only transactions, roles and databases let non-GQL languages and the direct API write. A session whose graph was dropped also fell back to the default graph; it now fails. +- **Transactions**: a failed statement inside a transaction kept its partial writes (it is now undone, and the transaction goes on); a commit that failed with a write-write conflict left the transaction active ([#409](https://github.com/GrafeoDB/grafeo/issues/409)); a transaction could not delete what it had created; transactions in different named graphs raised false write conflicts; and rolled-back nodes still counted in the planner's statistics. +- **Writes that should fail went through**: `SET`, `REMOVE` and label changes on a node or edge deleted earlier in the statement or transaction (now `... does not exist or has been deleted in this transaction`, as in openCypher), writes from a session reading at an earlier epoch (`set_viewing_epoch`, `execute_at_epoch`), and writes after `compact()` on a database opened read-only. All of these now fail. +- **Writes skipped schema checks**: `SET`, `REMOVE`, `SET n:Label` and `MERGE` did not check property types, `NOT NULL`, `UNIQUE`, `NODE KEY`, edge types, endpoint labels or `DEFAULT` values, `UNIQUE` did not hold within a transaction, and multi-property `UNIQUE` and `NODE KEY` were checked one property at a time. Every write is now checked like `INSERT`, and a failed `ALTER NODE TYPE` or `ALTER EDGE TYPE` applies all or none. +- **Parameterized queries skipped schema and constraint checks** ([#526](https://github.com/GrafeoDB/grafeo/issues/526)), so `execute(query, params)` in Python and Node.js could insert a duplicate `UNIQUE` value. They now run exactly like the same statement with literals. +- **Parameters**: a statement that used a parameter nobody supplied stored the text `$e` (`INSERT (:P {e: $e})`) or failed with an internal error, and now fails with `Missing parameter: $e` before it writes anything; unaliased columns were named after a parameter's value (`RETURN $x` gave a column `5`) and are now named after the query text; and `PROFILE` was ignored when parameters were passed ([#460](https://github.com/GrafeoDB/grafeo/issues/460)). +- **Removed properties stayed visible as null** without the `temporal` feature (as in the Python and Node.js packages); `REMOVE n.p` and `SET n.p = NULL` now remove the property. +- **Commit and rollback got slower as the database grew** ([#410](https://github.com/GrafeoDB/grafeo/issues/410)): a one-node insert took 36 ms at 1M nodes. They now cost only as much as the change. +- **Slow paths on large databases**: setting a property got slower with the node's property count (`NodeRecord::props_count` is now always 0), every hundredth commit scanned the whole store, and concurrent writes could deadlock with `tiered-storage`. +- **Lookups by `id()` or an indexed property scanned every node** ([#454](https://github.com/GrafeoDB/grafeo/issues/454), [#356](https://github.com/GrafeoDB/grafeo/issues/356)), once per row for keys like `UNWIND $rows AS row MATCH (n {id: row.id})`, and with a label (`MATCH (n:File {id: 'x'})`) every node of that label was collected. They now seek directly, also for `IN` lists and edge endpoints. +- **Regular expressions and `LIKE` patterns were compiled for every row** ([#458](https://github.com/GrafeoDB/grafeo/issues/458)); each is now compiled once. +- **Nodes and edges came back as IDs, as `0`, or read another entity's properties** ([#482](https://github.com/GrafeoDB/grafeo/issues/482)): after `ORDER BY`, `SKIP`, `LIMIT`, `DISTINCT`, a later `UNION` branch, a second `MATCH`, `OPTIONAL MATCH`, `EXISTS`, `CALL`, `UNWIND`, grouping or a write, a returned edge could be `0`, a node or edge a raw ID, and a property could come from the entity of the other kind with the same ID (`MATCH (a)-[r]->(b) RETURN r ORDER BY r.w`). `collect()` and group keys returned nodes and edges as IDs, `nodes(p)` and `relationships(p)` returned internal ids, `UNWIND` of them or of a function call after `MATCH` returned no rows, and `EXCEPT` and `INTERSECT` over nodes or edges returned the wrong rows. Nodes, edges and lists of them now keep their kind through every clause, also in Gremlin `union(outE(), out())`. +- **`keys(r)` of an edge returned null**; it now lists the edge's keys, and `keys()` of a node or edge returns them sorted, like `properties()`. +- **A variable used as both a node and an edge matched by a coincidence of IDs** (`MATCH (r) MATCH ()-[r]->()`); it now fails with an error that names the variable. +- **`ORDER BY` over values of different types failed or depended on the input order**: values of different types compared as equal, numbers among them and around NaN came out of order, lists and maps were not ordered, and such a sort (also in `percentileCont` and `percentileDisc`) could fail with `does not correctly implement a total order` (a `PanicException` in Python). `ORDER BY` now uses the openCypher total order: maps, lists, paths, temporal values, strings, booleans, numbers, then null; integers and floats compare exactly and NaN sorts after infinity. +- **Other `ORDER BY` fixes**: GQL `NULLS FIRST` and `NULLS LAST` were reversed with `DESC`, `RETURN * ORDER BY ...` failed with `Variable '*' not found in input`, and sort keys could show up as extra columns (`RETURN r AS e ORDER BY e.w` returned an `e_w` column). +- **Patterns through variables bound earlier matched too much**: a path back to an earlier node (`MATCH (a)-->(b)-->(a)`) returned every two-hop path instead of the cycles, and a pattern through a bound edge (`MATCH ()-[r]->() MATCH (x)-[r]->(y)`) matched every edge of its shape, also after `WITH`, in `CALL` subqueries, in stored procedures and after a shortest path. Both now use what the variable holds. +- **Variable-length edges**: the edge variable held only the last edge, a property map was checked only on the last hop, and a one-hop quantifier (`*1..1`, `*1`, GQL `{1,1}`) bound a single edge, so `size(rs)` returned null. The variable is now the list of the path's edges. +- **Shortest-path searches returned pairs without a path and ignored hop bounds** ([#514](https://github.com/GrafeoDB/grafeo/issues/514)). Unreachable pairs no longer get a row, `->+` finds the shortest cycle, and the path must fit the quantifier. +- **Clauses after a write or a `WITH` in one statement** ([#479](https://github.com/GrafeoDB/grafeo/issues/479), [#480](https://github.com/GrafeoDB/grafeo/issues/480)): a `MATCH` after a write missed the written data (and when it found nothing, the write did not run), and a GQL `MATCH` after `WITH` failed with `Variable '...' not found in input`. Clauses now read the rows the earlier ones pass on. +- **`EXISTS` and `COUNT` subqueries gave wrong answers** ([#543](https://github.com/GrafeoDB/grafeo/issues/543)): they checked only a path's first edge (missing multi-hop matches and minimums like `*2..`), ignored the rest of their pattern (`EXISTS { MATCH (a)-[:KNOWS]->(b), (c:Robot) }` held without any `Robot`), used only their start node from the row (`EXISTS { MATCH (b)-[:KNOWS]->(a) }` held whenever `b` knew anyone), ignored a tie to the row through a value (`{id: s.id}`, an inner `WHERE`), could run in `WHERE` before their variables were bound, and counted edges the query could not see. Each row now gets its own answer from the whole pattern, and a null node makes `EXISTS` false and `COUNT` 0. +- **GQL subqueries**: in `EXISTS`, `COUNT` and `VALUE`, each `MATCH` clause matched on its own; an `EXISTS` that starts with `OPTIONAL MATCH` filtered rows; `VALUE { ... RETURN count(x) }` ignored its argument and `DISTINCT`; and a `VALUE` subquery that is not a count gave every row the same answer, or failed in `WHERE` and `WITH`. They now behave as the same clauses do outside a subquery. +- **Cypher `OPTIONAL MATCH` at the start of a query and in `EXISTS` and `COUNT` failed**; it now runs, with one row of nulls when nothing matches. +- **`CALL` subqueries saw the wrong outer variables**: a GQL `CALL` did not see the outer row, so every row got every match; an importing `WITH` (in GQL, the first `WITH` of the body) ignored its `WHERE` and `WITH a AS b` imported the wrong value; and a Cypher `CALL` without an importing `WITH` read outer variables as null. A GQL `CALL` now sees every outer variable and its `WITH` is an ordinary one, a Cypher `CALL` sees those its importing `WITH` lists, and the scope clause `CALL (a, b) { ... }` (`CALL (*)`, `CALL ()`), which failed to parse, names them in both languages. +- **Nodes and edges a `CALL` subquery returned were copies of their properties**: a later `MATCH` from them failed or matched everything, and `id(b)` and `type(r)` were null. `RETURN *` in a `CALL` also returned nothing, and a two-edge chain in a `CALL` no rows. The subquery now passes on the nodes and edges themselves, and `RETURN *` returns the variables it binds. +- **Cypher `CALL` bodies rejected `ORDER BY`, `SKIP`, `LIMIT`, `UNION`, `DELETE`, `MERGE`, `REMOVE`, `FOREACH` and a nested `CALL`**; they now run, as in openCypher (an ordered `LIMIT` gives the top rows per outer row). A GQL `CALL` body takes `UNION`, `EXCEPT`, `INTERSECT` and `OTHERWISE`. +- **An `OPTIONAL MATCH` condition that reads a variable bound before it dropped rows**: in `MATCH (a)-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c) WHERE c.age > a.age`, a `b` whose matches all failed the condition lost its row instead of keeping it with nulls. The condition now decides which matches count, in GQL and Cypher. +- **`MERGE` bound only the first of several matches**: with two `(:Tag {k: 1})` nodes, `MERGE (t:Tag {k: 1}) SET t.seen = true` set one of them. As in openCypher, `MERGE` now binds every match, one row each, also for relationships and in `upsert_edges`. +- **GQL `NEXT` did not pass rows on**: the statement after `NEXT` matched from every node and returned the first statement's columns too. It now reads the rows the one before returns, and only the last `RETURN` is the result. +- **CDC history mixed graphs**: `history(id)` and `changes_between` returned the events of every graph's entity with that id. Each event now names its graph (`graph`, also in Python and Node.js), `history` reads the database's default graph or the session's current graph, and `changes_between` returns every graph's events. +- **A variable a `WITH` left out failed with an internal error** (`MATCH (a) WITH 1 AS x RETURN a.name`); it now fails with `Undefined variable 'a'`. +- **A Cypher pattern comprehension inside a function call, an aggregate or a map projection failed** (`RETURN size([(a)-->(b) | b])`); it now runs. +- **A transaction on a `.grafeo` file, a WAL directory or a compacted database did not see the type of an edge it had created** until the commit. +- **List comprehensions, list predicates and `reduce()` lost items** ([#538](https://github.com/GrafeoDB/grafeo/issues/538)) when the expression called a function, read a property of a `relationships(p)` item or used another row variable. +- **`EXCEPT`, `INTERSECT` and `OTHERWISE` did not check their branches' columns** ([#481](https://github.com/GrafeoDB/grafeo/issues/481)), and SQL/PGQ checked none of its set operations; they now fail like `UNION`. +- **Cypher `=~` matched any part of the string**; it now has to match the whole string, as in openCypher. +- **Cypher and SQL/PGQ checked only the first label of some node patterns** ([#513](https://github.com/GrafeoDB/grafeo/issues/513)), e.g. `MATCH ()-[r]->(n:A:B)`. +- **GQL `INSERT` with unlabeled or already bound endpoints** failed with `Undefined variable` or attached edges to the wrong node. +- **A `LOAD CSV` row returned whole was null** instead of the row's map. +- **Times with UTC offsets compared inconsistently**; they now compare by instant. +- **`toString()` returned an internal form for temporal values, lists and maps**; it now gives ISO 8601 text and literal form. +- **Very long `^` chains or runs of signs or `NOT`s could overflow the parser stack**. +- **Docs**: fixed the Discord invite and the GQL constraint examples ([#344](https://github.com/GrafeoDB/grafeo/issues/344)). ### Deriva Changes for [Deriva](https://github.com/StevenBtw/deriva), which generates ArchiMate models from software repositories and keeps its graph in an embedded Grafeo database through the Python binding. -- **Louvain gives the same communities on every run**: `db.algorithms.louvain()` and `CALL grafeo.louvain()` returned different partitions of the same graph, with community ids numbered at random. The result is now the same on every run, communities are numbered 0, 1, 2 and so on by their smallest node id, and `CALL grafeo.louvain()` returns its rows in node id order. -- **Python: `grafeo.features()` and `grafeo.build_info()`**: `features()` lists the query languages and capabilities compiled into the build (`['gql', 'cypher', ...]`), and `build_info()` adds the version, the git commit the module was built from, whether that tree had uncommitted changes, and the build profile. The installation guide shows how to build a wheel with the release's features: passing `--features` to maturin replaces that list. -- **Dotted access into map-valued properties**: `n.meta.route` reads key `route` of the map in `n.meta`, like `n.meta['route']`, in GQL and Cypher, also chained (`n.meta.a.b`). It failed with a syntax error in GQL and with `Nested property access not supported` in Cypher. A missing key or a property value that is not a map gives null, and a column without an alias is named as written (`n.meta.route`). On a node or edge that an expression returns, such as `startNode(r).name` or `head(collect(n)).name`, dotted access is an error that says to match the node with a variable instead. -- **The direct API follows the selected graph**: after `set_graph("model")` (Rust: `set_current_graph`), `execute()` used `model`, but `create_node`, `create_edge`, the property and label calls, the batch calls, `get_node`, `find_nodes_by_property` and the property index calls still used the default graph. They now use the selected graph. -- **Graph handles: `db.graph(name)`** (Rust: `GrafeoDB::graph`) returns a handle on a named graph with `execute()`, `execute_cypher()`, the direct API, `find_nodes_by_property()`, the property index calls and `begin_transaction()`, all working in that graph without switching the graph `set_graph()` selects. Handles on different graphs can be used side by side and from several threads, and a call through a handle whose graph no longer exists raises an error instead of writing to another graph. -- **Direct writes skipped the schema and constraints**: `create_node`, `create_edge`, `set_node_property`, `add_node_label` and the batch calls accepted a duplicate `UNIQUE` value, a missing `NOT NULL` property or a value of the wrong type, all of which `INSERT` and `SET` reject. They are now checked the same way and fail with the constraint's error, and a batch call creates all of its nodes or none. -- **Direct writes did not advance the epoch**: `current_epoch()` stayed the same across `create_node`, `create_edge` and the other direct writes, so `changes_between` could not separate the writes before a point from those after it, and `get_node_at_epoch` at a pinned `current_epoch()` saw nodes created later. Each direct write now commits at a new epoch. -- **CDC events did not say what changed**: in Python and Node.js, change events had no labels, edge type or endpoints at all. In Rust, a node created with `create_node` had no labels in its create event, an edge delete had no type or endpoints, a node deleted with `delete_node` had no labels, and a node created by `INSERT` got a create event without properties followed by an update event for them. Events now carry a node's `labels` on create and delete, an edge's `edge_type`, `src_id` and `dst_id` on create and delete, and the last properties in `before` on a delete. A node or edge created in a transaction has one create event with the labels and properties it has when the transaction commits. -- **Query results report what their writes changed**: `result.counters` (a dict in Python, an object in Node.js, a `WriteCounters` in Rust) counts the nodes and edges created and deleted, the properties set and the labels added and removed, like the summary counters of other graph databases. A `MERGE` counts only what it creates. -- **Upserts by key**: `upsert_nodes(labels, rows, key="id", replace=False)` and `upsert_edges(edge_type, rows, ...)` (Python, also on graph handles; `upsertNodes` and `upsertEdges` in Node.js; `GrafeoDB::upsert_nodes` and `upsert_edges` in Rust) create or update many nodes or edges by a key property in one statement, checked like a query, and report how many rows they created, updated and skipped, with the indices of the skipped rows. Nodes match on the key and all the labels; edges connect existing nodes found by a key property (a row with a missing endpoint is skipped, never created). By default a row's properties are merged in; `replace=True` makes the entity equal to the row. -- **Row order without `ORDER BY`, and a switch to test for it**: the docs now say that rows without `ORDER BY` come in no particular order, which can change between runs, builds and versions. The new test option `shuffle_unordered` (Python: `GrafeoDB(shuffle_unordered=True)`; Node.js: `GrafeoDB.create(path, { shuffleUnordered: true })`; Rust: `Config::with_shuffle_unordered(true)`) returns every result without `ORDER BY` in random order, in every query language, so tests find code that relies on the order anyway. It is off by default and costs nothing then. -- **Batch edges with properties, batch nodes with several labels** ([#462](https://github.com/GrafeoDB/grafeo/issues/462)): `batch_create_edges` (Python: `(src, dst, type[, properties])` tuples; Node.js: `batchCreateEdges` with `{src, dst, type, properties}`; Rust: `GrafeoDB::batch_create_edges` with `BatchEdge`) creates many edges, each with its own type and properties, in one transaction, and `batch_create_nodes_with_props` takes a list of labels (Node.js: `batchCreateNodesWithProps`; Rust: `batch_create_nodes_with_labels`). Graph handles have both. -- **Cypher pattern predicates in `WHERE`**: `MATCH (d:Dir) WHERE (d)-[:CONTAINS]->() RETURN d` and `WHERE NOT (d)-->()` were syntax errors. A pattern with a relationship in `WHERE` is now a predicate that holds when the pattern has a match, as in openCypher and like `EXISTS { MATCH ... }`. +- **Deterministic Louvain**: the same partition on every run, with communities numbered from 0 by their smallest node id. +- **Python: `grafeo.features()` and `grafeo.build_info()`**: the compiled-in languages and capabilities, plus version, git commit and build profile. +- **Dotted access into map properties**: `n.meta.route` (also chained) in GQL and Cypher; a missing key gives null. +- **The direct API follows the selected graph** set with `set_graph()`, like `execute()`. +- **Graph handles**: `db.graph(name)` scopes queries, the direct API and transactions to one graph, usable side by side and across threads. +- **Direct writes are checked against the schema and constraints** like `INSERT`, and advance the epoch, so `changes_between` and `get_node_at_epoch` see them in order. +- **CDC events say what changed**: labels, edge type and endpoints, the last properties on delete, and one create event per entity created in a transaction. +- **Write counters**: `result.counters` reports nodes, edges, properties and labels created, set or removed. +- **Upserts by key**: `upsert_nodes(labels, rows, key="id")` and `upsert_edges(...)` create or update many entities by a key property and report created, updated and skipped rows. +- **`shuffle_unordered` test option**: returns rows without `ORDER BY` in random order, to catch code that relies on an order that is not guaranteed. +- **Batch edges with properties, batch nodes with several labels** ([#462](https://github.com/GrafeoDB/grafeo/issues/462)). +- **Cypher pattern predicates in `WHERE`**: `WHERE (d)-[:CONTAINS]->()` and `WHERE NOT (d)-->()`. ### Internal -- **CI gate and policy checks** ([#511](https://github.com/GrafeoDB/grafeo/issues/511)): a `CI Gate` job fails when any required job fails, so one check can be required before merging, and a `Policy` job checks crate boundaries, workflow toolchain pins and public-text rules (`scripts/check_policy.py`). -- **CI toolchain pins restored** ([#509](https://github.com/GrafeoDB/grafeo/issues/509)): a Dependabot bump to a Rust version that does not exist stopped CI at its first job; Dependabot no longer bumps the toolchain. -- **Pull request eligibility check**: a `PR Policy` check asks contributor pull requests for a planned issue and the release branch as target, an ownership confirmation when AI tools helped, maintainer approval (by label) for new dependencies, CI changes, new crates and feature flags, and a test for every fix. The rules are listed in CONTRIBUTING.md. +- **CI gate and policy checks** ([#511](https://github.com/GrafeoDB/grafeo/issues/511)): a single `CI Gate` check to require before merging, and a `Policy` job for crate boundaries, toolchain pins and public-text rules. +- **CI toolchain pins restored** ([#509](https://github.com/GrafeoDB/grafeo/issues/509)); Dependabot no longer bumps the Rust toolchain. +- **Pull request eligibility check**: a `PR Policy` check applies the contribution rules in CONTRIBUTING.md. +- **Differential test**: `scripts/difftest` runs a GQL and Cypher query corpus on the previous release and on a release build of a commit, and fails on any changed result that was not reviewed for the release; it also compares GQL with Cypher within one run. +- **Browser WASM size limit**: the CI limit for the gzipped browser build is now 760 KB (warning at 740 KB); the build is about 730 KB after this release's fixes. ## [0.5.43] - 2026-09-27 @@ -111,37 +119,37 @@ Stabilization release. Fixes for silent wrong results (`ORDER BY` + `LIMIT`, `UN ### Changed - **Breaking (Rust API): `QueryResult::new`, `with_types` and `from_rows` return `Result`** and reject repeated column names ([#371](https://github.com/GrafeoDB/grafeo/issues/371)). Add `?` or `.unwrap()` at call sites. -- **Unaliased columns are named after their expression, as written**: `RETURN id(a), n.a + 1, count(b), 'x'` yields `id(a)`, `n.a + 1`, `count(b)` and `'x'` instead of `id(...)`, `expr`, `count(...)` and `String("x")`. `CASE`, `reduce`, list comprehensions, map projections and aggregate parameters (`percentile_cont(x, 0.9)`) are rendered too. Aliased columns are unchanged. ([#350](https://github.com/GrafeoDB/grafeo/pull/350), [#372](https://github.com/GrafeoDB/grafeo/pull/372)) +- **Unaliased columns are named after their expression, as written**: `RETURN id(a), n.a + 1, count(b)` yields `id(a)`, `n.a + 1` and `count(b)` instead of `id(...)`, `expr` and `count(...)`. Aliased columns are unchanged. ([#350](https://github.com/GrafeoDB/grafeo/pull/350), [#372](https://github.com/GrafeoDB/grafeo/pull/372)) - **`restore_to_epoch()` refuses to overwrite an existing database** or its `.wal` sidecar; restore to a fresh path instead. ([#363](https://github.com/GrafeoDB/grafeo/pull/363), [@teipsum](https://github.com/teipsum)) - **Node.js: transaction queries run on a worker thread**, like `Database.execute()`, so they no longer block the event loop. `commit()` and `rollback()` now throw while a query from the same transaction is still running. -- **Queries with trailing input are rejected** ([#380](https://github.com/GrafeoDB/grafeo/issues/380)): GQL, Cypher and Gremlin now fail with a syntax error when text follows the statement, when several statements are separated by `;` (a trailing `;` is fine), or, in GQL, when clauses come in an order the parser does not support yet (e.g. `SET ... DELETE`). These used to run only the first part. +- **Queries with trailing input are rejected** ([#380](https://github.com/GrafeoDB/grafeo/issues/380)): GQL, Cypher and Gremlin fail with a syntax error on text after the statement, including `;`-separated statements (a trailing `;` is fine) and GQL clause orders not supported yet (e.g. `SET ... DELETE`). These used to run only the first part. - **Breaking (Rust, `grafeo-core`): `ColumnCodec::write_to`, `write_to_v2`, `write_to_v3` and `CsrAdjacency::write_to` return `Result`**, failing instead of writing a truncated size ([#392](https://github.com/GrafeoDB/grafeo/issues/392)). - **Rust toolchain pinned to 1.98.1** for local builds and CI. The MSRV stays 1.91.1, and the `grafeo` crate now declares it ([#390](https://github.com/GrafeoDB/grafeo/issues/390)). ### Fixed -- **GQL ran only the first statement and ignored the rest** ([#380](https://github.com/GrafeoDB/grafeo/issues/380)): `INSERT ... INSERT ...` (as in the quickstart) created only the first node, `INSERT ... RETURN` ignored its `RETURN`, and any other text after a statement was silently dropped. Consecutive `INSERT` clauses and `INSERT ... RETURN` now work. Gremlin (text after an unknown character) and Cypher (text after schema statements) had the same gap. +- **GQL ran only the first statement and ignored the rest** ([#380](https://github.com/GrafeoDB/grafeo/issues/380)): `INSERT ... INSERT ...` (as in the quickstart) created only the first node and `INSERT ... RETURN` ignored its `RETURN`; both now work. Gremlin and Cypher had similar gaps. - **GQL `^` (power) was not parsed**: `RETURN 2 ^ 10` returned `2`. It now computes the power, binding tighter than `*` and right associative. - **`ALTER NODE TYPE / ALTER EDGE TYPE ... ADD PROPERTY name TYPE` added a property called `PROPERTY`** of type `name` and dropped the real type; `DROP PROPERTY name` dropped the wrong property. `PROPERTY` is now an optional keyword. -- **Databases over the storage format's limits were written corrupt** ([#392](https://github.com/GrafeoDB/grafeo/issues/392)): an LPG section over 4 GiB, more than 65,535 blocks, or more than 65,535 labels on one node or versions of one property wrapped a size field, and the file failed to open with `block N CRC mismatch`. Such a checkpoint now fails with an error naming the limit, the sidecar WAL is kept and nothing is lost. RDF and compact-store sections get the same checks; the compact store used to panic on names over 64 KiB. Support for larger sections is planned for 0.5.45. -- **Rolling back a transaction on a persistent database did not undo `SET`, `REMOVE` or label changes**: they stayed applied after `rollback()`, and single-file databases wrote them to disk on `close()`. In-memory databases were not affected. +- **Databases over the storage format's limits were written corrupt** ([#392](https://github.com/GrafeoDB/grafeo/issues/392)): an LPG section over 4 GiB or 65,535 blocks, or over 65,535 labels on one node or versions of one property, wrapped a size field, and the file failed to open with `CRC mismatch`. Such a checkpoint now fails with an error naming the limit and keeps the sidecar WAL, so nothing is lost. Larger sections are planned for 0.5.45. +- **Rolling back a transaction on a persistent database did not undo `SET`, `REMOVE` or label changes**, and single-file databases wrote them to disk on `close()`. - **`ORDER BY ... LIMIT` over a whole-node `RETURN` returned raw NodeIds** instead of node maps, both for property sort keys ([#335](https://github.com/GrafeoDB/grafeo/issues/335)) and for expression keys such as `text_score(...)`, `CASE` or arithmetic ([#347](https://github.com/GrafeoDB/grafeo/issues/347)). ([#337](https://github.com/GrafeoDB/grafeo/pull/337), [#349](https://github.com/GrafeoDB/grafeo/pull/349), [@temporaryfix](https://github.com/temporaryfix)) -- **Duplicate column names silently lost data** ([#371](https://github.com/GrafeoDB/grafeo/issues/371)), e.g. `RETURN id(s), id(t)` or `RETURN a.name, a.name`. Distinct expressions now get distinct names, and a result that would still repeat a name is an error asking for an alias. In SPARQL, an `AS ?x` that repeats another projected name or a variable bound in `WHERE` is a parse error. ([#372](https://github.com/GrafeoDB/grafeo/pull/372), [@teipsum](https://github.com/teipsum); [#350](https://github.com/GrafeoDB/grafeo/pull/350), [@temporaryfix](https://github.com/temporaryfix)) -- **`UNION` with differing branches** ([#365](https://github.com/GrafeoDB/grafeo/issues/365)): SPARQL now returns every variable from every branch, unbound where a branch lacks it. GQL and Cypher reject branches with a different number of columns or differently named columns instead of padding or truncating them, also in chains (`a UNION b UNION c`) and for branches that return only aggregates. ([#366](https://github.com/GrafeoDB/grafeo/pull/366), [@teipsum](https://github.com/teipsum)) -- **Aggregates next to aliased or later items** (GQL and Cypher): `RETURN n.city AS city, count(n)` and `RETURN count(n), sum(n.v) + 1` failed with `Undefined variable '_agg_0'`, `RETURN count(n), n.city` returned its columns in the wrong order, a `GROUP BY` key that was not returned appeared as an extra column, and `CALL ... RETURN count(x) GROUP BY y` ignored its `GROUP BY`. -- **SPARQL updates ignored open transactions**: `INSERT ... WHERE`, `DELETE WHERE` and `DELETE/INSERT ... WHERE` applied immediately, so `rollback()` did not undo them, and `INSERT DATA` / `DELETE DATA` on a named graph inside a transaction was dropped on commit. All updates now apply on commit and are discarded on rollback, and queries inside a transaction see its own earlier writes. +- **Duplicate column names silently lost data** ([#371](https://github.com/GrafeoDB/grafeo/issues/371)), e.g. `RETURN id(s), id(t)`. Distinct expressions now get distinct names, and a result that would still repeat a name is an error asking for an alias; in SPARQL, an `AS ?x` that repeats a projected or bound name is a parse error. ([#372](https://github.com/GrafeoDB/grafeo/pull/372), [@teipsum](https://github.com/teipsum); [#350](https://github.com/GrafeoDB/grafeo/pull/350), [@temporaryfix](https://github.com/temporaryfix)) +- **`UNION` with differing branches** ([#365](https://github.com/GrafeoDB/grafeo/issues/365)): SPARQL now returns every variable from every branch, and GQL and Cypher reject branches with different columns instead of padding or truncating them. ([#366](https://github.com/GrafeoDB/grafeo/pull/366), [@teipsum](https://github.com/teipsum)) +- **Aggregates next to aliased or later items** (GQL and Cypher): `RETURN n.city AS city, count(n)` failed with `Undefined variable '_agg_0'`, columns came back in the wrong order, and unreturned `GROUP BY` keys appeared as extra columns. +- **SPARQL updates ignored open transactions**: `INSERT/DELETE ... WHERE` applied immediately and was not undone by `rollback()`, and `INSERT DATA` / `DELETE DATA` on a named graph was dropped on commit. Updates now apply on commit, and queries in a transaction see its earlier writes. - **SPARQL `FROM` and `WITH ` did not apply inside subqueries**: a nested `SELECT` read the default graph instead of the query's dataset. - **SPARQL updates on named graphs went to the wrong graph or were lost** ([#367](https://github.com/GrafeoDB/grafeo/issues/367)): `INSERT`/`DELETE ... WHERE` with `GRAPH ` or `WITH ` now targets the named graph and survives a restart. `USING` / `USING NAMED` is rejected until it is supported. ([#368](https://github.com/GrafeoDB/grafeo/pull/368), [@teipsum](https://github.com/teipsum)) - **SPARQL `path+` returned only direct neighbours** ([#369](https://github.com/GrafeoDB/grafeo/issues/369)); it now returns the full transitive closure. ([#370](https://github.com/GrafeoDB/grafeo/pull/370), [@teipsum](https://github.com/teipsum)) - **Edges disappeared after `compact()`** once an endpoint was modified ([#345](https://github.com/GrafeoDB/grafeo/issues/345)). Deleted edges also no longer reappear or show up as neighbours. ([#346](https://github.com/GrafeoDB/grafeo/pull/346), [@temporaryfix](https://github.com/temporaryfix)) - **Updating an indexed vector could make other vectors unfindable** ([#374](https://github.com/GrafeoDB/grafeo/issues/374)). Removing or replacing the HNSW entry point also no longer leaves the upper index layers unreachable. ([#375](https://github.com/GrafeoDB/grafeo/pull/375), [@jarmen423](https://github.com/jarmen423)) -- **Filters on edge properties or map values returned no rows** when nodes had a property with the same name: `MATCH ()-[r]->() WHERE r.id = 'e1'`, `-[r {id: 'e1'}]->`, `UNWIND [{id: 'a'}] AS m WHERE m.id = 'a'` and similar filters were answered from node statistics. Filters on values written earlier in the same query (`SET ... WITH ... WHERE`) had the same problem. +- **Filters on edge properties or map values returned no rows** when nodes had a property with the same name, e.g. `MATCH ()-[r]->() WHERE r.id = 'e1'` or `UNWIND [{id: 'a'}] AS m WHERE m.id = 'a'`, because they were answered from node statistics. - **Property maps on anonymous edges were ignored**: `()-[:T {since: 2020}]->()` in Cypher and GQL matched every `T` edge. -- **Deleted nodes stayed in property and vector indexes**: `find_nodes_by_property` kept returning a deleted node, so a re-created node with the same key was found twice, and a vector search after `DELETE` could still return it. The lookup API also returned nodes from transactions that had not committed yet. Rolling back a delete or a `SET` now restores the index entries. -- **Computed property values in `CREATE`, `INSERT` and `MERGE`**: `CREATE (:X {id: toString(i)})`, `i * 2` or `'v' + toString(i)` failed with an internal error, and `MERGE (:X {id: toString(i)})` silently used `null`, so every row matched or created the same node. They are now evaluated per row. +- **Deleted nodes stayed in property and vector indexes**, so `find_nodes_by_property` and vector search could still return them, and the lookup API returned uncommitted nodes. Rolling back a delete or `SET` now restores the index entries. +- **Computed property values in `CREATE`, `INSERT` and `MERGE`**: `CREATE (:X {id: toString(i)})` failed with an internal error, and `MERGE` silently used `null`, so every row matched or created the same node. They are now evaluated per row. - **`MERGE` created duplicate relationships** when several rows of one statement merged the same edge (`UNWIND [1, 2] AS i MATCH (a), (b) MERGE (a)-[:T]->(b)` created two). A `null` in a relationship's `MERGE` pattern now also matches an absent property, as it already did for nodes. - **`=` and `<>` on dates, times, datetimes, durations and vectors were always false / true** in expressions, e.g. `RETURN date('2024-01-01') = date('2024-01-01')`, `x IN [date(...)]` or a comparison after `WITH`. Filters answered directly from a node property were not affected. -- **Python: naive `datetime` values were read as local time** but returned as UTC, so they shifted by the machine's UTC offset on a round trip; they are now UTC both ways. Microseconds are no longer rounded, dates before 1970 work on Windows, and the Python 3.12 `utcfromtimestamp` deprecation warning is gone. +- **Python: naive `datetime` values were read as local time** but returned as UTC; they are now UTC both ways. Microseconds are no longer rounded and dates before 1970 work on Windows. - **Storage docs**: the `.grafeo` sidecar WAL is removed on a clean `close()`, not after every checkpoint. ([#364](https://github.com/GrafeoDB/grafeo/pull/364), [@teipsum](https://github.com/teipsum)) ### Security diff --git a/Cargo.lock b/Cargo.lock index 11793fee3..087913a17 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -1508,7 +1508,7 @@ dependencies = [ [[package]] name = "grafeo" -version = "0.5.43" +version = "0.5.44" dependencies = [ "grafeo-adapters", "grafeo-common", @@ -1520,7 +1520,7 @@ dependencies = [ [[package]] name = "grafeo-adapters" -version = "0.5.43" +version = "0.5.44" dependencies = [ "bincode", "grafeo-common", @@ -1538,7 +1538,7 @@ dependencies = [ [[package]] name = "grafeo-bindings-common" -version = "0.5.43" +version = "0.5.44" dependencies = [ "grafeo-common", "grafeo-engine", @@ -1547,7 +1547,7 @@ dependencies = [ [[package]] name = "grafeo-c" -version = "0.5.43" +version = "0.5.44" dependencies = [ "grafeo-bindings-common", "grafeo-common", @@ -1559,7 +1559,7 @@ dependencies = [ [[package]] name = "grafeo-cli" -version = "0.5.43" +version = "0.5.44" dependencies = [ "anyhow", "clap", @@ -1580,7 +1580,7 @@ dependencies = [ [[package]] name = "grafeo-common" -version = "0.5.43" +version = "0.5.44" dependencies = [ "aes-gcm", "arcstr", @@ -1607,7 +1607,7 @@ dependencies = [ [[package]] name = "grafeo-core" -version = "0.5.43" +version = "0.5.44" dependencies = [ "arc-swap", "arcstr", @@ -1639,7 +1639,7 @@ dependencies = [ [[package]] name = "grafeo-engine" -version = "0.5.43" +version = "0.5.44" dependencies = [ "arcstr", "arrow-array", @@ -1679,14 +1679,14 @@ dependencies = [ [[package]] name = "grafeo-examples" -version = "0.5.43" +version = "0.5.44" dependencies = [ "grafeo", ] [[package]] name = "grafeo-node" -version = "0.5.43" +version = "0.5.44" dependencies = [ "grafeo-adapters", "grafeo-bindings-common", @@ -1704,7 +1704,7 @@ dependencies = [ [[package]] name = "grafeo-python" -version = "0.5.43" +version = "0.5.44" dependencies = [ "grafeo-adapters", "grafeo-bindings-common", @@ -1721,7 +1721,7 @@ dependencies = [ [[package]] name = "grafeo-spec-tests" -version = "0.5.43" +version = "0.5.44" dependencies = [ "grafeo-common", "grafeo-engine", @@ -1731,7 +1731,7 @@ dependencies = [ [[package]] name = "grafeo-storage" -version = "0.5.43" +version = "0.5.44" dependencies = [ "bincode", "byteorder", @@ -1753,7 +1753,7 @@ dependencies = [ [[package]] name = "grafeo-wasm" -version = "0.5.43" +version = "0.5.44" dependencies = [ "console_error_panic_hook", "getrandom 0.4.3", diff --git a/Cargo.toml b/Cargo.toml index 96ff9bd56..72ff4a0b8 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -18,7 +18,7 @@ members = [ ] [workspace.package] -version = "0.5.43" +version = "0.5.44" edition = "2024" rust-version = "1.91.1" license = "Apache-2.0" @@ -28,13 +28,13 @@ description = "A high-performance, embeddable graph database with support for LP [workspace.dependencies] # Internal crates (default-features = false to allow per-crate control) -grafeo-common = { path = "crates/grafeo-common", version = "0.5.43" } -grafeo-core = { path = "crates/grafeo-core", version = "0.5.43", default-features = false } -grafeo-adapters = { path = "crates/grafeo-adapters", version = "0.5.43", default-features = false } -grafeo-storage = { path = "crates/grafeo-storage", version = "0.5.43", default-features = false } -grafeo-engine = { path = "crates/grafeo-engine", version = "0.5.43", default-features = false } -grafeo = { path = "crates/grafeo", version = "0.5.43" } -grafeo-bindings-common = { path = "crates/bindings/common", version = "0.5.43", default-features = false } +grafeo-common = { path = "crates/grafeo-common", version = "0.5.44" } +grafeo-core = { path = "crates/grafeo-core", version = "0.5.44", default-features = false } +grafeo-adapters = { path = "crates/grafeo-adapters", version = "0.5.44", default-features = false } +grafeo-storage = { path = "crates/grafeo-storage", version = "0.5.44", default-features = false } +grafeo-engine = { path = "crates/grafeo-engine", version = "0.5.44", default-features = false } +grafeo = { path = "crates/grafeo", version = "0.5.44" } +grafeo-bindings-common = { path = "crates/bindings/common", version = "0.5.44", default-features = false } # Error handling thiserror = "2.0" diff --git a/_typos.toml b/_typos.toml index b6f18b8c4..bd3695386 100644 --- a/_typos.toml +++ b/_typos.toml @@ -10,8 +10,15 @@ extend-exclude = [ ".mypy_cache/", ".pytest_cache/", ".ruff_cache/", + # LDBC SNB test data, copied verbatim: generated text with truncated words, DBpedia names + "tests/spec/datasets/ldbc_snb_mini.setup", + "tests/spec/lpg/*/ldbc_snb_interactive.gtest", ] +[default] +# Result fingerprints in the differential test's reviewed lists are hex digests, not words +extend-ignore-re = ['(?m)^[A-Z]+[0-9]+\|[a-z]+ [0-9a-f]{10} '] + [default.extend-words] # "fof" = friend-of-friend (standard graph query pattern) fof = "fof" diff --git a/crates/bindings/csharp/src/Grafeo/Grafeo.csproj b/crates/bindings/csharp/src/Grafeo/Grafeo.csproj index 8d7dc1213..5cb88e480 100644 --- a/crates/bindings/csharp/src/Grafeo/Grafeo.csproj +++ b/crates/bindings/csharp/src/Grafeo/Grafeo.csproj @@ -6,7 +6,7 @@ Grafeo - 0.5.43 + 0.5.44 C# bindings for the Grafeo graph database. Supports GQL queries, ACID transactions, and vector search. GrafeoDB Apache-2.0 diff --git a/crates/bindings/dart/CHANGELOG.md b/crates/bindings/dart/CHANGELOG.md index 64e71bcba..abb594c74 100644 --- a/crates/bindings/dart/CHANGELOG.md +++ b/crates/bindings/dart/CHANGELOG.md @@ -1,5 +1,9 @@ # Changelog +## 0.5.44 + +- Version bump to match workspace release + ## 0.5.43 - Version bump to match workspace release diff --git a/crates/bindings/dart/README.md b/crates/bindings/dart/README.md index 4ef70647e..a150dbc88 100644 --- a/crates/bindings/dart/README.md +++ b/crates/bindings/dart/README.md @@ -6,7 +6,7 @@ Dart FFI bindings for the [Grafeo](https://grafeo.dev) graph database. Wraps the ```yaml dependencies: - grafeo: ^0.5.43 + grafeo: ^0.5.44 ``` You also need the `grafeo-c` native library for your platform. See [Building from Source](#building-from-source) below. diff --git a/crates/bindings/dart/pubspec.yaml b/crates/bindings/dart/pubspec.yaml index a8334e1fb..aba719023 100644 --- a/crates/bindings/dart/pubspec.yaml +++ b/crates/bindings/dart/pubspec.yaml @@ -1,5 +1,5 @@ name: grafeo -version: 0.5.43 +version: 0.5.44 description: >- High-performance embeddable graph database with GQL support. Dart bindings for grafeo-c via dart:ffi. diff --git a/crates/bindings/node/__test__/index.spec.mjs b/crates/bindings/node/__test__/index.spec.mjs index 58b564c92..c568a98ff 100644 --- a/crates/bindings/node/__test__/index.spec.mjs +++ b/crates/bindings/node/__test__/index.spec.mjs @@ -740,31 +740,44 @@ describe('transaction edge cases', () => { const r = await db.execute('MATCH (p:Person) RETURN p.name') expect(r.length).toBe(1) } else { - // Commit won the race: the queued query must fail, not run auto-committed. - await expect(pending).rejects.toThrow(/no longer active/) + // Commit went first. A query that had not started yet must fail, not + // run auto-committed; one that had already finished ran inside the + // transaction and was committed with it. (The rollback test below shows + // that a query never runs after its transaction ended.) + const ran = await pending.then( + () => true, + (e) => { + expect(e.message).toMatch(/no longer active/) + return false + }, + ) const r = await db.execute('MATCH (p:Person) RETURN p.name') - expect(r.length).toBe(0) + expect(r.length).toBe(ran ? 1 : 0) } db.close() } }) it('should refuse rollback while a query is running, then allow it', async () => { - const db = GrafeoDB.create() - const tx = db.beginTransaction() - const pending = tx.execute("INSERT (:Person {name: 'Jules'})") - try { - tx.rollback() - } catch (e) { - expect(e.message).toMatch(/still running/) - await pending - tx.rollback() + for (let i = 0; i < 20; i++) { + const db = GrafeoDB.create() + const tx = db.beginTransaction() + const pending = tx.execute("INSERT (:Person {name: 'Jules'})") + try { + tx.rollback() + } catch (e) { + expect(e.message).toMatch(/still running/) + await pending + tx.rollback() + } + // Whether the query ran before the rollback, was still running or had not + // started, nothing it wrote survives. + await pending.catch(() => {}) + expect(tx.isActive).toBe(false) + const r = await db.execute('MATCH (p:Person) RETURN p.name') + expect(r.length).toBe(0) + db.close() } - await pending.catch(() => {}) - expect(tx.isActive).toBe(false) - const r = await db.execute('MATCH (p:Person) RETURN p.name') - expect(r.length).toBe(0) - db.close() }) }) @@ -1572,6 +1585,17 @@ describe('ID validation', () => { it('should reject negative edge ID', () => { expect(() => db.getEdge(-1)).toThrow(/Invalid edge ID/) }) + + it('should reject a fractional ID instead of truncating it', async () => { + expect(() => db.getNode(1.5)).toThrow(/Invalid node ID/) + expect(() => db.getEdge(0.5)).toThrow(/Invalid edge ID/) + const alix = db.createNode(['Person']).id + const gus = db.createNode(['Person']).id + await expect( + db.batchCreateEdges([{ src: alix + 0.9, dst: gus, type: 'KNOWS' }]) + ).rejects.toThrow(/Invalid node ID/) + expect(db.edgeCount()).toBe(0) + }) }) // ── Concurrent database instances ─────────────────────────────────── @@ -1599,8 +1623,17 @@ describe('concurrent instances', () => { // -- Upserts ----------------------------------------------------------- describe('upserts', () => { + let db + + beforeEach(() => { + db = GrafeoDB.create() + }) + + afterEach(() => { + db.close() + }) + it('should create then update nodes and edges by key', async () => { - const db = GrafeoDB.create() const nodes = await db.upsertNodes( ['Graph', 'File'], [{ id: 'f1', size: 3 }, { size: 4 }, { id: 'f2' }, { id: 'f1', lang: 'rs' }] @@ -1619,11 +1652,9 @@ describe('upserts', () => { await db.upsertNodes(['Graph', 'File'], [{ id: 'f1', size: 9 }], { replace: true }) const file = (await db.execute("MATCH (n:File {id: 'f1'}) RETURN n.size, n.lang")).toArray() expect(file).toEqual([{ 'n.size': 9, 'n.lang': null }]) - db.close() }) it('should take edge options', async () => { - const db = GrafeoDB.create() await db.upsertNodes(['File'], [{ id: 'f1' }, { id: 'f2' }]) const result = await db.upsertEdges('CALLS', [{ from: 'f1', to: 'f2', rid: 'c1' }], { key: 'rid', @@ -1632,15 +1663,23 @@ describe('upserts', () => { dstField: 'to', }) expect(result.created).toBe(1) - db.close() }) }) // -- Batch writes -------------------------------------------------------- describe('batch writes', () => { + let db + + beforeEach(() => { + db = GrafeoDB.create() + }) + + afterEach(() => { + db.close() + }) + it('should create nodes with several labels and edges with their own types', async () => { - const db = GrafeoDB.create() const [alix, gus, vincent] = await db.batchCreateNodesWithProps( ['Graph', 'Person'], [{ name: 'Alix' }, { name: 'Gus' }, { name: 'Vincent' }] @@ -1659,11 +1698,9 @@ describe('batch writes', () => { { 'a.name': 'Alix', 'type(r)': 'KNOWS', 'r.since': 2020 }, { 'a.name': 'Gus', 'type(r)': 'LIKES', 'r.since': null }, ]) - db.close() }) it('should create no edge of a failing batch', async () => { - const db = GrafeoDB.create() const [alix, gus] = await db.batchCreateNodesWithProps('Person', [{}, {}]) await expect( db.batchCreateEdges([ @@ -1672,7 +1709,6 @@ describe('batch writes', () => { ]) ).rejects.toThrow(/does not exist/) expect(db.edgeCount()).toBe(0) - db.close() }) }) diff --git a/crates/bindings/node/index.d.ts b/crates/bindings/node/index.d.ts index 88d6b609e..0c75ce9a8 100644 --- a/crates/bindings/node/index.d.ts +++ b/crates/bindings/node/index.d.ts @@ -287,8 +287,10 @@ export declare class GrafeoDB { * * Each row names its endpoints in the source and target fields (`src` * and `dst` by default) and holds the edge key; every other field is an - * edge property. A row whose endpoint does not exist, or without the - * key, is skipped, never created. + * edge property. A row is skipped when it lacks the edge key, the + * source field or the target field, or when no node or more than one + * node has its endpoint key; endpoints are never created. The key and + * the two endpoint fields must be different fields. */ upsertEdges(edgeType: string, rows: Array, options?: UpsertEdgesOptions | undefined | null): Promise } @@ -355,6 +357,8 @@ export declare class QueryResult { edges(): Array /** Returns the result formatted as a Unicode table. */ toString(): string + /** Get all rows as an array of arrays (no column names). */ + rows(): object /** * Returns the result as Arrow IPC stream bytes (Buffer). * @@ -365,8 +369,6 @@ export declare class QueryResult { * ``` */ toArrowIPC(): Buffer - /** Get all rows as an array of arrays (no column names). */ - rows(): object } /** @@ -525,8 +527,9 @@ export interface UpsertSummary { /** Rows that updated an existing node or edge. */ updated: number /** - * Rows that were not written: without their key, or an edge row whose - * endpoint does not exist. + * Rows that were not written: without their key, edge rows without a + * source or target field, and edge rows whose endpoint key matches no + * node or more than one node. */ skipped: number /** The indices of the skipped rows, in order (at most 1,000). */ diff --git a/crates/bindings/node/index.js b/crates/bindings/node/index.js index f90de1b7a..1ccdc4302 100644 --- a/crates/bindings/node/index.js +++ b/crates/bindings/node/index.js @@ -77,8 +77,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-android-arm64') const bindingPackageVersion = require('@grafeo-db/js-android-arm64/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -93,8 +93,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-android-arm-eabi') const bindingPackageVersion = require('@grafeo-db/js-android-arm-eabi/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -114,8 +114,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-win32-x64-gnu') const bindingPackageVersion = require('@grafeo-db/js-win32-x64-gnu/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -130,8 +130,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-win32-x64-msvc') const bindingPackageVersion = require('@grafeo-db/js-win32-x64-msvc/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -147,8 +147,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-win32-ia32-msvc') const bindingPackageVersion = require('@grafeo-db/js-win32-ia32-msvc/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -163,8 +163,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-win32-arm64-msvc') const bindingPackageVersion = require('@grafeo-db/js-win32-arm64-msvc/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -182,8 +182,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-darwin-universal') const bindingPackageVersion = require('@grafeo-db/js-darwin-universal/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -198,8 +198,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-darwin-x64') const bindingPackageVersion = require('@grafeo-db/js-darwin-x64/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -214,8 +214,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-darwin-arm64') const bindingPackageVersion = require('@grafeo-db/js-darwin-arm64/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -234,8 +234,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-freebsd-x64') const bindingPackageVersion = require('@grafeo-db/js-freebsd-x64/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -250,8 +250,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-freebsd-arm64') const bindingPackageVersion = require('@grafeo-db/js-freebsd-arm64/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -271,8 +271,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-linux-x64-musl') const bindingPackageVersion = require('@grafeo-db/js-linux-x64-musl/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -287,8 +287,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-linux-x64-gnu') const bindingPackageVersion = require('@grafeo-db/js-linux-x64-gnu/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -305,8 +305,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-linux-arm64-musl') const bindingPackageVersion = require('@grafeo-db/js-linux-arm64-musl/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -321,8 +321,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-linux-arm64-gnu') const bindingPackageVersion = require('@grafeo-db/js-linux-arm64-gnu/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -339,8 +339,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-linux-arm-musleabihf') const bindingPackageVersion = require('@grafeo-db/js-linux-arm-musleabihf/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -355,8 +355,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-linux-arm-gnueabihf') const bindingPackageVersion = require('@grafeo-db/js-linux-arm-gnueabihf/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -373,8 +373,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-linux-loong64-musl') const bindingPackageVersion = require('@grafeo-db/js-linux-loong64-musl/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -389,8 +389,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-linux-loong64-gnu') const bindingPackageVersion = require('@grafeo-db/js-linux-loong64-gnu/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -407,8 +407,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-linux-riscv64-musl') const bindingPackageVersion = require('@grafeo-db/js-linux-riscv64-musl/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -423,8 +423,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-linux-riscv64-gnu') const bindingPackageVersion = require('@grafeo-db/js-linux-riscv64-gnu/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -440,8 +440,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-linux-ppc64-gnu') const bindingPackageVersion = require('@grafeo-db/js-linux-ppc64-gnu/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -456,8 +456,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-linux-s390x-gnu') const bindingPackageVersion = require('@grafeo-db/js-linux-s390x-gnu/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -476,8 +476,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-openharmony-arm64') const bindingPackageVersion = require('@grafeo-db/js-openharmony-arm64/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -492,8 +492,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-openharmony-x64') const bindingPackageVersion = require('@grafeo-db/js-openharmony-x64/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { @@ -508,8 +508,8 @@ function requireNative() { try { const binding = require('@grafeo-db/js-openharmony-arm') const bindingPackageVersion = require('@grafeo-db/js-openharmony-arm/package.json').version - if (bindingPackageVersion !== '0.5.43' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { - throw new Error(`Native binding package version mismatch, expected 0.5.43 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) + if (bindingPackageVersion !== '0.5.44' && process.env.NAPI_RS_ENFORCE_VERSION_CHECK && process.env.NAPI_RS_ENFORCE_VERSION_CHECK !== '0') { + throw new Error(`Native binding package version mismatch, expected 0.5.44 but got ${bindingPackageVersion}. You can reinstall dependencies to fix this issue.`) } return binding } catch (e) { diff --git a/crates/bindings/node/npm/darwin-arm64/package.json b/crates/bindings/node/npm/darwin-arm64/package.json index 89977b000..c43bed2a3 100644 --- a/crates/bindings/node/npm/darwin-arm64/package.json +++ b/crates/bindings/node/npm/darwin-arm64/package.json @@ -1,6 +1,6 @@ { "name": "@grafeo-db/js-darwin-arm64", - "version": "0.5.43", + "version": "0.5.44", "os": [ "darwin" ], diff --git a/crates/bindings/node/npm/darwin-x64/package.json b/crates/bindings/node/npm/darwin-x64/package.json index 25dfe5b43..2a3f17999 100644 --- a/crates/bindings/node/npm/darwin-x64/package.json +++ b/crates/bindings/node/npm/darwin-x64/package.json @@ -1,6 +1,6 @@ { "name": "@grafeo-db/js-darwin-x64", - "version": "0.5.43", + "version": "0.5.44", "os": [ "darwin" ], diff --git a/crates/bindings/node/npm/linux-arm64-gnu/package.json b/crates/bindings/node/npm/linux-arm64-gnu/package.json index d156fbe71..ffa9e5ad0 100644 --- a/crates/bindings/node/npm/linux-arm64-gnu/package.json +++ b/crates/bindings/node/npm/linux-arm64-gnu/package.json @@ -1,6 +1,6 @@ { "name": "@grafeo-db/js-linux-arm64-gnu", - "version": "0.5.43", + "version": "0.5.44", "os": [ "linux" ], diff --git a/crates/bindings/node/npm/linux-arm64-musl/package.json b/crates/bindings/node/npm/linux-arm64-musl/package.json index 165167bb2..1f7d8d71e 100644 --- a/crates/bindings/node/npm/linux-arm64-musl/package.json +++ b/crates/bindings/node/npm/linux-arm64-musl/package.json @@ -1,6 +1,6 @@ { "name": "@grafeo-db/js-linux-arm64-musl", - "version": "0.5.43", + "version": "0.5.44", "os": [ "linux" ], diff --git a/crates/bindings/node/npm/linux-x64-gnu/package.json b/crates/bindings/node/npm/linux-x64-gnu/package.json index 1b0548a85..6fb7b32ae 100644 --- a/crates/bindings/node/npm/linux-x64-gnu/package.json +++ b/crates/bindings/node/npm/linux-x64-gnu/package.json @@ -1,6 +1,6 @@ { "name": "@grafeo-db/js-linux-x64-gnu", - "version": "0.5.43", + "version": "0.5.44", "os": [ "linux" ], diff --git a/crates/bindings/node/npm/win32-x64-msvc/package.json b/crates/bindings/node/npm/win32-x64-msvc/package.json index e83f1f6c5..10a64bc61 100644 --- a/crates/bindings/node/npm/win32-x64-msvc/package.json +++ b/crates/bindings/node/npm/win32-x64-msvc/package.json @@ -1,6 +1,6 @@ { "name": "@grafeo-db/js-win32-x64-msvc", - "version": "0.5.43", + "version": "0.5.44", "os": [ "win32" ], diff --git a/crates/bindings/node/package.json b/crates/bindings/node/package.json index ef16399d2..e858d4074 100644 --- a/crates/bindings/node/package.json +++ b/crates/bindings/node/package.json @@ -1,6 +1,6 @@ { "name": "@grafeo-db/js", - "version": "0.5.43", + "version": "0.5.44", "description": "Node.js/TypeScript bindings for Grafeo - a high-performance embeddable graph database", "main": "index.js", "types": "index.d.ts", @@ -20,12 +20,12 @@ "index.d.ts" ], "optionalDependencies": { - "@grafeo-db/js-darwin-x64": "0.5.43", - "@grafeo-db/js-win32-x64-msvc": "0.5.43", - "@grafeo-db/js-linux-x64-gnu": "0.5.43", - "@grafeo-db/js-darwin-arm64": "0.5.43", - "@grafeo-db/js-linux-arm64-gnu": "0.5.43", - "@grafeo-db/js-linux-arm64-musl": "0.5.43" + "@grafeo-db/js-darwin-x64": "0.5.44", + "@grafeo-db/js-win32-x64-msvc": "0.5.44", + "@grafeo-db/js-linux-x64-gnu": "0.5.44", + "@grafeo-db/js-darwin-arm64": "0.5.44", + "@grafeo-db/js-linux-arm64-gnu": "0.5.44", + "@grafeo-db/js-linux-arm64-musl": "0.5.44" }, "keywords": [ "graph", diff --git a/crates/bindings/node/src/database.rs b/crates/bindings/node/src/database.rs index c25b3831e..75a3c4939 100644 --- a/crates/bindings/node/src/database.rs +++ b/crates/bindings/node/src/database.rs @@ -35,40 +35,35 @@ fn convert_json_filters( Ok(Some(result)) } -/// Validate a JavaScript number as a safe node ID. -/// -/// JavaScript numbers are f64, but entity IDs are u64. This rejects -/// negative values, NaN, Infinity, and values beyond `Number.MAX_SAFE_INTEGER`. -fn validate_node_id(id: f64) -> Result { - if !(0.0..=9_007_199_254_740_991.0).contains(&id) { - return Err(NodeGrafeoError::InvalidArgument(format!("Invalid node ID: {id}")).into()); - } - // reason: Range check above guarantees the value is in [0, 2^53-1], safe for u64 +/// A JavaScript number as an unsigned 64-bit id: a whole number from 0 to +/// `Number.MAX_SAFE_INTEGER`. Negative values, fractions (`1.9` is not node +/// 1), NaN and Infinity give `None`. +fn whole_id(value: f64) -> Option { + let whole = (0.0..=9_007_199_254_740_991.0).contains(&value) && value.fract() == 0.0; + // reason: a whole number in [0, 2^53-1] converts to u64 exactly #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)] - Ok(NodeId(id as u64)) + whole.then_some(value as u64) +} + +/// Validate a JavaScript number as a node ID (see [`whole_id`]). +fn validate_node_id(id: f64) -> Result { + whole_id(id) + .map(NodeId) + .ok_or_else(|| NodeGrafeoError::InvalidArgument(format!("Invalid node ID: {id}")).into()) } -/// Validate a JavaScript number as a safe edge ID. +/// Validate a JavaScript number as an edge ID (see [`whole_id`]). fn validate_edge_id(id: f64) -> Result { - if !(0.0..=9_007_199_254_740_991.0).contains(&id) { - return Err(NodeGrafeoError::InvalidArgument(format!("Invalid edge ID: {id}")).into()); - } - // reason: Range check above guarantees the value is in [0, 2^53-1], safe for u64 - #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)] - Ok(EdgeId(id as u64)) + whole_id(id) + .map(EdgeId) + .ok_or_else(|| NodeGrafeoError::InvalidArgument(format!("Invalid edge ID: {id}")).into()) } -/// Validate a JavaScript number as a non-negative epoch ID. -/// -/// Rejects negative values, NaN, Infinity, and values beyond -/// `Number.MAX_SAFE_INTEGER`. Epochs are unsigned 64-bit integers internally. +/// Validate a JavaScript number as an epoch (see [`whole_id`]). fn validate_epoch(epoch: f64) -> Result { - if !(0.0..=9_007_199_254_740_991.0).contains(&epoch) { - return Err(NodeGrafeoError::InvalidArgument(format!("Invalid epoch: {epoch}")).into()); - } - // reason: Range check above guarantees the value is in [0, 2^53-1], safe for u64 - #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)] - Ok(grafeo_common::types::EpochId::new(epoch as u64)) + whole_id(epoch) + .map(grafeo_common::types::EpochId::new) + .ok_or_else(|| NodeGrafeoError::InvalidArgument(format!("Invalid epoch: {epoch}")).into()) } /// Your connection to a Grafeo database. @@ -1624,6 +1619,7 @@ fn change_event_to_json(event: &grafeo_engine::cdc::ChangeEvent) -> serde_json:: serde_json::json!({ "entity_id": event.entity_id.as_u64(), "entity_type": entity_type, + "graph": event.graph, "kind": kind, "epoch": event.epoch.0, "timestamp": event.timestamp, @@ -1637,7 +1633,9 @@ fn change_event_to_json(event: &grafeo_engine::cdc::ChangeEvent) -> serde_json:: }) } -// After `JsGrafeoDB`: napi takes the class's JS name from the struct, so the -// `impl` blocks of these modules must come after it. +// After `JsGrafeoDB`: napi-derive records a struct's `js_name` when it expands +// the struct and looks it up when it expands an `impl` block, falling back to +// the Rust name when the struct has not expanded yet. Declared earlier, these +// modules' methods would land on a class named after the Rust struct. mod batch; mod upsert; diff --git a/crates/bindings/node/src/database/upsert.rs b/crates/bindings/node/src/database/upsert.rs index 74f00b51c..4d589ccf6 100644 --- a/crates/bindings/node/src/database/upsert.rs +++ b/crates/bindings/node/src/database/upsert.rs @@ -47,8 +47,9 @@ pub struct UpsertSummary { pub created: u32, /// Rows that updated an existing node or edge. pub updated: u32, - /// Rows that were not written: without their key, or an edge row whose - /// endpoint does not exist. + /// Rows that were not written: without their key, edge rows without a + /// source or target field, and edge rows whose endpoint key matches no + /// node or more than one node. pub skipped: u32, /// The indices of the skipped rows, in order (at most 1,000). pub skipped_rows: Vec, @@ -122,8 +123,10 @@ impl JsGrafeoDB { /// /// Each row names its endpoints in the source and target fields (`src` /// and `dst` by default) and holds the edge key; every other field is an - /// edge property. A row whose endpoint does not exist, or without the - /// key, is skipped, never created. + /// edge property. A row is skipped when it lacks the edge key, the + /// source field or the target field, or when no node or more than one + /// node has its endpoint key; endpoints are never created. The key and + /// the two endpoint fields must be different fields. #[napi(js_name = "upsertEdges")] pub async fn upsert_edges( &self, diff --git a/crates/bindings/node/src/query.rs b/crates/bindings/node/src/query.rs index 8c6216219..f72c1e9a2 100644 --- a/crates/bindings/node/src/query.rs +++ b/crates/bindings/node/src/query.rs @@ -150,38 +150,6 @@ impl QueryResult { ) } - /// Returns the result as Arrow IPC stream bytes (Buffer). - /// - /// Use with the `apache-arrow` npm package: - /// ```js - /// import { tableFromIPC } from 'apache-arrow'; - /// const table = tableFromIPC(result.toArrowIPC()); - /// ``` - #[cfg(feature = "arrow-export")] - #[napi(js_name = "toArrowIPC")] - pub fn to_arrow_ipc(&self) -> Result { - let col_types = vec![grafeo_common::LogicalType::Any; self.columns.len()]; - let batch = grafeo_engine::database::arrow::query_result_to_record_batch( - &self.columns, - &col_types, - &self.rows, - ) - .map_err(|e| { - napi::Error::new( - napi::Status::GenericFailure, - format!("Arrow export failed: {e}"), - ) - })?; - let ipc_bytes = grafeo_engine::database::arrow::record_batch_to_ipc_stream(&batch) - .map_err(|e| { - napi::Error::new( - napi::Status::GenericFailure, - format!("Arrow IPC failed: {e}"), - ) - })?; - Ok(ipc_bytes.into()) - } - /// Get all rows as an array of arrays (no column names). #[napi] pub fn rows(&self, env: Env) -> Result> { @@ -215,6 +183,43 @@ impl QueryResult { } } +// A separate block: napi-derive registers every method of a `#[napi]` +// impl, so a method behind a cfg needs a block behind that cfg. +#[cfg(feature = "arrow-export")] +#[napi] +impl QueryResult { + /// Returns the result as Arrow IPC stream bytes (Buffer). + /// + /// Use with the `apache-arrow` npm package: + /// ```js + /// import { tableFromIPC } from 'apache-arrow'; + /// const table = tableFromIPC(result.toArrowIPC()); + /// ``` + #[napi(js_name = "toArrowIPC")] + pub fn to_arrow_ipc(&self) -> Result { + let col_types = vec![grafeo_common::LogicalType::Any; self.columns.len()]; + let batch = grafeo_engine::database::arrow::query_result_to_record_batch( + &self.columns, + &col_types, + &self.rows, + ) + .map_err(|e| { + napi::Error::new( + napi::Status::GenericFailure, + format!("Arrow export failed: {e}"), + ) + })?; + let ipc_bytes = grafeo_engine::database::arrow::record_batch_to_ipc_stream(&batch) + .map_err(|e| { + napi::Error::new( + napi::Status::GenericFailure, + format!("Arrow IPC failed: {e}"), + ) + })?; + Ok(ipc_bytes.into()) + } +} + impl QueryResult { /// Convert a row to a JS object with column names as keys. fn row_to_object(&self, env: sys::napi_env, idx: usize) -> Result> { diff --git a/crates/bindings/python/pyproject.toml b/crates/bindings/python/pyproject.toml index dc9a03161..317122608 100644 --- a/crates/bindings/python/pyproject.toml +++ b/crates/bindings/python/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "maturin" [project] name = "grafeo" -version = "0.5.43" +version = "0.5.44" description = "A high-performance, embeddable graph database with Python bindings" readme = "README.md" license = { text = "Apache-2.0" } diff --git a/crates/bindings/python/src/database.rs b/crates/bindings/python/src/database.rs index d90fa08da..a340a354f 100644 --- a/crates/bindings/python/src/database.rs +++ b/crates/bindings/python/src/database.rs @@ -1464,9 +1464,11 @@ impl PyGrafeoDB { /// Each row names its endpoints in `src_field` and `dst_field` (the /// nodes whose `endpoint_key` has that value, with all of /// `endpoint_labels` when given) and holds the edge's `key`; every other - /// field is an edge property. A row whose endpoint does not exist, or - /// without the key, is skipped, never created. An edge is identified by - /// its endpoints, type and key. By default a row's properties are merged + /// field is an edge property. A row is skipped when it lacks the key, + /// `src_field` or `dst_field`, or when no node or more than one node has + /// its endpoint key; endpoints are never created. `key`, `src_field` and + /// `dst_field` must be different fields. An edge is identified by its + /// endpoints, type and key. By default a row's properties are merged /// into the edge's; with `replace=True` they become exactly the row's. /// A property index on `endpoint_key` makes the endpoint lookups fast. /// @@ -4033,6 +4035,16 @@ fn change_event_to_dict( }; map.insert("labels".to_string(), labels_py(&event.labels)); map.insert("before_labels".to_string(), labels_py(&event.before_labels)); + // The graph the entity is in (None for the default graph): entity ids + // repeat across graphs. + map.insert( + "graph".to_string(), + event + .graph + .clone() + .into_py_any(py) + .expect("str to Python conversion"), + ); map.insert( "edge_type".to_string(), event diff --git a/crates/bindings/python/src/direct.rs b/crates/bindings/python/src/direct.rs index 30187aad3..ef3dcd3f7 100644 --- a/crates/bindings/python/src/direct.rs +++ b/crates/bindings/python/src/direct.rs @@ -129,9 +129,12 @@ pub(crate) trait DirectTarget { properties: Vec<(PropertyKey, Value)>, ) -> grafeo_common::utils::error::Result; - fn node(&self, id: NodeId) -> Option; + /// The node as the target now sees it; an error when the target itself + /// is gone (a dropped graph). + fn node(&self, id: NodeId) -> grafeo_common::utils::error::Result>; - fn edge(&self, id: EdgeId) -> Option; + /// The edge as the target now sees it (see [`node`](Self::node)). + fn edge(&self, id: EdgeId) -> grafeo_common::utils::error::Result>; fn upsert_nodes( &self, @@ -168,12 +171,12 @@ impl DirectTarget for Session { Session::create_edge_with_props(self, source, target, edge_type, properties) } - fn node(&self, id: NodeId) -> Option { - self.get_node(id) + fn node(&self, id: NodeId) -> grafeo_common::utils::error::Result> { + Ok(self.get_node(id)) } - fn edge(&self, id: EdgeId) -> Option { - self.get_edge(id) + fn edge(&self, id: EdgeId) -> grafeo_common::utils::error::Result> { + Ok(self.get_edge(id)) } fn upsert_nodes( @@ -215,12 +218,12 @@ impl DirectTarget for GrafeoDB { GrafeoDB::create_edge_with_props(self, source, target, edge_type, properties) } - fn node(&self, id: NodeId) -> Option { - self.get_node(id) + fn node(&self, id: NodeId) -> grafeo_common::utils::error::Result> { + Ok(self.get_node(id)) } - fn edge(&self, id: EdgeId) -> Option { - self.get_edge(id) + fn edge(&self, id: EdgeId) -> grafeo_common::utils::error::Result> { + Ok(self.get_edge(id)) } fn upsert_nodes( @@ -262,12 +265,12 @@ impl DirectTarget for GraphHandle<'_> { GraphHandle::create_edge_with_props(self, source, target, edge_type, properties) } - fn node(&self, id: NodeId) -> Option { - self.get_node(id).ok().flatten() + fn node(&self, id: NodeId) -> grafeo_common::utils::error::Result> { + self.get_node(id) } - fn edge(&self, id: EdgeId) -> Option { - self.get_edge(id).ok().flatten() + fn edge(&self, id: EdgeId) -> grafeo_common::utils::error::Result> { + self.get_edge(id) } fn upsert_nodes( @@ -378,6 +381,7 @@ pub(crate) fn create_node( .map_err(PyGrafeoError::from)?; target .node(id) + .map_err(PyGrafeoError::from)? .map(node) .ok_or_else(|| PyGrafeoError::database("Failed to create node").into()) } @@ -395,6 +399,7 @@ pub(crate) fn create_edge( .map_err(PyGrafeoError::from)?; target .edge(id) + .map_err(PyGrafeoError::from)? .map(edge) .ok_or_else(|| PyGrafeoError::database("Failed to create edge").into()) } diff --git a/crates/bindings/python/src/graph_handle.rs b/crates/bindings/python/src/graph_handle.rs index b92b10853..af46d9fdd 100644 --- a/crates/bindings/python/src/graph_handle.rs +++ b/crates/bindings/python/src/graph_handle.rs @@ -220,9 +220,11 @@ impl PyGraphHandle { /// Each row names its endpoints in `src_field` and `dst_field` (the /// nodes whose `endpoint_key` has that value, with all of /// `endpoint_labels` when given) and holds the edge's `key`; every other - /// field is an edge property. A row whose endpoint does not exist, or - /// without the key, is skipped, never created. An edge is identified by - /// its endpoints, type and key. By default a row's properties are merged + /// field is an edge property. A row is skipped when it lacks the key, + /// `src_field` or `dst_field`, or when no node or more than one node has + /// its endpoint key; endpoints are never created. `key`, `src_field` and + /// `dst_field` must be different fields. An edge is identified by its + /// endpoints, type and key. By default a row's properties are merged /// into the edge's; with `replace=True` they become exactly the row's. /// A property index on `endpoint_key` makes the endpoint lookups fast. /// diff --git a/crates/bindings/python/src/query.rs b/crates/bindings/python/src/query.rs index 2c3c1037a..b0c1921bd 100644 --- a/crates/bindings/python/src/query.rs +++ b/crates/bindings/python/src/query.rs @@ -163,15 +163,20 @@ impl PyQueryResult { /// if result.rows_scanned: /// print(f"Scanned {result.rows_scanned} rows") /// ``` + #[getter] + fn rows_scanned(&self) -> Option { + self.rows_scanned + } + /// What the query's writes changed, as a dict: `nodes_created`, /// `nodes_deleted`, `edges_created`, `edges_deleted`, `properties_set`, /// `labels_added` and `labels_removed`. /// /// Example: - /// ```python - /// result = db.execute("INSERT (:Person {name: 'Alix'})") - /// result.counters["nodes_created"] # 1 - /// ``` + /// ```python + /// result = db.execute("INSERT (:Person {name: 'Alix'})") + /// result.counters["nodes_created"] # 1 + /// ``` #[getter] fn counters<'py>(&self, py: Python<'py>) -> PyResult> { let c = &self.counters; @@ -190,11 +195,6 @@ impl PyQueryResult { Ok(dict) } - #[getter] - fn rows_scanned(&self) -> Option { - self.rows_scanned - } - /// Convert to a pandas DataFrame. /// /// Requires pandas to be installed (`uv add pandas`). Each column in the diff --git a/crates/bindings/python/tests/lpg/test_upserts.py b/crates/bindings/python/tests/lpg/test_upserts.py index 3033857e1..ec6b3d23d 100644 --- a/crates/bindings/python/tests/lpg/test_upserts.py +++ b/crates/bindings/python/tests/lpg/test_upserts.py @@ -41,6 +41,23 @@ def test_edges_connect_existing_nodes_only(db): assert values(db, "MATCH ()-[r]->() RETURN type(r), r.id, r.w") == [["Graph:USES", "u1", 5]] +def test_an_ambiguous_endpoint_skips_the_row(db): + db.upsert_nodes(["File"], [{"id": "f1"}, {"id": "f2"}, {"id": "f3"}]) + db.execute("INSERT (:Other {id: 'f2'})") + result = db.upsert_edges( + "USES", + [ + {"src": "f1", "dst": "f2", "id": "u1"}, + {"src": "f3", "dst": "f1", "id": "u3"}, + {"src": "f2", "dst": "f1", "id": "u2"}, + ], + ) + assert result == {"created": 1, "updated": 0, "skipped": 2, "skipped_rows": [0, 2]} + assert values(db, "MATCH (s)-[r]->(d) RETURN s.id, d.id, r.id") == [["f3", "f1", "u3"]] + with pytest.raises(Exception, match="different fields"): + db.upsert_edges("USES", [{"src": "f1", "dst": "f2"}], key="src") + + def test_edge_options(db): db.upsert_nodes(["File"], [{"id": "f1"}, {"id": "f2"}]) db.execute("INSERT (:Other {id: 'f2'})") diff --git a/crates/bindings/wasm/package-lite.json b/crates/bindings/wasm/package-lite.json index c045e8e42..f30ff9a6a 100644 --- a/crates/bindings/wasm/package-lite.json +++ b/crates/bindings/wasm/package-lite.json @@ -1,7 +1,7 @@ { "name": "@grafeo-db/wasm-lite", "type": "module", - "version": "0.5.43", + "version": "0.5.44", "description": "WebAssembly bindings for Grafeo - GQL-only lightweight variant", "keywords": [ "graph", diff --git a/crates/bindings/wasm/package.json b/crates/bindings/wasm/package.json index 0afa13e66..20705df76 100644 --- a/crates/bindings/wasm/package.json +++ b/crates/bindings/wasm/package.json @@ -1,7 +1,7 @@ { "name": "@grafeo-db/wasm", "type": "module", - "version": "0.5.43", + "version": "0.5.44", "description": "WebAssembly bindings for Grafeo - a high-performance embeddable graph database", "keywords": [ "graph", diff --git a/crates/grafeo-adapters/src/query/cypher/ast.rs b/crates/grafeo-adapters/src/query/cypher/ast.rs index 084a7c999..fadf3c418 100644 --- a/crates/grafeo-adapters/src/query/cypher/ast.rs +++ b/crates/grafeo-adapters/src/query/cypher/ast.rs @@ -83,8 +83,20 @@ pub enum Clause { Remove(RemoveClause), /// CALL procedure clause. Call(CallClause), - /// CALL { subquery } (inline subquery). - CallSubquery(Query), + /// CALL [(scope)] { subquery } (inline subquery). + CallSubquery { + /// The subquery. + query: Query, + /// The variable scope clause, `CALL (a, b) { ... }`: the outer + /// variables the subquery sees (`*` for all of them, none for `()`). + /// `None` without one, when its importing `WITH` names them. + scope: Option>, + /// Further subqueries joined to `query` by `UNION` (`UNION ALL` when + /// `union_all`), each with an importing `WITH` of its own. + unions: Vec, + /// Whether the `unions` are joined by `UNION ALL` (duplicates kept). + union_all: bool, + }, /// FOREACH (variable IN list | update_clauses). ForEach(ForEachClause), /// LOAD CSV clause. diff --git a/crates/grafeo-adapters/src/query/cypher/parser.rs b/crates/grafeo-adapters/src/query/cypher/parser.rs index cc598aec3..7996a240f 100644 --- a/crates/grafeo-adapters/src/query/cypher/parser.rs +++ b/crates/grafeo-adapters/src/query/cypher/parser.rs @@ -148,14 +148,10 @@ impl<'a> Parser<'a> { return Err(self.error("UNION requires query statements")); }; let mut queries = vec![first_query]; - let mut is_all = false; + let mut union_all = None; while self.current.kind == TokenKind::Union { - self.advance(); // consume UNION - is_all = self.current.kind == TokenKind::All; - if is_all { - self.advance(); // consume ALL - } + self.parse_union_keyword(&mut union_all)?; let next_stmt = self.parse_statement()?; match next_stmt { Statement::Query(q) => queries.push(q), @@ -167,10 +163,26 @@ impl<'a> Parser<'a> { Ok(Statement::Union { queries, - all: is_all, + all: union_all.unwrap_or(false), }) } + /// Consumes `UNION` or `UNION ALL`. As in openCypher, one chain of + /// queries uses one of them: `union_all` holds the kind seen so far, and + /// a different one is an error. + fn parse_union_keyword(&mut self, union_all: &mut Option) -> Result<()> { + self.expect(TokenKind::Union)?; + let all = self.current.kind == TokenKind::All; + if all { + self.advance(); // consume ALL + } + if union_all.is_some_and(|seen| seen != all) { + return Err(self.error("Invalid combination of UNION and UNION ALL")); + } + *union_all = Some(all); + Ok(()) + } + fn parse_statement(&mut self) -> Result { // Parse reading/writing clauses into a query let mut clauses = Vec::new(); @@ -224,18 +236,7 @@ impl<'a> Parser<'a> { self.advance(); clauses.push(Clause::Limit(self.parse_expression()?)); } - TokenKind::Call => { - // CALL { subquery } vs CALL procedure(...) - if self.peek_kind() == TokenKind::LBrace { - self.advance(); // consume CALL - self.advance(); // consume { - let inner = self.parse_subquery_body()?; - self.expect(TokenKind::RBrace)?; - clauses.push(Clause::CallSubquery(inner)); - } else { - clauses.push(Clause::Call(self.parse_call_clause()?)); - } - } + TokenKind::Call => clauses.push(self.parse_call()?), _ => { // FOREACH and LOAD are contextual keywords (not reserved) if self.can_be_identifier() @@ -261,6 +262,37 @@ impl<'a> Parser<'a> { })) } + /// Parses `CALL [(scope)] { subquery [UNION [ALL] subquery]* }` or a + /// procedure call: a procedure name comes before any `(`. + fn parse_call(&mut self) -> Result { + if !matches!(self.peek_kind(), TokenKind::LBrace | TokenKind::LParen) { + return Ok(Clause::Call(self.parse_call_clause()?)); + } + self.advance(); // consume CALL + let scope = if self.current.kind == TokenKind::LParen { + Some(self.parse_call_scope()?) + } else { + None + }; + self.expect(TokenKind::LBrace)?; + self.enter_nesting()?; + let query = self.parse_subquery_body()?; + let mut unions = Vec::new(); + let mut union_all = None; + while self.current.kind == TokenKind::Union { + self.parse_union_keyword(&mut union_all)?; + unions.push(self.parse_subquery_body()?); + } + self.exit_nesting(); + self.expect(TokenKind::RBrace)?; + Ok(Clause::CallSubquery { + query, + scope, + unions, + union_all: union_all.unwrap_or(false), + }) + } + /// Parses a CALL clause: `CALL name.space(args) [YIELD field [AS alias], ...]`. fn parse_call_clause(&mut self) -> Result { let span_start = self.current.span.start; @@ -442,6 +474,33 @@ impl<'a> Parser<'a> { TokenKind::Set => { clauses.push(Clause::Set(self.parse_set_clause()?)); } + TokenKind::Merge => { + clauses.push(Clause::Merge(self.parse_merge_clause()?)); + } + TokenKind::Delete | TokenKind::Detach => { + clauses.push(Clause::Delete(self.parse_delete_clause()?)); + } + TokenKind::Remove => { + clauses.push(Clause::Remove(self.parse_remove_clause()?)); + } + TokenKind::Order => { + clauses.push(Clause::OrderBy(self.parse_order_by_clause()?)); + } + TokenKind::Skip => { + self.advance(); + clauses.push(Clause::Skip(self.parse_expression()?)); + } + TokenKind::Limit => { + self.advance(); + clauses.push(Clause::Limit(self.parse_expression()?)); + } + TokenKind::Call => clauses.push(self.parse_call()?), + // FOREACH is a contextual keyword (not reserved) + _ if self.can_be_identifier() + && self.get_identifier_text().eq_ignore_ascii_case("FOREACH") => + { + clauses.push(Clause::ForEach(self.parse_foreach_clause()?)); + } _ => break, } } @@ -479,13 +538,21 @@ impl<'a> Parser<'a> { }) } - /// Parses the inner query of an EXISTS subquery. - /// Accepts one or more MATCH clauses and an optional WHERE clause. + /// Parses the inner query of an EXISTS or COUNT subquery: MATCH and + /// OPTIONAL MATCH clauses and an optional WHERE. fn parse_exists_inner_query(&mut self) -> Result { let mut clauses = Vec::new(); - while self.current.kind == TokenKind::Match || self.current.kind == TokenKind::Optional { - clauses.push(Clause::Match(self.parse_match_clause()?)); + loop { + match self.current.kind { + TokenKind::Match => clauses.push(Clause::Match(self.parse_match_clause()?)), + TokenKind::Optional => { + self.advance(); + self.expect(TokenKind::Match)?; + clauses.push(Clause::OptionalMatch(self.parse_match_clause_body()?)); + } + _ => break, + } } // Bare pattern form: EXISTS { (a)-[r]->(b) WHERE ... } @@ -1971,6 +2038,27 @@ impl<'a> Parser<'a> { } } + /// Parses the variable scope clause of `CALL (a, b) { ... }`: the names + /// in parentheses, `*` for all outer variables, none for `()`. + fn parse_call_scope(&mut self) -> Result> { + self.expect(TokenKind::LParen)?; + let mut names = Vec::new(); + if self.current.kind == TokenKind::Star { + self.advance(); + names.push("*".to_string()); + } else if self.current.kind != TokenKind::RParen { + loop { + names.push(self.expect_identifier()?); + if self.current.kind != TokenKind::Comma { + break; + } + self.advance(); + } + } + self.expect(TokenKind::RParen)?; + Ok(names) + } + fn expect_identifier(&mut self) -> Result { if self.can_be_identifier() { let text = self.get_identifier_text(); @@ -4238,4 +4326,108 @@ mod tests { "Cypher errors should be prefixed with [Cypher], got: {err}" ); } + + /// The variable scope clause of `CALL (a, b) { ... }`: the names, `*` for + /// `(*)`, none for `()`, and `None` without a clause. A procedure call + /// has its name before any parenthesis. + #[test] + fn test_call_subquery_scope_clause() { + let scope_of = |query: &str| { + let Statement::Query(statement) = parse_ok(query) else { + panic!("expected a query: {query}"); + }; + statement + .clauses + .iter() + .find_map(|clause| match clause { + Clause::CallSubquery { scope, .. } => Some(scope.clone()), + _ => None, + }) + .unwrap_or_else(|| panic!("expected a CALL subquery: {query}")) + }; + let names = |names: &[&str]| Some(names.iter().map(ToString::to_string).collect()); + assert_eq!( + scope_of("MATCH (a), (b) CALL (a, b) { RETURN 1 AS x } RETURN x"), + names(&["a", "b"]) + ); + assert_eq!( + scope_of("MATCH (a) CALL (*) { RETURN 1 AS x } RETURN x"), + names(&["*"]) + ); + assert_eq!( + scope_of("MATCH (a) CALL () { RETURN 1 AS x } RETURN x"), + names(&[]) + ); + assert_eq!( + scope_of("MATCH (a) CALL { WITH a RETURN 1 AS x } RETURN x"), + None + ); + parse_err("MATCH (a) CALL (1) { RETURN 1 AS x } RETURN x"); + let Statement::Query(procedure) = parse_ok("CALL db.labels() YIELD label RETURN label") + else { + panic!("expected a query"); + }; + assert!(matches!(procedure.clauses[0], Clause::Call(_))); + } + + /// A CALL subquery body takes ORDER BY, SKIP and LIMIT after its RETURN, + /// and parts joined by UNION or UNION ALL (one kind per chain). + #[test] + fn test_call_subquery_body_paging_and_union() { + let call_of = |query: &str| { + let Statement::Query(statement) = parse_ok(query) else { + panic!("expected a query: {query}"); + }; + statement + .clauses + .into_iter() + .find_map(|clause| match clause { + Clause::CallSubquery { + query, + unions, + union_all, + .. + } => Some((query, unions, union_all)), + _ => None, + }) + .unwrap_or_else(|| panic!("expected a CALL subquery: {query}")) + }; + let (body, unions, _) = call_of( + "MATCH (a) CALL (a) { MATCH (a)-->(b) RETURN b ORDER BY b.x DESC SKIP 1 LIMIT 2 } RETURN b", + ); + assert!(unions.is_empty()); + assert!(matches!( + body.clauses[..], + [ + Clause::Match(_), + Clause::Return(_), + Clause::OrderBy(_), + Clause::Skip(_), + Clause::Limit(_) + ] + )); + let (_, unions, union_all) = + call_of("CALL { RETURN 1 AS x UNION RETURN 2 AS x UNION RETURN 3 AS x } RETURN x"); + assert_eq!((unions.len(), union_all), (2, false)); + let (_, unions, union_all) = + call_of("CALL { RETURN 1 AS x UNION ALL RETURN 1 AS x } RETURN x"); + assert_eq!((unions.len(), union_all), (1, true)); + parse_err("CALL { RETURN 1 AS x UNION RETURN 2 AS x UNION ALL RETURN 3 AS x } RETURN x"); + parse_err("RETURN 1 AS x UNION ALL RETURN 2 AS x UNION RETURN 3 AS x"); + } + + /// EXISTS and COUNT subqueries take OPTIONAL MATCH clauses. + #[test] + fn test_exists_and_count_take_optional_match() { + for query in [ + "MATCH (a) WHERE EXISTS { MATCH (a)-->(b) OPTIONAL MATCH (b)-->(c) } RETURN a", + "MATCH (a) RETURN COUNT { MATCH (a)-->(b) OPTIONAL MATCH (b)-->(c) WHERE c.x > 1 } AS n", + ] { + let Statement::Query(statement) = parse_ok(query) else { + panic!("expected a query: {query}"); + }; + let text = format!("{statement:?}"); + assert!(text.contains("OptionalMatch"), "{query}: {text}"); + } + } } diff --git a/crates/grafeo-adapters/src/query/gql/ast.rs b/crates/grafeo-adapters/src/query/gql/ast.rs index 59e6423b8..ee4dc9a73 100644 --- a/crates/grafeo-adapters/src/query/gql/ast.rs +++ b/crates/grafeo-adapters/src/query/gql/ast.rs @@ -233,16 +233,27 @@ pub enum QueryClause { Delete(DeleteStatement), /// A SET clause. Set(SetClause), + /// A REMOVE clause. + Remove(RemoveClause), /// A MERGE clause. Merge(MergeClause), /// A LET clause (variable bindings). Let(Vec<(String, Expression)>), + /// A WITH clause: the clauses after it read the rows it passes on. + With(WithClause), /// An inline CALL { subquery } clause (optional = OPTIONAL CALL { ... }). InlineCall { /// The inner subquery. subquery: QueryStatement, + /// Further subqueries combined with `subquery`, left to right, by + /// `UNION`, `EXCEPT`, `INTERSECT` or `OTHERWISE` (never `NEXT`). + combined: Vec<(CompositeOp, QueryStatement)>, /// Whether this is OPTIONAL CALL (left-join semantics). optional: bool, + /// The variable scope clause, `CALL (a, b) { ... }`: the outer + /// variables the subquery sees. `None` without one, when it sees all + /// of them. + scope: Option>, }, /// A CALL procedure clause within a query. CallProcedure(CallStatement), diff --git a/crates/grafeo-adapters/src/query/gql/parser.rs b/crates/grafeo-adapters/src/query/gql/parser.rs index 518b57854..634182f2e 100644 --- a/crates/grafeo-adapters/src/query/gql/parser.rs +++ b/crates/grafeo-adapters/src/query/gql/parser.rs @@ -302,56 +302,7 @@ impl<'a> Parser<'a> { } // Check for composite query operators (UNION, EXCEPT, INTERSECT, OTHERWISE) - while matches!( - self.current.kind, - TokenKind::Union | TokenKind::Except | TokenKind::Intersect | TokenKind::Otherwise - ) { - let op = match self.current.kind { - TokenKind::Union => { - self.advance(); - if self.current.kind == TokenKind::All { - self.advance(); - CompositeOp::UnionAll - } else { - // UNION DISTINCT is explicit form of the default - if self.current.kind == TokenKind::Distinct { - self.advance(); - } - CompositeOp::Union - } - } - TokenKind::Except => { - self.advance(); - if self.current.kind == TokenKind::All { - self.advance(); - CompositeOp::ExceptAll - } else { - // EXCEPT DISTINCT is explicit form of the default - if self.current.kind == TokenKind::Distinct { - self.advance(); - } - CompositeOp::Except - } - } - TokenKind::Intersect => { - self.advance(); - if self.current.kind == TokenKind::All { - self.advance(); - CompositeOp::IntersectAll - } else { - // INTERSECT DISTINCT is explicit form of the default - if self.current.kind == TokenKind::Distinct { - self.advance(); - } - CompositeOp::Intersect - } - } - TokenKind::Otherwise => { - self.advance(); - CompositeOp::Otherwise - } - _ => unreachable!(), - }; + while let Some(op) = self.parse_composite_op() { let right = self.parse_single_statement()?; left = Statement::CompositeQuery { left: Box::new(left), @@ -363,6 +314,32 @@ impl<'a> Parser<'a> { Ok(left) } + /// Consumes a set operator between two queries (`UNION [ALL | DISTINCT]`, + /// `EXCEPT ...`, `INTERSECT ...`, `OTHERWISE`), or returns `None` when + /// the current token starts none. + fn parse_composite_op(&mut self) -> Option { + let (distinct, all) = match self.current.kind { + TokenKind::Union => (CompositeOp::Union, CompositeOp::UnionAll), + TokenKind::Except => (CompositeOp::Except, CompositeOp::ExceptAll), + TokenKind::Intersect => (CompositeOp::Intersect, CompositeOp::IntersectAll), + TokenKind::Otherwise => { + self.advance(); + return Some(CompositeOp::Otherwise); + } + _ => return None, + }; + self.advance(); + if self.current.kind == TokenKind::All { + self.advance(); + return Some(all); + } + // DISTINCT is the explicit form of the default + if self.current.kind == TokenKind::Distinct { + self.advance(); + } + Some(distinct) + } + fn parse_single_statement(&mut self) -> Result { match self.current.kind { TokenKind::Match @@ -390,8 +367,8 @@ impl<'a> Parser<'a> { } } TokenKind::Call => { - if self.peek_kind() == TokenKind::LBrace { - // CALL { subquery } RETURN ... : treat as a query + if matches!(self.peek_kind(), TokenKind::LBrace | TokenKind::LParen) { + // CALL [(scope)] { subquery } RETURN ... : treat as a query self.parse_query().map(Statement::Query) } else { self.parse_call_statement().map(Statement::Call) @@ -552,20 +529,53 @@ impl<'a> Parser<'a> { }) } - /// Parses an inline CALL { subquery }. + /// Parses an inline CALL { subquery } and its variable scope clause, if + /// any: `CALL (a, b) { ... }` sees the outer variables `a` and `b`, + /// `CALL () { ... }` none, and `CALL { ... }` all of them. The body may + /// combine queries with `UNION`, `EXCEPT`, `INTERSECT` or `OTHERWISE`. /// /// ```text - /// CALL { [WITH var [, var]*] query_body RETURN ... } + /// CALL [( [var [, var]*] )] { query_body RETURN ... [UNION ...] } /// ``` - fn parse_inline_call(&mut self) -> Result { + fn parse_inline_call(&mut self, optional: bool) -> Result { self.expect(TokenKind::Call)?; + let scope = if self.current.kind == TokenKind::LParen { + self.advance(); + let mut names = Vec::new(); + if self.current.kind != TokenKind::RParen { + loop { + if !self.is_identifier() { + return Err(self.error("Expected a variable in the variable scope clause")); + } + names.push(self.get_identifier_name()); + self.advance(); + if self.current.kind != TokenKind::Comma { + break; + } + self.advance(); + } + } + self.expect(TokenKind::RParen)?; + Some(names) + } else { + None + }; self.expect(TokenKind::LBrace)?; // Parse the inner query body (MATCH ... RETURN ...) - let inner = self.parse_query()?; + let subquery = self.parse_query()?; + let mut combined = Vec::new(); + while let Some(op) = self.parse_composite_op() { + combined.push((op, self.parse_query()?)); + } self.expect(TokenKind::RBrace)?; - Ok(inner) + Ok(QueryClause::InlineCall { + subquery, + combined, + optional, + scope, + }) } /// Parses a YIELD item list: `field [AS alias] { , field [AS alias] }`. @@ -651,13 +661,9 @@ impl<'a> Parser<'a> { let pk = self.peek_kind(); if pk == TokenKind::Call { self.advance(); // consume OPTIONAL - if self.peek_kind() == TokenKind::LBrace { - // OPTIONAL CALL { subquery } - let subquery = self.parse_inline_call()?; - ordered_clauses.push(QueryClause::InlineCall { - subquery, - optional: true, - }); + if matches!(self.peek_kind(), TokenKind::LBrace | TokenKind::LParen) { + // OPTIONAL CALL [(scope)] { subquery } + ordered_clauses.push(self.parse_inline_call(true)?); } else { // OPTIONAL CALL procedure(...) let call = self.parse_call_statement()?; @@ -700,13 +706,10 @@ impl<'a> Parser<'a> { delete_clauses.push(clause); } TokenKind::Call => { - // CALL { subquery } (inline) or CALL procedure(...) (within query) - if self.peek_kind() == TokenKind::LBrace { - let subquery = self.parse_inline_call()?; - ordered_clauses.push(QueryClause::InlineCall { - subquery, - optional: false, - }); + // CALL [(scope)] { subquery } (inline) or CALL procedure(...) + // (within query): a procedure name comes before any `(` + if matches!(self.peek_kind(), TokenKind::LBrace | TokenKind::LParen) { + ordered_clauses.push(self.parse_inline_call(false)?); } else { let call = self.parse_call_statement()?; ordered_clauses.push(QueryClause::CallProcedure(call)); @@ -768,7 +771,9 @@ impl<'a> Parser<'a> { // Parse REMOVE clauses let mut remove_clauses = Vec::new(); while self.current.kind == TokenKind::Remove { - remove_clauses.push(self.parse_remove_clause()?); + let clause = self.parse_remove_clause()?; + ordered_clauses.push(QueryClause::Remove(clause.clone())); + remove_clauses.push(clause); } // Parse WITH clauses @@ -781,6 +786,7 @@ impl<'a> Parser<'a> { wc.let_bindings = self.parse_let_clause()?; } + ordered_clauses.push(QueryClause::With(wc.clone())); with_clauses.push(wc); // After WITH (+ optional LET), we can have more clauses @@ -10396,6 +10402,27 @@ mod tests { } } + // --- Clause order --- + + #[test] + fn test_ordered_clauses_keep_remove_and_with_in_place() { + let mut parser = Parser::new("MATCH (a:P) REMOVE a.w WITH a MATCH (a)-[:K]->(b) RETURN b"); + let Statement::Query(query) = parser.parse().unwrap() else { + panic!("Expected Query statement"); + }; + let kinds: Vec<&str> = query + .ordered_clauses + .iter() + .map(|clause| match clause { + QueryClause::Match(_) => "MATCH", + QueryClause::Remove(_) => "REMOVE", + QueryClause::With(_) => "WITH", + _ => "other", + }) + .collect(); + assert_eq!(kinds, ["MATCH", "REMOVE", "WITH", "MATCH"]); + } + // --- LOAD DATA --- #[test] @@ -11143,4 +11170,87 @@ mod tests { result.err() ); } + + /// An inline CALL body combines queries with set operators, left to right. + #[test] + fn test_parse_inline_call_combined_body() { + let combined_of = |query: &str| { + let Statement::Query(statement) = Parser::new(query).parse().unwrap() else { + panic!("expected a query: {query}"); + }; + statement + .ordered_clauses + .into_iter() + .find_map(|clause| match clause { + QueryClause::InlineCall { combined, .. } => { + Some(combined.into_iter().map(|(op, _)| op).collect::>()) + } + _ => None, + }) + .unwrap_or_else(|| panic!("expected an inline CALL: {query}")) + }; + assert_eq!(combined_of("CALL { RETURN 1 AS x } RETURN x"), []); + assert_eq!( + combined_of( + "MATCH (a) CALL (a) { RETURN 1 AS x UNION ALL RETURN 2 AS x EXCEPT RETURN 3 AS x } RETURN x" + ), + [CompositeOp::UnionAll, CompositeOp::Except] + ); + assert_eq!( + combined_of("CALL { RETURN 1 AS x UNION DISTINCT RETURN 1 AS x } RETURN x"), + [CompositeOp::Union] + ); + } + + /// The variable scope clause of an inline CALL: `(a, b)` names the outer + /// variables the subquery sees, `()` none, and no clause all of them. A + /// procedure call has its name before any parenthesis. + #[test] + fn test_parse_inline_call_scope_clause() { + let scope_of = |query: &str| { + let Statement::Query(statement) = Parser::new(query).parse().unwrap() else { + panic!("expected a query: {query}"); + }; + statement + .ordered_clauses + .iter() + .find_map(|clause| match clause { + QueryClause::InlineCall { + scope, optional, .. + } => Some((scope.clone(), *optional)), + _ => None, + }) + .unwrap_or_else(|| panic!("expected an inline CALL: {query}")) + }; + let names = |names: &[&str]| Some(names.iter().map(ToString::to_string).collect()); + assert_eq!( + scope_of("MATCH (a), (b) CALL (a, b) { RETURN 1 AS x } RETURN x"), + (names(&["a", "b"]), false) + ); + assert_eq!( + scope_of("MATCH (a) CALL () { RETURN 1 AS x } RETURN x"), + (names(&[]), false) + ); + assert_eq!( + scope_of("MATCH (a) CALL { RETURN 1 AS x } RETURN x"), + (None, false) + ); + assert_eq!( + scope_of("MATCH (a) OPTIONAL CALL (a) { RETURN 1 AS x } RETURN x"), + (names(&["a"]), true) + ); + assert_eq!( + scope_of("CALL () { RETURN 1 AS x } RETURN x"), + (names(&[]), false) + ); + assert!( + Parser::new("MATCH (a) CALL (1) { RETURN 1 AS x } RETURN x") + .parse() + .is_err() + ); + assert!(matches!( + Parser::new("CALL db.labels()").parse().unwrap(), + Statement::Call(_) + )); + } } diff --git a/crates/grafeo-common/src/testing/child_process.rs b/crates/grafeo-common/src/testing/child_process.rs index 8d9305d94..fef0c989b 100644 --- a/crates/grafeo-common/src/testing/child_process.rs +++ b/crates/grafeo-common/src/testing/child_process.rs @@ -8,17 +8,26 @@ //! opens it again at that moment finds it "locked by another process". //! //! [`run`] and [`output`] start a child only while no database lock is being -//! taken, and the storage layer takes its locks inside [`lock_acquisition`], -//! which waits while a child starts. Outside tests nothing starts children -//! this way, so the guard is never contended. +//! taken, and the storage layer takes its locks through [`take_lock`], which +//! waits while a child starts. Outside tests nothing starts children this +//! way, so the guard is never contended and a held lock fails at once. use std::io; use std::process::{Command, ExitStatus, Output, Stdio}; -use std::sync::{PoisonError, RwLock, RwLockReadGuard}; +use std::sync::atomic::{AtomicBool, Ordering}; +use std::sync::{PoisonError, RwLock}; +use std::time::{Duration, Instant}; /// Held for writing while a child starts, for reading while a lock is taken. static CHILD_START: RwLock<()> = RwLock::new(()); +/// Whether this process has started a child with [`run`] or [`output`]. +static CHILDREN_STARTED: AtomicBool = AtomicBool::new(false); + +/// How long [`take_lock`] tries a held lock again in a process that starts +/// children. +const RETRY_WINDOW: Duration = Duration::from_millis(250); + /// Starts `command` and waits for it, like [`Command::status`]. /// /// # Errors @@ -29,8 +38,10 @@ pub fn run(command: &mut Command) -> io::Result { let _starting = CHILD_START.write().unwrap_or_else(PoisonError::into_inner); // `spawn` reports a failed exec, so it returns only once the child // runs its own program, which closes the copies (Rust opens files - // close-on-exec). - command.spawn()? + // close-on-exec), give or take the moment `take_lock` allows for. + let child = command.spawn()?; + CHILDREN_STARTED.store(true, Ordering::Relaxed); + child }; child.wait() } @@ -44,19 +55,47 @@ pub fn run(command: &mut Command) -> io::Result { pub fn output(command: &mut Command) -> io::Result { let child = { let _starting = CHILD_START.write().unwrap_or_else(PoisonError::into_inner); - command + let child = command .stdin(Stdio::null()) .stdout(Stdio::piped()) .stderr(Stdio::piped()) - .spawn()? + .spawn()?; + CHILDREN_STARTED.store(true, Ordering::Relaxed); + child }; child.wait_with_output() } -/// Holds off [`run`] and [`output`] while the caller takes a database lock, -/// so no starting child holds a copy of the file being locked. -pub fn lock_acquisition() -> RwLockReadGuard<'static, ()> { - CHILD_START.read().unwrap_or_else(PoisonError::into_inner) +/// Takes a database lock with `try_lock` while no child starts, so no +/// starting child holds a copy of the file being locked. +/// +/// On Linux the kernel lets the parent of a starting child go on just before +/// it closes the child's close-on-exec files, so for a moment after [`run`] +/// or [`output`] started a child, that child can still hold the lock of a +/// database a test has just closed. In a process that started children this +/// way, a lock that `try_lock` finds held (an error `held` accepts) is +/// therefore tried again for up to 250 ms before the error stands. Other +/// errors, and every error in other processes, are returned at once. +/// +/// # Errors +/// +/// Returns the last error of `try_lock`. +pub fn take_lock( + mut try_lock: impl FnMut() -> Result, + held: impl Fn(&E) -> bool, +) -> Result { + let _no_child_start = CHILD_START.read().unwrap_or_else(PoisonError::into_inner); + let deadline = CHILDREN_STARTED + .load(Ordering::Relaxed) + .then(|| Instant::now() + RETRY_WINDOW); + loop { + match (try_lock(), deadline) { + (Err(error), Some(deadline)) if held(&error) && Instant::now() < deadline => { + std::thread::sleep(Duration::from_millis(2)); + } + (result, _) => return result, + } + } } #[cfg(test)] @@ -91,4 +130,49 @@ mod tests { "{stdout}" ); } + + /// A lock released a moment after the first try, as by a starting child + /// that still holds a copy of it, is taken once it is free; a lock that + /// stays held fails after the window. + #[test] + #[cfg_attr(miri, ignore = "Miri cannot start child processes")] + fn a_lock_released_a_moment_later_is_taken() { + assert!(run(&mut list_tests()).unwrap().success()); + let path = std::env::temp_dir().join(format!("grafeo-take-lock-{}", std::process::id())); + let open = || { + std::fs::OpenOptions::new() + .read(true) + .write(true) + .create(true) + .truncate(false) + .open(&path) + .unwrap() + }; + + let held = open(); + held.lock().unwrap(); + let release = std::thread::spawn(move || { + std::thread::sleep(Duration::from_millis(20)); + drop(held); + }); + let held = + |error: &std::fs::TryLockError| matches!(error, std::fs::TryLockError::WouldBlock); + let ours = open(); + assert!(take_lock(|| ours.try_lock(), held).is_ok()); + release.join().unwrap(); + + let started = Instant::now(); + assert!(take_lock(|| open().try_lock(), held).is_err()); + assert!(started.elapsed() >= RETRY_WINDOW); + + // An error other than a held lock is returned at once. + let started = Instant::now(); + assert_eq!( + take_lock(|| Err::<(), _>("broken"), |_| false), + Err("broken") + ); + assert!(started.elapsed() < RETRY_WINDOW); + drop(ours); + std::fs::remove_file(&path).unwrap(); + } } diff --git a/crates/grafeo-core/benches/top_k.rs b/crates/grafeo-core/benches/top_k.rs index 6a2876003..b2ada6ad1 100644 --- a/crates/grafeo-core/benches/top_k.rs +++ b/crates/grafeo-core/benches/top_k.rs @@ -88,12 +88,7 @@ fn bench_top_k_vs_sort_limit_time(c: &mut Criterion) { group.bench_function(format!("top_k/N={n}"), |b| { b.iter(|| { let source = Box::new(VecSource::new(&values, 2048)); - let mut op = TopKOperator::new( - source, - vec![SortKey::descending(0)], - K, - vec![LogicalType::Int64], - ); + let mut op = TopKOperator::new(source, vec![SortKey::descending(0)], K); black_box(drain(&mut op)) }); }); @@ -101,12 +96,8 @@ fn bench_top_k_vs_sort_limit_time(c: &mut Criterion) { group.bench_function(format!("sort_limit/N={n}"), |b| { b.iter(|| { let source = Box::new(VecSource::new(&values, 2048)); - let sort = Box::new(SortOperator::new( - source, - vec![SortKey::descending(0)], - vec![LogicalType::Int64], - )); - let mut limit = LimitOperator::new(sort, K, vec![LogicalType::Int64]); + let sort = Box::new(SortOperator::new(source, vec![SortKey::descending(0)])); + let mut limit = LimitOperator::new(sort, K); black_box(drain(&mut limit)) }); }); @@ -125,12 +116,7 @@ fn bench_top_k_drain(c: &mut Criterion) { group.bench_function(format!("N={n}"), |b| { b.iter(|| { let source = Box::new(VecSource::new(&values, 2048)); - let mut op = TopKOperator::new( - source, - vec![SortKey::descending(0)], - K, - vec![LogicalType::Int64], - ); + let mut op = TopKOperator::new(source, vec![SortKey::descending(0)], K); black_box(drain(&mut op)) }); }); diff --git a/crates/grafeo-core/src/execution/chunk.rs b/crates/grafeo-core/src/execution/chunk.rs index c755aa5b6..866ee9481 100644 --- a/crates/grafeo-core/src/execution/chunk.rs +++ b/crates/grafeo-core/src/execution/chunk.rs @@ -141,6 +141,15 @@ impl DataChunk { &self.columns } + /// Returns the types of the columns. + #[must_use] + pub fn column_types(&self) -> Vec { + self.columns + .iter() + .map(|column| column.data_type().clone()) + .collect() + } + /// Returns the total number of rows (ignoring selection). #[must_use] pub fn total_row_count(&self) -> usize { @@ -419,7 +428,8 @@ impl DataChunk { /// Returns a slice of this chunk. /// - /// Returns a new DataChunk containing rows [offset, offset + count). + /// Returns a new DataChunk containing the selected rows [offset, offset + + /// count). Each column keeps its type, so every value is copied as it is. #[must_use] pub fn slice(&self, offset: usize, count: usize) -> DataChunk { if offset >= self.len() || count == 0 { @@ -430,7 +440,7 @@ impl DataChunk { let mut result_columns = Vec::with_capacity(self.columns.len()); for col in &self.columns { - let mut new_col = ValueVector::new(); + let mut new_col = ValueVector::with_capacity(col.data_type().clone(), actual_count); for i in offset..(offset + actual_count) { let actual_idx = if let Some(sel) = &self.selection { sel.get(i).unwrap_or(i) @@ -460,6 +470,81 @@ impl DataChunk { } } +/// The column types for rows taken from the chunks of one input. +/// +/// A column keeps its type while every chunk with rows has that type there and +/// becomes [`LogicalType::Any`] where they differ; a chunk without rows counts +/// only until one with rows comes. Rows copied into columns of these types keep +/// every value: a typed column stores a value of another type as that type's +/// default (see [`ValueVector::push_value`]), so an operator that only reorders, +/// cuts or deduplicates rows must not copy them by a declared schema. +#[derive(Debug, Default, Clone)] +pub(crate) struct ColumnTypes { + /// The types so far (none before the first chunk). + types: Option>, + /// Whether they come from chunks with rows. + from_rows: bool, +} + +impl ColumnTypes { + /// Takes the column types of `chunk` into account. + pub(crate) fn add(&mut self, chunk: &DataChunk) { + let has_rows = chunk.row_count() > 0; + match &mut self.types { + Some(types) if self.from_rows && has_rows => { + for (known, column) in types.iter_mut().zip(chunk.columns()) { + if known != column.data_type() { + *known = LogicalType::Any; + } + } + } + // A chunk without rows says nothing about the values: its types + // count only while no other chunk gave any. + Some(_) if self.from_rows || !has_rows => {} + // The first chunk, or the first with rows after chunks without. + _ => { + self.types = Some(chunk.column_types()); + self.from_rows = has_rows; + } + } + } + + /// The column types of the chunks seen so far (none before the first). + pub(crate) fn types(&self) -> &[LogicalType] { + self.types.as_deref().unwrap_or(&[]) + } +} + +/// The type for a column copied from an input column of type `input`, which +/// the planner declared as `declared`: the input's, so that every value is +/// copied as it is, except where the input's is [`LogicalType::Any`] and the +/// planner declared a node or edge: an ID whose kind the input lost is that +/// entity again. +pub(crate) fn copied_column_type( + input: &LogicalType, + declared: Option<&LogicalType>, +) -> LogicalType { + match (input, declared) { + (LogicalType::Any, Some(entity @ (LogicalType::Node | LogicalType::Edge))) => { + entity.clone() + } + _ => input.clone(), + } +} + +/// [`copied_column_type`] for each of the `input` types, declared as the +/// `declared` types at the same positions. +pub(crate) fn copied_column_types( + input: &[LogicalType], + declared: &[LogicalType], +) -> Vec { + input + .iter() + .enumerate() + .map(|(i, input_type)| copied_column_type(input_type, declared.get(i))) + .collect() +} + impl Clone for DataChunk { fn clone(&self) -> Self { Self { @@ -539,6 +624,55 @@ mod tests { use super::*; use grafeo_common::types::Value; + /// Chunks without rows do not turn a node column into `Any`: their types + /// count only until a chunk with rows comes. Chunks with rows of different + /// types do. + #[test] + fn column_types_ignore_chunks_without_rows() { + let empty = || DataChunk::with_capacity(&[LogicalType::Any], 0); + let nodes = || { + let mut builder = DataChunkBuilder::new(&[LogicalType::Node]); + builder + .column_mut(0) + .unwrap() + .push_node_id(grafeo_common::types::NodeId::new(1)); + builder.advance_row(); + builder.finish() + }; + + let mut types = ColumnTypes::default(); + types.add(&empty()); + assert_eq!(types.types(), [LogicalType::Any]); + types.add(&nodes()); + types.add(&empty()); + assert_eq!(types.types(), [LogicalType::Node]); + + let mut edges = DataChunkBuilder::new(&[LogicalType::Edge]); + edges + .column_mut(0) + .unwrap() + .push_edge_id(grafeo_common::types::EdgeId::new(1)); + edges.advance_row(); + types.add(&edges.finish()); + assert_eq!(types.types(), [LogicalType::Any]); + } + + /// A copied column keeps its input type; a declared node or edge only + /// gives an `Any` column back its entity kind, never a scalar type. + #[test] + fn copied_columns_keep_the_input_type_or_a_declared_entity() { + use LogicalType::{Any, Edge, Int64, Node}; + let text = LogicalType::String; + + assert_eq!( + copied_column_types( + &[Any, Any, Any, Node, Int64, text.clone(), Any], + &[Edge, Node, Int64, Edge, Any, Node], + ), + [Edge, Node, Any, Node, Int64, text, Any] + ); + } + #[test] fn test_chunk_creation() { let schema = [LogicalType::Int64, LogicalType::String]; diff --git a/crates/grafeo-core/src/execution/operators/aggregate.rs b/crates/grafeo-core/src/execution/operators/aggregate.rs index 27478d5ab..505af75dd 100644 --- a/crates/grafeo-core/src/execution/operators/aggregate.rs +++ b/crates/grafeo-core/src/execution/operators/aggregate.rs @@ -17,7 +17,7 @@ use grafeo_common::types::{LogicalType, PropertyKey, Value}; use super::accumulator::{AggregateExpr, AggregateFunction, HashableValue}; use super::{Operator, OperatorError, OperatorResult}; use crate::execution::DataChunk; -use crate::execution::chunk::DataChunkBuilder; +use crate::execution::chunk::{ColumnTypes, DataChunkBuilder, copied_column_type}; /// State for a single aggregation computation. /// @@ -493,7 +493,7 @@ impl AggregateState { Value::Null } else { let mut sorted = values.clone(); - sorted.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal)); + sorted.sort_by(|a, b| compare_floats(*a, *b)); // Index calculation per SQL standard: floor(p * (n - 1)) // reason: percentile index is bounded by sorted.len(), fits usize #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)] @@ -507,7 +507,7 @@ impl AggregateState { Value::Null } else { let mut sorted = values.clone(); - sorted.sort_by(|a, b| a.partial_cmp(b).unwrap_or(std::cmp::Ordering::Equal)); + sorted.sort_by(|a, b| compare_floats(*a, *b)); // Linear interpolation per SQL standard let rank = percentile * (sorted.len() - 1) as f64; // reason: rank is bounded by sorted.len() - 1, fits usize @@ -632,7 +632,7 @@ impl AggregateState { } } -use super::value_utils::{compare_values, value_to_f64}; +use super::value_utils::{compare_floats, compare_values, value_to_f64}; /// Converts a Value to its string representation for GROUP_CONCAT. fn agg_value_to_string(val: &Value) -> String { @@ -746,7 +746,8 @@ impl GroupKey { /// Hash-based aggregate operator. /// -/// Groups input by key columns and computes aggregations for each group. +/// Groups input by key columns and computes aggregations for each group. A +/// group key keeps its input column's type: a node or edge stays one. pub struct HashAggregateOperator { /// Child operator to read from. child: Box, @@ -756,6 +757,8 @@ pub struct HashAggregateOperator { aggregates: Vec, /// Output schema. output_schema: Vec, + /// The column types of the input chunks. + input_types: ColumnTypes, /// Ordered map: group key -> aggregate states (IndexMap for deterministic iteration order). groups: IndexMap>, /// Whether aggregation is complete. @@ -783,12 +786,27 @@ impl HashAggregateOperator { group_columns, aggregates, output_schema, + input_types: ColumnTypes::default(), groups: IndexMap::new(), aggregation_complete: false, results: None, } } + /// The output column types: the group keys' input types (see + /// [`ColumnTypes`]; declared before any input), then the aggregates'. + fn output_types(&self) -> Vec { + let mut types = self.output_schema.clone(); + for (i, &column) in self.group_columns.iter().enumerate() { + if let (Some(key_type), Some(input_type)) = + (types.get_mut(i), self.input_types.types().get(column)) + { + *key_type = copied_column_type(input_type, Some(&*key_type)); + } + } + types + } + /// Decomposes this operator for push-based conversion. pub fn into_parts(self) -> (Box, Vec, Vec) { (self.child, self.group_columns, self.aggregates) @@ -797,6 +815,7 @@ impl HashAggregateOperator { /// Performs the aggregation. fn aggregate(&mut self) -> Result<(), OperatorError> { while let Some(chunk) = self.child.next()? { + self.input_types.add(&chunk); for row in chunk.selected_indices() { let key = GroupKey::from_row(&chunk, row, &self.group_columns); @@ -905,11 +924,12 @@ impl Operator for HashAggregateOperator { return Ok(Some(builder.finish())); } + let types = self.output_types(); let Some(results) = &mut self.results else { return Ok(None); }; - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, 2048); + let mut builder = DataChunkBuilder::with_capacity(&types, 2048); for (key, states) in results.by_ref() { // Output group key columns @@ -944,6 +964,7 @@ impl Operator for HashAggregateOperator { fn reset(&mut self) { self.child.reset(); + self.input_types = ColumnTypes::default(); self.groups.clear(); self.aggregation_complete = false; self.results = None; @@ -1143,6 +1164,47 @@ mod tests { } } + /// A node group key keeps its column type; the count its declared one. + #[test] + fn group_keys_keep_their_column_types() { + use grafeo_common::types::NodeId; + + let mut builder = DataChunkBuilder::new(&[LogicalType::Node]); + for node in [101, 101, 102] { + builder + .column_mut(0) + .unwrap() + .push_node_id(NodeId::new(node)); + builder.advance_row(); + } + let mut aggregate = HashAggregateOperator::new( + Box::new(MockOperator::new(vec![builder.finish()])), + vec![0], + vec![AggregateExpr::count_star()], + vec![LogicalType::Any, LogicalType::Int64], + ); + + let chunk = aggregate.next().unwrap().unwrap(); + assert_eq!( + chunk.column_types(), + [LogicalType::Node, LogicalType::Int64] + ); + let mut groups: Vec<(u64, Option)> = chunk + .selected_indices() + .map(|row| { + ( + chunk.column(0).unwrap().get_node_id(row).unwrap().as_u64(), + chunk.column(1).unwrap().get_value(row), + ) + }) + .collect(); + groups.sort_by_key(|(node, _)| *node); + assert_eq!( + groups, + [(101, Some(Value::Int64(2))), (102, Some(Value::Int64(1)))] + ); + } + fn create_test_chunk() -> DataChunk { // Create: [(group, value)] = [(1, 10), (1, 20), (2, 30), (2, 40), (2, 50)] let mut builder = DataChunkBuilder::new(&[LogicalType::Int64, LogicalType::Int64]); @@ -1562,6 +1624,35 @@ mod tests { assert!((p100 - 9.0).abs() < 0.01); } + /// NaN sorts after every other number in percentiles too. Sorting it as + /// equal to everything was not a total order: the sort could panic. + #[test] + fn percentiles_sort_nan_last() { + let input = || { + let mut builder = DataChunkBuilder::new(&[LogicalType::Float64]); + for value in [3.0, f64::NAN, 1.0, 2.0, f64::NAN, 4.0, 0.5, 2.5] { + builder.column_mut(0).unwrap().push_float64(value); + builder.advance_row(); + } + MockOperator::new(vec![builder.finish()]) + }; + let mut agg = SimpleAggregateOperator::new( + Box::new(input()), + vec![ + AggregateExpr::percentile_disc(0, 0.5), + AggregateExpr::percentile_cont(0, 0.5), + AggregateExpr::percentile_disc(0, 1.0), + ], + vec![LogicalType::Float64; 3], + ); + + let result = agg.next().unwrap().unwrap(); + // Sorted: [0.5, 1, 2, 2.5, 3, 4, NaN, NaN]; rank 0.5 * 7 = 3.5. + assert_eq!(result.column(0).unwrap().get_float64(0), Some(2.5)); + assert_eq!(result.column(1).unwrap().get_float64(0), Some(2.75)); + assert!(result.column(2).unwrap().get_float64(0).unwrap().is_nan()); + } + #[test] fn test_stdev_single_value() { // Single value should return null for sample stdev diff --git a/crates/grafeo-core/src/execution/operators/apply.rs b/crates/grafeo-core/src/execution/operators/apply.rs index f8db19cac..3ccd0de9d 100644 --- a/crates/grafeo-core/src/execution/operators/apply.rs +++ b/crates/grafeo-core/src/execution/operators/apply.rs @@ -14,6 +14,7 @@ use grafeo_common::types::{LogicalType, Value}; use super::parameter_scan::ParameterState; use super::{DataChunk, Operator, OperatorResult}; +use crate::execution::chunk::ColumnTypes; use crate::execution::vector::ValueVector; /// Apply (lateral join) operator. @@ -25,6 +26,9 @@ use crate::execution::vector::ValueVector; /// When `param_state` is set, outer row values for the specified column indices /// are injected into the shared [`ParameterState`] before each inner execution, /// allowing the inner plan's [`ParameterScanOperator`](super::ParameterScanOperator) to read them. +/// +/// A row keeps the column types of the rows it combines (and the injected +/// values those of their outer columns): a node or edge stays one. pub struct ApplyOperator { outer: Box, inner: Box, @@ -58,6 +62,8 @@ enum ApplyState { outer_row: usize, /// Accumulated output rows (combined outer + inner). output: Vec>, + /// The column types of the inner chunks of these rows. + inner_types: ColumnTypes, }, /// All outer input exhausted. Done, @@ -141,14 +147,35 @@ impl ApplyOperator { values } - /// Builds a DataChunk from accumulated rows. - fn build_chunk(rows: &[Vec]) -> DataChunk { + /// The column types of output rows from `outer_chunk`: its own, then the + /// EXISTS flag's or the inner chunks' (see [`ColumnTypes`]). + fn output_types( + exists_flag: bool, + exists_mode: Option, + outer_chunk: &DataChunk, + inner_types: &ColumnTypes, + ) -> Vec { + let mut types = outer_chunk.column_types(); + if exists_flag { + types.push(LogicalType::Bool); + } else if exists_mode.is_none() { + types.extend_from_slice(inner_types.types()); + } + types + } + + /// Builds a DataChunk from accumulated rows, in `types` (a column past + /// them, from inner chunks never seen, is of any type). + fn build_chunk(rows: &[Vec], types: &[LogicalType]) -> DataChunk { if rows.is_empty() { return DataChunk::empty(); } let num_cols = rows[0].len(); let mut columns: Vec = (0..num_cols) - .map(|_| ValueVector::with_capacity(LogicalType::Any, rows.len())) + .map(|i| { + let column_type = types.get(i).cloned().unwrap_or(LogicalType::Any); + ValueVector::with_capacity(column_type, rows.len()) + }) .collect(); for row in rows { @@ -172,6 +199,7 @@ impl Operator for ApplyOperator { outer_chunk: chunk, outer_row: 0, output: Vec::new(), + inner_types: ColumnTypes::default(), }; } None => { @@ -183,20 +211,31 @@ impl Operator for ApplyOperator { outer_chunk, outer_row, output, + inner_types, } => { let selected: Vec = outer_chunk.selected_indices().collect(); while *outer_row < selected.len() { let row = selected[*outer_row]; let outer_values = Self::extract_row(outer_chunk, row); - // Inject outer values into the inner plan's parameter state + // Inject outer values into the inner plan's parameter state, + // with the types of the columns they come from if let Some(ref param_state) = self.param_state { let injected: Vec = self .param_col_indices .iter() .map(|&idx| outer_values.get(idx).cloned().unwrap_or(Value::Null)) .collect(); - param_state.set_values(injected); + let types: Vec = self + .param_col_indices + .iter() + .map(|&idx| { + outer_chunk + .column(idx) + .map_or(LogicalType::Any, |col| col.data_type().clone()) + }) + .collect(); + param_state.set_typed_values(injected, types); } // Reset and run inner plan for this outer row @@ -218,6 +257,7 @@ impl Operator for ApplyOperator { } else { let pre_len = output.len(); while let Some(inner_chunk) = self.inner.next()? { + inner_types.add(&inner_chunk); for inner_row in inner_chunk.selected_indices() { let inner_values = Self::extract_row(&inner_chunk, inner_row); let mut combined = outer_values.clone(); @@ -241,15 +281,28 @@ impl Operator for ApplyOperator { // Flush when we have enough rows if output.len() >= 1024 { - let chunk = Self::build_chunk(output); + let types = Self::output_types( + self.exists_flag, + self.exists_mode, + outer_chunk, + inner_types, + ); + let chunk = Self::build_chunk(output, &types); output.clear(); + *inner_types = ColumnTypes::default(); return Ok(Some(chunk)); } } // Finished this outer chunk; flush any remaining output if !output.is_empty() { - let chunk = Self::build_chunk(output); + let types = Self::output_types( + self.exists_flag, + self.exists_mode, + outer_chunk, + inner_types, + ); + let chunk = Self::build_chunk(output, &types); output.clear(); self.state = ApplyState::Init; return Ok(Some(chunk)); @@ -277,3 +330,90 @@ impl Operator for ApplyOperator { self } } + +#[cfg(test)] +mod tests { + use grafeo_common::types::NodeId; + + use super::super::ParameterScanOperator; + use super::*; + use crate::execution::chunk::DataChunkBuilder; + + struct MockOperator { + chunks: Vec, + position: usize, + } + + impl Operator for MockOperator { + fn next(&mut self) -> OperatorResult { + let chunk = self.chunks.get(self.position).cloned(); + self.position += 1; + Ok(chunk) + } + + fn reset(&mut self) { + self.position = 0; + } + + fn name(&self) -> &'static str { + "Mock" + } + + fn into_any(self: Box) -> Box { + self + } + } + + /// Outer rows with the nodes 101 and 102. + fn outer_nodes() -> Box { + let mut builder = DataChunkBuilder::new(&[LogicalType::Node]); + for id in [101, 102] { + builder.column_mut(0).unwrap().push_node_id(NodeId::new(id)); + builder.advance_row(); + } + Box::new(MockOperator { + chunks: vec![builder.finish()], + position: 0, + }) + } + + /// The inner plan reads the injected outer node as a node, and the joined + /// rows keep the column types of both sides. + #[test] + fn apply_keeps_the_column_types_of_outer_and_inner_rows() { + let state = Arc::new(ParameterState::new(vec!["a".to_string()])); + let inner = Box::new(ParameterScanOperator::new(Arc::clone(&state))); + let mut apply = ApplyOperator::new_correlated(outer_nodes(), inner, state, vec![0]); + + let chunk = apply.next().unwrap().unwrap(); + assert_eq!(chunk.column_types(), [LogicalType::Node, LogicalType::Node]); + let rows: Vec<(u64, u64)> = chunk + .selected_indices() + .map(|row| { + ( + chunk.column(0).unwrap().get_node_id(row).unwrap().as_u64(), + chunk.column(1).unwrap().get_node_id(row).unwrap().as_u64(), + ) + }) + .collect(); + assert_eq!(rows, [(101, 101), (102, 102)]); + assert!(chunk.column(1).unwrap().get_edge_id(0).is_none()); + assert!(apply.next().unwrap().is_none()); + } + + /// The EXISTS flag is a boolean column after the outer row's own types. + #[test] + fn an_exists_flag_follows_the_outer_types() { + let state = Arc::new(ParameterState::new(vec!["a".to_string()])); + let inner = Box::new(ParameterScanOperator::new(Arc::clone(&state))); + let mut apply = + ApplyOperator::new_correlated(outer_nodes(), inner, state, vec![0]).with_exists_flag(); + + let chunk = apply.next().unwrap().unwrap(); + assert_eq!(chunk.column_types(), [LogicalType::Node, LogicalType::Bool]); + assert_eq!( + chunk.column(1).unwrap().get_value(0), + Some(Value::Bool(true)) + ); + } +} diff --git a/crates/grafeo-core/src/execution/operators/distinct.rs b/crates/grafeo-core/src/execution/operators/distinct.rs index 91ac842ae..88d74af12 100644 --- a/crates/grafeo-core/src/execution/operators/distinct.rs +++ b/crates/grafeo-core/src/execution/operators/distinct.rs @@ -5,7 +5,7 @@ use std::collections::HashSet; -use grafeo_common::types::{LogicalType, Value}; +use grafeo_common::types::Value; use super::{Operator, OperatorResult}; use crate::execution::DataChunk; @@ -56,25 +56,23 @@ impl RowKey { /// Distinct operator. /// -/// Removes duplicate rows from the input. Can operate on all columns or a subset. +/// Removes duplicate rows from the input. Can operate on all columns or a +/// subset. The rows it keeps keep their columns' types and values. pub struct DistinctOperator { /// Child operator. child: Box, /// Columns to consider for uniqueness (None = all columns). distinct_columns: Option>, - /// Output schema. - output_schema: Vec, /// Set of seen row keys. seen: HashSet, } impl DistinctOperator { /// Creates a new distinct operator that considers all columns. - pub fn new(child: Box, output_schema: Vec) -> Self { + pub fn new(child: Box) -> Self { Self { child, distinct_columns: None, - output_schema, seen: HashSet::new(), } } @@ -85,15 +83,10 @@ impl DistinctOperator { } /// Creates a distinct operator that considers only specified columns. - pub fn on_columns( - child: Box, - columns: Vec, - output_schema: Vec, - ) -> Self { + pub fn on_columns(child: Box, columns: Vec) -> Self { Self { child, distinct_columns: Some(columns), - output_schema, seen: HashSet::new(), } } @@ -106,7 +99,7 @@ impl Operator for DistinctOperator { return Ok(None); }; - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, 2048); + let mut builder = DataChunkBuilder::with_capacity(&chunk.column_types(), 2048); for row in chunk.selected_indices() { let key = match &self.distinct_columns { @@ -160,6 +153,7 @@ impl Operator for DistinctOperator { mod tests { use super::*; use crate::execution::chunk::DataChunkBuilder; + use grafeo_common::types::LogicalType; struct MockOperator { chunks: Vec, @@ -224,10 +218,7 @@ mod tests { fn test_distinct_all_columns() { let mock = MockOperator::new(vec![create_chunk_with_duplicates()]); - let mut distinct = DistinctOperator::new( - Box::new(mock), - vec![LogicalType::Int64, LogicalType::String], - ); + let mut distinct = DistinctOperator::new(Box::new(mock)); let mut results = Vec::new(); while let Some(chunk) = distinct.next().unwrap() { @@ -262,11 +253,7 @@ mod tests { fn test_distinct_single_column() { let mock = MockOperator::new(vec![create_chunk_with_duplicates()]); - let mut distinct = DistinctOperator::on_columns( - Box::new(mock), - vec![0], // Only consider first column - vec![LogicalType::Int64, LogicalType::String], - ); + let mut distinct = DistinctOperator::on_columns(Box::new(mock), vec![0]); let mut results = Vec::new(); while let Some(chunk) = distinct.next().unwrap() { @@ -298,7 +285,7 @@ mod tests { let mock = MockOperator::new(vec![builder1.finish(), builder2.finish()]); - let mut distinct = DistinctOperator::new(Box::new(mock), vec![LogicalType::Int64]); + let mut distinct = DistinctOperator::new(Box::new(mock)); let mut results = Vec::new(); while let Some(chunk) = distinct.next().unwrap() { @@ -316,7 +303,7 @@ mod tests { #[test] fn test_distinct_into_any() { let mock = MockOperator::new(vec![]); - let op = DistinctOperator::new(Box::new(mock), vec![LogicalType::Int64]); + let op = DistinctOperator::new(Box::new(mock)); let any = Box::new(op).into_any(); assert!(any.downcast::().is_ok()); } @@ -324,11 +311,7 @@ mod tests { #[test] fn test_distinct_into_parts() { let mock = MockOperator::new(vec![]); - let op = DistinctOperator::on_columns( - Box::new(mock), - vec![0, 2], - vec![LogicalType::Int64, LogicalType::String, LogicalType::Int64], - ); + let op = DistinctOperator::on_columns(Box::new(mock), vec![0, 2]); let (mut child, distinct_columns) = op.into_parts(); assert_eq!(distinct_columns, Some(vec![0, 2])); assert!(child.next().unwrap().is_none()); @@ -337,7 +320,7 @@ mod tests { #[test] fn test_distinct_into_parts_all_columns() { let mock = MockOperator::new(vec![]); - let op = DistinctOperator::new(Box::new(mock), vec![LogicalType::Int64]); + let op = DistinctOperator::new(Box::new(mock)); let (_child, distinct_columns) = op.into_parts(); assert!(distinct_columns.is_none()); } diff --git a/crates/grafeo-core/src/execution/operators/factorized_expand.rs b/crates/grafeo-core/src/execution/operators/factorized_expand.rs index 1f8ceba35..dcce4d5b9 100644 --- a/crates/grafeo-core/src/execution/operators/factorized_expand.rs +++ b/crates/grafeo-core/src/execution/operators/factorized_expand.rs @@ -605,8 +605,9 @@ pub struct ExpandStep { pub struct LazyFactorizedChainOperator { /// The graph store. store: Arc, - /// The source operator (filter, scan, etc). - source: Option>, + /// The source operator (filter, scan, etc). It stays here, so a reset + /// can run the chain again. + source: Box, /// The expand steps to execute. steps: Vec, /// Transaction ID for MVCC visibility. @@ -632,7 +633,7 @@ impl LazyFactorizedChainOperator { ) -> Self { Self { store, - source: Some(source), + source, steps, transaction_id: None, viewing_epoch: None, @@ -666,13 +667,19 @@ impl LazyFactorizedChainOperator { /// factorized chunk without flattening, allowing O(n) aggregation instead /// of O(n²) or worse. fn execute_factorized(&mut self) -> Result, OperatorError> { - let Some(source) = self.source.take() else { + // The chain expands the source's rows as one chunk; reading them here + // keeps the source, so a reset can run the chain again (a correlated + // subquery runs it once per outer row). + let Some(input) = FactorizedExpandChain::collect_all_batches(&mut *self.source)? else { return Ok(None); }; // Build and execute the chain - let mut chain = FactorizedExpandChain::new(Arc::clone(&self.store), source) - .with_read_only(self.read_only); + let mut chain = FactorizedExpandChain::new( + Arc::clone(&self.store), + Box::new(SingleChunkOperator::new(input)), + ) + .with_read_only(self.read_only); if let Some(epoch) = self.viewing_epoch { chain = chain.with_transaction_context(epoch, self.transaction_id); @@ -734,10 +741,10 @@ impl Operator for LazyFactorizedChainOperator { } fn reset(&mut self) { - // Cannot reset - source has been consumed + self.source.reset(); self.result = None; self.factorized_result = None; - self.executed = true; + self.executed = false; } fn name(&self) -> &'static str { @@ -986,6 +993,41 @@ mod tests { assert!(any.downcast::().is_ok()); } + /// After a reset the chain runs again, as a correlated subquery runs it + /// once per outer row: Alix knows Gus, who lives in Berlin, both times. + #[test] + fn lazy_chain_runs_again_after_a_reset() { + let store = Arc::new(LpgStore::new().unwrap()); + let alix = store.create_node(&["Person"]); + let gus = store.create_node(&["Person"]); + let berlin = store.create_node(&["City"]); + store.create_edge(alix, gus, "KNOWS"); + store.create_edge(gus, berlin, "LIVES_IN"); + + let scan = Box::new(ScanOperator::with_label(store.clone(), "Person")); + let step = |edge_type: &str, source_column| ExpandStep { + source_column, + direction: Direction::Outgoing, + edge_types: vec![edge_type.to_string()], + }; + let mut chain = LazyFactorizedChainOperator::new( + store.clone(), + scan, + vec![step("KNOWS", 0), step("LIVES_IN", 1)], + ); + let rows = |chain: &mut LazyFactorizedChainOperator| { + let mut rows = 0; + while let Some(chunk) = chain.next().unwrap() { + rows += chunk.row_count(); + } + rows + }; + + assert_eq!(rows(&mut chain), 1); + chain.reset(); + assert_eq!(rows(&mut chain), 1); + } + #[test] fn test_lazy_factorized_chain_into_any() { let store = Arc::new(LpgStore::new().unwrap()); diff --git a/crates/grafeo-core/src/execution/operators/filter.rs b/crates/grafeo-core/src/execution/operators/filter.rs index 39db975b5..b7db103eb 100644 --- a/crates/grafeo-core/src/execution/operators/filter.rs +++ b/crates/grafeo-core/src/execution/operators/filter.rs @@ -6,7 +6,8 @@ use crate::graph::Direction; use crate::graph::GraphStoreSearch; use crate::graph::lpg::{Edge, Node}; use grafeo_common::types::{ - EdgeId, EpochId, HashableValue, LogicalType, NodeId, PropertyKey, TransactionId, Value, + EdgeId, EpochId, HashableValue, LogicalType, NodeId, PropertyKey, PropertyMap, TransactionId, + Value, }; #[cfg(feature = "regex")] use regex::Regex; @@ -415,30 +416,45 @@ pub enum FilterExpression { /// The predicate to test for each element. predicate: Box, }, - /// EXISTS subquery: evaluates inner plan and returns true if results exist. + /// EXISTS subquery over one edge, or one variable-length edge, between + /// `start_var` and `end_var` (fast path): whether the pattern matches for + /// the row. A pattern variable the row binds must match what the row + /// holds (a null matches nothing); see [`ExpressionPredicate`] for a row + /// that binds the end but not the start, or neither. ExistsSubquery { - /// The start node variable from outer query. + /// The pattern's start node variable. start_var: String, - /// Direction of edge traversal. + /// The pattern's other end. + end_var: String, + /// The pattern's edge variable, if it has one. + edge_var: Option, + /// Direction of edge traversal, from the start. direction: Direction, /// Edge type filter (empty = match all types, multiple = match any). edge_types: Vec, - /// Optional end node labels filter. + /// Labels the other end must have (single-edge patterns only). end_labels: Option>, - /// Minimum number of hops (for variable-length patterns). + /// `Some(1)` for a variable-length pattern, `None` for one edge. min_hops: Option, - /// Maximum number of hops (for variable-length patterns). + /// Maximum number of hops of a variable-length pattern (`None`: + /// unbounded). max_hops: Option, }, - /// COUNT subquery: counts matching edges from a node (fast path). + /// COUNT subquery over one edge between `start_var` and `end_var` (fast + /// path): how many edges match for the row, with the same rules as + /// [`ExistsSubquery`](Self::ExistsSubquery). CountSubquery { - /// The start node variable from outer query. + /// The pattern's start node variable. start_var: String, - /// Direction of edge traversal. + /// The pattern's other end. + end_var: String, + /// The pattern's edge variable, if it has one. + edge_var: Option, + /// Direction of edge traversal, from the start. direction: Direction, /// Edge type filter (empty = match all types, multiple = match any). edge_types: Vec, - /// Optional end node labels filter. + /// Labels the other end must have. end_labels: Option>, }, /// reduce() accumulator: `reduce(acc = init, x IN list | expr)`. @@ -590,31 +606,305 @@ impl ExpressionPredicate { edge_types: &[String], end_labels: &Option>, ) -> bool { - // Check edge type if specified - if !edge_types.is_empty() { - let type_ok = if let Some(actual_type) = self.store.edge_type(edge_id) { - edge_types - .iter() - .any(|t| actual_type.as_str().eq_ignore_ascii_case(t.as_str())) - } else { - false - }; - if !type_ok { - return false; + self.edge_type_matches(edge_id, edge_types) && self.has_labels(other_node_id, end_labels) + } + + /// Whether the edge has one of `edge_types` (any type when empty). The + /// type is read as this query sees the edge, so a transaction finds the + /// type of an edge it created. + fn edge_type_matches(&self, edge_id: EdgeId, edge_types: &[String]) -> bool { + if edge_types.is_empty() { + return true; + } + let actual = if let (Some(ep), Some(tx)) = (self.viewing_epoch, self.transaction_id) { + self.store.edge_type_versioned(edge_id, ep, tx) + } else { + self.store.edge_type(edge_id) + }; + actual.is_some_and(|actual| { + edge_types + .iter() + .any(|t| actual.as_str().eq_ignore_ascii_case(t.as_str())) + }) + } + + /// The edges of `node` in `direction` that this query sees, with the + /// node at their other end: not one created after the viewing epoch, by + /// another transaction that has not committed, or deleted by this one + /// (the checks the expand operators make). + fn visible_edges_from(&self, node: NodeId, direction: Direction) -> Vec<(NodeId, EdgeId)> { + let mut edges = self.store.edges_from(node, direction); + if let Some(epoch) = self.viewing_epoch { + edges.retain(|&(other, edge)| { + if let Some(tx) = self.transaction_id { + self.store.is_edge_visible_versioned(edge, epoch, tx) + && self.store.is_node_visible_versioned(other, epoch, tx) + } else { + self.store.is_edge_visible_at_epoch(edge, epoch) + && self.store.is_node_visible_at_epoch(other, epoch) + } + }); + } + edges + } + + /// Whether the node has every label of `labels` (any node when `None`). + fn has_labels(&self, node_id: NodeId, labels: &Option>) -> bool { + labels.as_ref().is_none_or(|labels| { + self.resolve_node(node_id) + .is_some_and(|node| labels.iter().all(|label| node.has_label(label))) + }) + } + + /// What the row binds the pattern variable `name` to, read with `read`. + fn bound( + &self, + name: &str, + chunk: &DataChunk, + row: usize, + read: impl FnOnce(&ValueVector, usize) -> Option, + ) -> Bound { + match self + .variable_columns + .get(name) + .and_then(|&index| chunk.column(index)) + { + None => Bound::No, + Some(column) => read(column, row).map_or(Bound::Null, Bound::To), + } + } + + /// The nodes this query sees. + fn visible_nodes(&self) -> impl Iterator + '_ { + self.store + .node_ids() + .into_iter() + .filter(|&id| self.resolve_node(id).is_some()) + } + + /// How many matches a fast-path `EXISTS` or `COUNT` pattern has for the + /// row, counting up to `limit` (`EXISTS` needs one). + /// + /// The subquery pattern shares a variable with the row when the row binds + /// it: an end or edge the row binds must be that node or edge, and one the + /// row holds as null matches nothing. A row that binds the edge is + /// answered from that edge; one that binds the end but not the start has + /// its edges found from the end; one that binds none of them, which makes + /// the subquery the same for every row, from every node. + fn subquery_matches( + &self, + pattern: &SubqueryPattern<'_>, + chunk: &DataChunk, + row: usize, + limit: usize, + ) -> usize { + let (start, end) = match ( + self.bound(pattern.start, chunk, row, ValueVector::get_node_id), + self.bound(pattern.end, chunk, row, ValueVector::get_node_id), + ) { + (Bound::Null, _) | (_, Bound::Null) => return 0, + (start, end) => (start.node(), end.node()), + }; + if let Hops::Path(max_hops) = pattern.hops { + return usize::from(self.path_exists(pattern, chunk, row, start, end, max_hops)); + } + match pattern.edge.map_or(Bound::No, |name| { + self.bound(name, chunk, row, ValueVector::get_edge_id) + }) { + Bound::Null => return 0, + Bound::To(edge) => { + return self + .bound_edge_matches(edge, start, end, pattern) + .min(limit); + } + Bound::No => {} + } + match (start, end) { + (Some(start), end) => self + .visible_edges_from(start, pattern.direction) + .into_iter() + .filter(|&(other, id)| { + end.is_none_or(|end| end == other) + && self.edge_matches(other, id, pattern.edge_types, pattern.end_labels) + }) + .take(limit) + .count(), + (None, Some(end)) => { + if !self.has_labels(end, pattern.end_labels) { + return 0; + } + self.visible_edges_from(end, pattern.direction.reverse()) + .into_iter() + .filter(|&(_, id)| self.edge_type_matches(id, pattern.edge_types)) + .take(limit) + .count() + } + (None, None) => { + let mut found = 0; + for start in self.visible_nodes() { + found += self + .visible_edges_from(start, pattern.direction) + .into_iter() + .filter(|&(other, id)| { + self.edge_matches(other, id, pattern.edge_types, pattern.end_labels) + }) + .take(limit - found) + .count(); + if found == limit { + break; + } + } + found } } + } - // Check end node labels if specified (e.g., (:Person)-[:KNOWS]->(n) requires - // the other endpoint to have the Person label after direction flipping). - if let Some(labels) = end_labels { - if let Some(node) = self.resolve_node(other_node_id) { - labels.iter().all(|l| node.has_label(l)) - } else { - false + /// How often the row's edge matches a one-edge pattern between `start` + /// and `end` (the nodes the row binds, if any): once, or once each way + /// for an undirected pattern with free ends, as walking the edges of + /// every node would find it (a self-loop too). + fn bound_edge_matches( + &self, + edge: EdgeId, + start: Option, + end: Option, + pattern: &SubqueryPattern<'_>, + ) -> usize { + let Some(record) = self.resolve_edge(edge) else { + return 0; + }; + if !self.edge_type_matches(edge, pattern.edge_types) { + return 0; + } + // The edge read from the start: forward from its source, backward + // from its target. + let forward = matches!(pattern.direction, Direction::Outgoing | Direction::Both) + .then_some((record.src, record.dst)); + let backward = matches!(pattern.direction, Direction::Incoming | Direction::Both) + .then_some((record.dst, record.src)); + [forward, backward] + .into_iter() + .flatten() + .filter(|&(from, to)| { + start.map_or_else(|| self.resolve_node(from).is_some(), |start| start == from) + && end.is_none_or(|end| end == to) + && self.has_labels(to, pattern.end_labels) + }) + .count() + } + + /// Whether a variable-length fast-path pattern (1 to `max_hops` edges, + /// any number when `None`) matches for the row; `start` and `end` are the + /// nodes the row binds. A path from an unbound end exists when its first + /// edge does. + fn path_exists( + &self, + pattern: &SubqueryPattern<'_>, + chunk: &DataChunk, + row: usize, + start: Option, + end: Option, + max_hops: Option, + ) -> bool { + if let Some(name) = pattern.edge { + match self.bound(name, chunk, row, edge_id_list) { + Bound::Null => return false, + Bound::To(edges) => { + return self.path_follows(&edges, start, end, pattern, max_hops); + } + Bound::No => {} } - } else { - true } + let has_edge = |node: NodeId, direction: Direction| { + self.visible_edges_from(node, direction) + .into_iter() + .any(|(_, id)| self.edge_type_matches(id, pattern.edge_types)) + }; + match (start, end) { + (Some(start), Some(end)) => self.reaches(start, end, pattern, max_hops), + (Some(start), None) => has_edge(start, pattern.direction), + (None, Some(end)) => has_edge(end, pattern.direction.reverse()), + (None, None) => self + .visible_nodes() + .any(|node| has_edge(node, pattern.direction)), + } + } + + /// Whether `to` is reached from `from` over 1 to `max_hops` edges of the + /// pattern (any number when `None`). + fn reaches( + &self, + from: NodeId, + to: NodeId, + pattern: &SubqueryPattern<'_>, + max_hops: Option, + ) -> bool { + // `from` is not marked seen, so a cycle back to it counts. + let mut seen = std::collections::HashSet::new(); + let mut frontier = vec![from]; + let mut hops = 0; + while !frontier.is_empty() && max_hops.is_none_or(|max| hops < max) { + hops += 1; + let mut next = Vec::new(); + for node in frontier { + for (other, id) in self.visible_edges_from(node, pattern.direction) { + if !self.edge_type_matches(id, pattern.edge_types) { + continue; + } + if other == to { + return true; + } + if seen.insert(other) { + next.push(other); + } + } + } + frontier = next; + } + false + } + + /// Whether `edges` are a path of the pattern, one after the other: 1 to + /// `max_hops` edges of its types in its direction, from `start` and to + /// `end` when the row binds them. + fn path_follows( + &self, + edges: &[EdgeId], + start: Option, + end: Option, + pattern: &SubqueryPattern<'_>, + max_hops: Option, + ) -> bool { + if max_hops.is_some_and(|max| usize::try_from(max).is_ok_and(|max| edges.len() > max)) { + return false; + } + let Some(first) = edges.first().and_then(|&id| self.resolve_edge(id)) else { + return false; + }; + let starts = match (start, pattern.direction) { + (Some(start), _) => vec![start], + (None, Direction::Outgoing) => vec![first.src], + (None, Direction::Incoming) => vec![first.dst], + (None, Direction::Both) => vec![first.src, first.dst], + }; + starts.into_iter().any(|mut node| { + for &id in edges { + let Some(edge) = self.resolve_edge(id) else { + return false; + }; + if !self.edge_type_matches(id, pattern.edge_types) { + return false; + } + node = match pattern.direction { + Direction::Outgoing if edge.src == node => edge.dst, + Direction::Incoming if edge.dst == node => edge.src, + Direction::Both if edge.src == node => edge.dst, + Direction::Both if edge.dst == node => edge.src, + _ => return false, + }; + } + end.is_none_or(|end| end == node) + }) } /// Resolves an edge using transaction-aware access when available. @@ -778,6 +1068,21 @@ impl ExpressionPredicate { return edge.get_property(key.as_str()).cloned(); } } + // One item of a node or edge list (`rs[0].w`, + // `head(rs).w`) is the ID of a node or edge. + if let Value::Int64(id) = &base_val + && let Ok(id) = u64::try_from(*id) + { + return match self.element_kind(base, chunk) { + ItemKind::Node => self + .resolve_node(NodeId::new(id)) + .and_then(|node| node.get_property(key.as_str()).cloned()), + ItemKind::Edge => self + .resolve_edge(EdgeId::new(id)) + .and_then(|edge| edge.get_property(key.as_str()).cloned()), + ItemKind::Value => None, + }; + } None } _ => None, @@ -950,51 +1255,50 @@ impl ExpressionPredicate { } FilterExpression::ExistsSubquery { start_var, + end_var, + edge_var, direction, edge_types, end_labels, - // min_hops/max_hops are always None from the fast path - // (extract_exists_pattern rejects multi-hop patterns). - .. + min_hops, + max_hops, } => { - // Get the start node ID from the current row - let col_idx = *self.variable_columns.get(start_var)?; - let col = chunk.column(col_idx)?; - let start_node_id = col.get_node_id(row)?; - - // Check if any matching edges exist - let exists = self - .store - .edges_from(start_node_id, *direction) - .into_iter() - .any(|(other_node_id, edge_id)| { - self.edge_matches(other_node_id, edge_id, edge_types, end_labels) - }); - - Some(Value::Bool(exists)) + let pattern = SubqueryPattern { + start: start_var, + end: end_var, + edge: edge_var.as_deref(), + direction: *direction, + edge_types, + end_labels, + hops: if min_hops.is_some() { + Hops::Path(*max_hops) + } else { + Hops::One + }, + }; + Some(Value::Bool( + self.subquery_matches(&pattern, chunk, row, 1) > 0, + )) } FilterExpression::CountSubquery { start_var, + end_var, + edge_var, direction, edge_types, end_labels, } => { - let col_idx = *self.variable_columns.get(start_var)?; - let col = chunk.column(col_idx)?; - let start_node_id = col.get_node_id(row)?; - - let count = self - .store - .edges_from(start_node_id, *direction) - .into_iter() - .filter(|(other_node_id, edge_id)| { - self.edge_matches(*other_node_id, *edge_id, edge_types, end_labels) - }) - .count(); - - // reason: edge count from a single node fits i64 - #[allow(clippy::cast_possible_wrap)] - Some(Value::Int64(count as i64)) + let pattern = SubqueryPattern { + start: start_var, + end: end_var, + edge: edge_var.as_deref(), + direction: *direction, + edge_types, + end_labels, + hops: Hops::One, + }; + let count = self.subquery_matches(&pattern, chunk, row, usize::MAX); + Some(Value::Int64(i64::try_from(count).unwrap_or(i64::MAX))) } FilterExpression::Reduce { accumulator, @@ -1033,6 +1337,42 @@ impl ExpressionPredicate { } } + /// The properties of the node or edge the variable's column holds at + /// `row`, taken from the entity the store resolves (no copy). An edge + /// column holds edges and a node column nodes; an untyped column holding + /// raw IDs holds a node when one has the ID, otherwise an edge. `None` for + /// a value that is neither. + fn element_properties( + &self, + variable: &str, + chunk: &DataChunk, + row: usize, + ) -> Option { + let column = chunk.column(*self.variable_columns.get(variable)?)?; + let node = || Some(self.resolve_node(column.get_node_id(row)?)?.properties); + let edge = || Some(self.resolve_edge(column.get_edge_id(row)?)?.properties); + match column.data_type() { + LogicalType::Edge => edge(), + LogicalType::Node => node(), + _ => node().or_else(edge), + } + } + + /// What one item taken from a list refers to: `list[i]`, `head(list)` + /// and `last(list)` are nodes or edges when the items of `list` are. + fn element_kind(&self, expr: &FilterExpression, chunk: &DataChunk) -> ItemKind { + match expr { + FilterExpression::IndexAccess { base, .. } => self.item_kind(base, chunk), + FilterExpression::FunctionCall { name, args, .. } + if name.eq_ignore_ascii_case("head") || name.eq_ignore_ascii_case("last") => + { + args.first() + .map_or(ItemKind::Value, |list| self.item_kind(list, chunk)) + } + _ => ItemKind::Value, + } + } + /// What the items of `list_expr` refer to: edges for `edges(p)`, /// `relationships(p)` and the variable of a variable-length edge pattern, /// nodes for `nodes(p)`, also after `reverse`, `tail` or a slice. @@ -1831,19 +2171,18 @@ impl ExpressionPredicate { if args.len() != 1 { return None; } - // keys(n) on a node variable: get property keys from the store - if let FilterExpression::Variable(var) = &args[0] { - let col_idx = *self.variable_columns.get(var)?; - let col = chunk.column(col_idx)?; - if let Some(node_id) = col.get_node_id(row) { - let node = self.resolve_node(node_id)?; - let keys: Vec = node - .properties - .iter() - .map(|(k, _)| Value::String(k.as_str().into())) - .collect(); - return Some(Value::List(keys.into())); - } + // keys(n) or keys(r) on a node or edge variable: the property + // keys from the store, sorted (as `properties` and map keys + // are), so their order does not depend on how they are stored + if let FilterExpression::Variable(var) = &args[0] + && let Some(properties) = self.element_properties(var, chunk, row) + { + let mut keys: Vec = properties + .into_iter() + .map(|(k, _)| Value::String(k.as_str().into())) + .collect(); + keys.sort_by(|a, b| a.as_str().cmp(&b.as_str())); + return Some(Value::List(keys.into())); } // keys(map) on a map value let val = self.eval_expr(&args[0], chunk, row)?; @@ -1863,25 +2202,11 @@ impl ExpressionPredicate { return None; } if let FilterExpression::Variable(var) = &args[0] { - let col_idx = *self.variable_columns.get(var)?; - let col = chunk.column(col_idx)?; - if let Some(node_id) = col.get_node_id(row) { - let node = self.resolve_node(node_id)?; - let map: std::collections::BTreeMap = node - .properties - .iter() - .map(|(k, v)| (k.clone(), v.clone())) - .collect(); - return Some(Value::Map(Arc::new(map))); - } else if let Some(edge_id) = col.get_edge_id(row) { - let edge = self.resolve_edge(edge_id)?; - let map: std::collections::BTreeMap = edge - .properties - .iter() - .map(|(k, v)| (k.clone(), v.clone())) - .collect(); - return Some(Value::Map(Arc::new(map))); - } + let map: std::collections::BTreeMap = self + .element_properties(var, chunk, row)? + .into_iter() + .collect(); + return Some(Value::Map(Arc::new(map))); } None } @@ -1891,20 +2216,16 @@ impl ExpressionPredicate { if args.len() != 1 { return None; } + // In key order, so they line up with keys(n) (a store keeps + // properties in no particular order). if let FilterExpression::Variable(var) = &args[0] { - let col_idx = *self.variable_columns.get(var)?; - let col = chunk.column(col_idx)?; - if let Some(node_id) = col.get_node_id(row) { - let node = self.resolve_node(node_id)?; - let vals: Vec = - node.properties.iter().map(|(_, v)| v.clone()).collect(); - return Some(Value::List(vals.into())); - } else if let Some(edge_id) = col.get_edge_id(row) { - let edge = self.resolve_edge(edge_id)?; - let vals: Vec = - edge.properties.iter().map(|(_, v)| v.clone()).collect(); - return Some(Value::List(vals.into())); - } + let mut properties: Vec<(PropertyKey, Value)> = self + .element_properties(var, chunk, row)? + .into_iter() + .collect(); + properties.sort_by(|(a, _), (b, _)| a.as_str().cmp(b.as_str())); + let values: Vec = properties.into_iter().map(|(_, v)| v).collect(); + return Some(Value::List(values.into())); } None } @@ -3568,6 +3889,63 @@ impl ExpressionPredicate { } } +/// A fast-path `EXISTS` or `COUNT` pattern (see +/// [`FilterExpression::ExistsSubquery`]). +struct SubqueryPattern<'a> { + start: &'a str, + end: &'a str, + edge: Option<&'a str>, + direction: Direction, + edge_types: &'a [String], + end_labels: &'a Option>, + hops: Hops, +} + +/// How many edges a subquery pattern has. +#[derive(Clone, Copy)] +enum Hops { + /// One edge. + One, + /// A variable-length edge: 1 to this many (`None`: any number). + Path(Option), +} + +/// What a row binds a variable of a subquery pattern to. +enum Bound { + /// The row has no such variable: the pattern binds it. + No, + /// The row holds null there, or not a node or edge: nothing matches it. + Null, + /// The row holds this node or edge. + To(T), +} + +impl Bound { + /// The node, when the row binds one. + fn node(self) -> Option { + match self { + Self::To(node) => Some(node), + Self::No | Self::Null => None, + } + } +} + +/// The edges of a path's edge list in `column`, as a variable-length expand +/// writes it. +fn edge_id_list(column: &ValueVector, row: usize) -> Option> { + let Value::List(items) = column.get_value(row)? else { + return None; + }; + items + .iter() + .map(|item| { + item.as_int64() + .and_then(|id| u64::try_from(id).ok()) + .map(EdgeId::new) + }) + .collect() +} + /// What the items of a list a comprehension, list predicate or `reduce` /// iterates over refer to. `edges(p)`, `relationships(p)` and `nodes(p)` /// hold entity ids; an item bound as an edge or a node is read like an edge @@ -6496,6 +6874,105 @@ mod text_fn_tests { ); } + /// `keys()`, `properties()` and `property_values()` read the entity the + /// column holds: an edge column reads the edge even when a node has the + /// same ID, and an untyped column holding a raw ID reads the edge when no + /// node has that ID (it used to stop at the missing node). + /// `property_values(n)` lists the values in key order, as `keys(n)` lists + /// the keys. + #[test] + fn property_values_line_up_with_keys() { + let store = Arc::new(LpgStore::new().unwrap()); + let alix = store.create_node(&["Person"]); + // Set in an order other than the key order. + store.set_node_property(alix, "name", Value::from("Alix")); + store.set_node_property(alix, "age", Value::Int64(30)); + let eval = |function: &str| { + let mut column = ValueVector::with_capacity(LogicalType::Node, 1); + column.push_node_id(alix); + ExpressionPredicate::new( + FilterExpression::FunctionCall { + name: function.to_string(), + args: vec![FilterExpression::Variable("n".to_string())], + }, + HashMap::from([("n".to_string(), 0)]), + Arc::clone(&store) as Arc, + ) + .eval(&DataChunk::new(vec![column]), 0) + }; + assert_eq!( + eval("keys"), + Some(Value::List( + vec![Value::from("age"), Value::from("name")].into() + )) + ); + assert_eq!( + eval("property_values"), + Some(Value::List( + vec![Value::Int64(30), Value::from("Alix")].into() + )) + ); + } + + #[test] + fn element_functions_read_the_entity_the_column_holds() { + let store = Arc::new(LpgStore::new().unwrap()); + let alix = store.create_node(&["Person"]); + store.set_node_property(alix, "name", Value::from("Alix")); + let gus = store.create_node(&["Person"]); + // Edge 0 has the ID of node 0 (Alix); edge 2 has no node of its ID. + let first = store.create_edge(alix, gus, "KNOWS"); + store.set_edge_property(first, "since", Value::Int64(2010)); + store.create_edge(gus, alix, "KNOWS"); + let third = store.create_edge(alix, alix, "KNOWS"); + store.set_edge_property(third, "w", Value::Int64(3)); + + let eval = |column: ValueVector, function: &str| { + let predicate = ExpressionPredicate::new( + FilterExpression::FunctionCall { + name: function.to_string(), + args: vec![FilterExpression::Variable("r".to_string())], + }, + HashMap::from([("r".to_string(), 0)]), + Arc::clone(&store) as Arc, + ); + predicate.eval(&DataChunk::new(vec![column]), 0) + }; + let keys = |names: &[&str]| { + Some(Value::List( + names + .iter() + .map(|name| Value::from(*name)) + .collect::>() + .into(), + )) + }; + + let mut typed = ValueVector::with_capacity(LogicalType::Edge, 1); + typed.push_edge_id(first); + assert_eq!(eval(typed.clone(), "keys"), keys(&["since"])); + assert_eq!( + eval(typed.clone(), "property_values"), + Some(Value::List(vec![Value::Int64(2010)].into())) + ); + let Some(Value::Map(map)) = eval(typed, "properties") else { + panic!("expected a map"); + }; + assert_eq!( + map.get(&PropertyKey::new("since")), + Some(&Value::Int64(2010)) + ); + assert_eq!(map.len(), 1); + + let mut untyped = ValueVector::with_capacity(LogicalType::Any, 1); + untyped.push_value(Value::Int64(i64::try_from(third.as_u64()).unwrap())); + assert_eq!(eval(untyped, "keys"), keys(&["w"])); + + let mut node = ValueVector::with_capacity(LogicalType::Node, 1); + node.push_node_id(alix); + assert_eq!(eval(node, "keys"), keys(&["name"])); + } + #[test] fn test_text_match_function() { let (store, n1, n2) = setup_store_with_text_index(); diff --git a/crates/grafeo-core/src/execution/operators/horizontal_aggregate.rs b/crates/grafeo-core/src/execution/operators/horizontal_aggregate.rs index dcabbcee1..151eff811 100644 --- a/crates/grafeo-core/src/execution/operators/horizontal_aggregate.rs +++ b/crates/grafeo-core/src/execution/operators/horizontal_aggregate.rs @@ -104,9 +104,15 @@ impl Operator for HorizontalAggregateOperator { return Ok(None); }; - // Build output columns: copy input columns + one new aggregate result column + // Build output columns: copy input columns, in their types (a node or + // edge stays one), + one new aggregate result column let mut output_columns: Vec = (0..self.input_column_count) - .map(|_| ValueVector::with_capacity(LogicalType::Any, input.row_count())) + .map(|col_idx| { + let column_type = input + .column(col_idx) + .map_or(LogicalType::Any, |column| column.data_type().clone()); + ValueVector::with_capacity(column_type, input.row_count()) + }) .collect(); let mut result_column = ValueVector::with_capacity(LogicalType::Float64, input.row_count()); @@ -245,6 +251,36 @@ mod tests { (Arc::new(store), node_ids) } + /// The copied input columns keep their types: a node stays a node. + #[test] + fn copied_columns_keep_their_types() { + let (store, edge_ids) = setup_store_with_edges(); + let mut builder = DataChunkBuilder::new(&[LogicalType::Node, LogicalType::Any]); + builder.column_mut(0).unwrap().push_node_id(NodeId::new(5)); + builder + .column_mut(1) + .unwrap() + .push_value(Value::List(edge_ids.into())); + builder.advance_row(); + let mut op = HorizontalAggregateOperator::new( + Box::new(MockOperator::new(vec![builder.finish()])), + 1, + EntityKind::Edge, + AggregateFunction::Sum, + "weight".to_string(), + store, + 2, + ); + + let result = op.next().unwrap().unwrap(); + assert_eq!(result.column_types()[0], LogicalType::Node); + assert_eq!( + result.column(0).unwrap().get_node_id(0).unwrap().as_u64(), + 5 + ); + assert!(result.column(0).unwrap().get_edge_id(0).is_none()); + } + #[test] fn test_horizontal_sum_over_edges() { let (store, edge_ids) = setup_store_with_edges(); diff --git a/crates/grafeo-core/src/execution/operators/join.rs b/crates/grafeo-core/src/execution/operators/join.rs index 50ed6fa38..126423fb1 100644 --- a/crates/grafeo-core/src/execution/operators/join.rs +++ b/crates/grafeo-core/src/execution/operators/join.rs @@ -5,13 +5,13 @@ //! - `NestedLoopJoinOperator`: General-purpose join for any condition use std::cmp::Ordering; -use std::collections::HashMap; +use std::collections::{HashMap, VecDeque}; use arcstr::ArcStr; use grafeo_common::types::{LogicalType, Value}; use super::{Operator, OperatorError, OperatorResult}; -use crate::execution::chunk::DataChunkBuilder; +use crate::execution::chunk::{ColumnTypes, DataChunkBuilder, copied_column_types}; use crate::execution::{DataChunk, ValueVector}; /// The type of join to perform. @@ -167,7 +167,8 @@ impl HashKey { /// Hash join operator. /// /// Builds a hash table from the build side (right) and probes with the probe side (left). -/// Efficient for equality joins on one or more columns. +/// Efficient for equality joins on one or more columns. A row keeps the column +/// types of the rows it joins: a node or edge stays one. pub struct HashJoinOperator { /// Left (probe) side operator. probe_side: Box, @@ -179,8 +180,14 @@ pub struct HashJoinOperator { build_keys: Vec, /// Join type. join_type: JoinType, - /// Output schema (combined from both sides). + /// Output schema (combined from both sides). Used only for a side that has + /// no rows: the build side of a left join that matched nothing, the probe + /// side of the unmatched build rows of a right join. output_schema: Vec, + /// The column types of the probe chunks read so far. + probe_types: ColumnTypes, + /// The column types of the materialized build side. + build_types: ColumnTypes, /// Hash table: key -> list of (chunk_index, row_index). hash_table: HashMap>, /// Materialized build side chunks. @@ -199,6 +206,12 @@ pub struct HashJoinOperator { probe_matched: Vec, /// For right/full outer joins: track which build rows were matched. build_matched: Vec>, + /// A condition on each pair of rows the keys match, beyond the keys (the + /// WHERE of an OPTIONAL MATCH that reads both sides): a pair that fails it + /// is no match, and a left row without one keeps nulls. + residual: Option>, + /// Whether a pair of the current probe row passed the residual condition. + current_probe_kept: bool, /// Whether we're in the emit unmatched phase (for outer joins). emitting_unmatched: bool, /// Current chunk index when emitting unmatched rows. @@ -232,6 +245,8 @@ impl HashJoinOperator { build_keys, join_type, output_schema, + probe_types: ColumnTypes::default(), + build_types: ColumnTypes::default(), hash_table: HashMap::new(), build_chunks: Vec::new(), build_complete: false, @@ -241,12 +256,23 @@ impl HashJoinOperator { current_matches: Vec::new(), probe_matched: Vec::new(), build_matched: Vec::new(), + residual: None, + current_probe_kept: false, emitting_unmatched: false, unmatched_chunk_idx: 0, unmatched_row_idx: 0, } } + /// Adds a condition each pair of rows the keys match must also meet: a + /// pair that fails it is no match (so in a left join, a probe row none of + /// whose pairs pass it keeps nulls). + #[must_use] + pub fn with_residual(mut self, condition: Box) -> Self { + self.residual = Some(condition); + self + } + /// Builds the hash table from the build side. fn build_hash_table(&mut self) -> Result<(), OperatorError> { while let Some(chunk) = self.build_side.next()? { @@ -277,6 +303,7 @@ impl HashJoinOperator { .push((chunk_idx, row)); } + self.build_types.add(&chunk); self.build_chunks.push(chunk); } @@ -284,6 +311,44 @@ impl HashJoinOperator { Ok(()) } + /// The column types of the rows joined from `probe_chunk`: its own, then + /// the build side's (only its own for a semi- or anti-join), so the copied + /// values keep their types (see [`ColumnTypes`]). The declared schema gives + /// the build side's types while it has no rows. + fn output_types(&self, probe_chunk: &DataChunk) -> Vec { + let mut types = probe_chunk.column_types(); + if matches!(self.join_type, JoinType::Semi | JoinType::Anti) { + return copied_column_types(&types, &self.output_schema); + } + if self.build_chunks.is_empty() { + types.extend( + self.output_schema + .iter() + .skip(probe_chunk.column_count()) + .cloned(), + ); + } else { + types.extend_from_slice(self.build_types.types()); + } + copied_column_types(&types, &self.output_schema) + } + + /// The column types of the unmatched build rows of a right or full join: + /// the probe side's (declared while it had no rows), then the build side's. + fn unmatched_build_types(&self, probe_col_count: usize) -> Vec { + let mut types = if self.probe_types.types().is_empty() { + self.output_schema + .iter() + .take(probe_col_count) + .cloned() + .collect() + } else { + self.probe_types.types().to_vec() + }; + types.extend_from_slice(self.build_types.types()); + copied_column_types(&types, &self.output_schema) + } + /// Extracts a hash key from a chunk row. fn extract_key( &self, @@ -362,21 +427,23 @@ impl HashJoinOperator { } } _ => { - // Emit nulls for build side (left outer join case) - if !self.build_chunks.is_empty() { - let build_col_count = self.build_chunks[0].column_count(); - for col_idx in 0..build_col_count { - let dst_col = - builder - .column_mut(probe_col_count + col_idx) - .ok_or_else(|| { - OperatorError::ColumnNotFound(format!( - "output column {}", - probe_col_count + col_idx - )) - })?; - dst_col.push_value(Value::Null); - } + // Emit nulls for build side (left outer join case), in every + // build column: the declared ones while the build side is empty + let build_col_count = self.build_chunks.first().map_or_else( + || self.output_schema.len().saturating_sub(probe_col_count), + DataChunk::column_count, + ); + for col_idx in 0..build_col_count { + let dst_col = + builder + .column_mut(probe_col_count + col_idx) + .ok_or_else(|| { + OperatorError::ColumnNotFound(format!( + "output column {}", + probe_col_count + col_idx + )) + })?; + dst_col.push_value(Value::Null); } } } @@ -393,6 +460,7 @@ impl HashJoinOperator { if matches!(self.join_type, JoinType::Left | JoinType::Full) { self.probe_matched = vec![false; c.row_count()]; } + self.probe_types.add(c); } let has_chunk = chunk.is_some(); self.current_probe_chunk = chunk; @@ -406,14 +474,14 @@ impl HashJoinOperator { return Ok(None); } - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, 2048); - // Determine probe column count from schema or first probe chunk let probe_col_count = if !self.build_chunks.is_empty() { self.output_schema.len() - self.build_chunks[0].column_count() } else { 0 }; + let mut builder = + DataChunkBuilder::with_capacity(&self.unmatched_build_types(probe_col_count), 2048); while self.unmatched_chunk_idx < self.build_chunks.len() { let chunk = &self.build_chunks[self.unmatched_chunk_idx]; @@ -479,8 +547,9 @@ impl Operator for HashJoinOperator { return self.emit_unmatched_build(); } - // Phase 2: Probe - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, 2048); + // Phase 2: Probe. Each returned chunk holds rows of one probe chunk, + // built in its column types. + let mut chunk_builder: Option = None; loop { // Get current probe chunk or fetch new one @@ -490,11 +559,8 @@ impl Operator for HashJoinOperator { self.emitting_unmatched = true; return self.emit_unmatched_build(); } - return if builder.row_count() > 0 { - Ok(Some(builder.finish())) - } else { - Ok(None) - }; + // A probe chunk's rows are returned when the chunk ends. + return Ok(None); } // Invariant: current_probe_chunk is Some here - the guard at line 396 either @@ -503,6 +569,9 @@ impl Operator for HashJoinOperator { .current_probe_chunk .as_ref() .expect("probe chunk is Some: guard at line 396 ensures this"); + let builder = chunk_builder.get_or_insert_with(|| { + DataChunkBuilder::with_capacity(&self.output_types(probe_chunk), 2048) + }); let probe_rows: Vec = probe_chunk.selected_indices().collect(); while self.current_probe_row < probe_rows.len() { @@ -512,10 +581,25 @@ impl Operator for HashJoinOperator { if self.current_matches.is_empty() && self.current_match_position == 0 { let key = self.extract_key(probe_chunk, probe_row, &self.probe_keys)?; - // Handle semi/anti joins differently + // Handle semi/anti joins differently: a probe row has a match + // when a pair with the same key passes the residual condition. + let has_match = || { + self.hash_table.get(&key).is_some_and(|candidates| { + self.residual.as_ref().is_none_or(|residual| { + candidates.iter().any(|&(chunk_idx, row)| { + residual.evaluate( + probe_chunk, + probe_row, + &self.build_chunks[chunk_idx], + row, + ) + }) + }) + }) + }; match self.join_type { JoinType::Semi => { - if self.hash_table.contains_key(&key) { + if has_match() { // Emit probe row only for col_idx in 0..probe_chunk.column_count() { if let (Some(src_col), Some(dst_col)) = @@ -531,7 +615,7 @@ impl Operator for HashJoinOperator { continue; } JoinType::Anti => { - if !self.hash_table.contains_key(&key) { + if !has_match() { // Emit probe row only for col_idx in 0..probe_chunk.column_count() { if let (Some(src_col), Some(dst_col)) = @@ -549,6 +633,7 @@ impl Operator for HashJoinOperator { _ => { self.current_matches = self.hash_table.get(&key).cloned().unwrap_or_default(); + self.current_probe_kept = false; } } } @@ -557,7 +642,7 @@ impl Operator for HashJoinOperator { if self.current_matches.is_empty() { // No matches - for left/full outer join, emit with nulls if matches!(self.join_type, JoinType::Left | JoinType::Full) { - self.produce_output_row(&mut builder, probe_chunk, probe_row, None, None)?; + self.produce_output_row(builder, probe_chunk, probe_row, None, None)?; } self.current_probe_row += 1; self.current_match_position = 0; @@ -568,6 +653,14 @@ impl Operator for HashJoinOperator { self.current_matches[self.current_match_position]; let build_chunk = &self.build_chunks[build_chunk_idx]; + if let Some(residual) = &self.residual + && !residual.evaluate(probe_chunk, probe_row, build_chunk, build_row) + { + self.current_match_position += 1; + continue; + } + self.current_probe_kept = true; + // Mark as matched for outer joins if matches!(self.join_type, JoinType::Left | JoinType::Full) && probe_row < self.probe_matched.len() @@ -582,7 +675,7 @@ impl Operator for HashJoinOperator { } self.produce_output_row( - &mut builder, + builder, probe_chunk, probe_row, Some(build_chunk), @@ -592,18 +685,25 @@ impl Operator for HashJoinOperator { self.current_match_position += 1; if builder.is_full() { - return Ok(Some(builder.finish())); + return Ok(chunk_builder.take().map(DataChunkBuilder::finish)); } } - // Done with this probe row + // Done with this probe row: without a pair that passed the + // residual condition, a left or full join keeps it with nulls. + if self.residual.is_some() + && !self.current_probe_kept + && matches!(self.join_type, JoinType::Left | JoinType::Full) + { + self.produce_output_row(builder, probe_chunk, probe_row, None, None)?; + } self.current_probe_row += 1; self.current_matches.clear(); self.current_match_position = 0; } if builder.is_full() { - return Ok(Some(builder.finish())); + return Ok(chunk_builder.take().map(DataChunkBuilder::finish)); } } @@ -611,8 +711,10 @@ impl Operator for HashJoinOperator { self.current_probe_chunk = None; self.current_probe_row = 0; - if builder.row_count() > 0 { - return Ok(Some(builder.finish())); + if let Some(done) = chunk_builder.take() + && done.row_count() > 0 + { + return Ok(Some(done.finish())); } } } @@ -620,6 +722,8 @@ impl Operator for HashJoinOperator { fn reset(&mut self) { self.probe_side.reset(); self.build_side.reset(); + self.probe_types = ColumnTypes::default(); + self.build_types = ColumnTypes::default(); self.hash_table.clear(); self.build_chunks.clear(); self.build_complete = false; @@ -629,6 +733,7 @@ impl Operator for HashJoinOperator { self.current_matches.clear(); self.probe_matched.clear(); self.build_matched.clear(); + self.current_probe_kept = false; self.emitting_unmatched = false; self.unmatched_chunk_idx = 0; self.unmatched_row_idx = 0; @@ -646,7 +751,8 @@ impl Operator for HashJoinOperator { /// Nested loop join operator. /// /// Performs a cartesian product of both sides, filtering by the join condition. -/// Less efficient than hash join but supports any join condition. +/// Less efficient than hash join but supports any join condition. A row keeps +/// the column types of the rows it joins: a node or edge stays one. pub struct NestedLoopJoinOperator { /// Left side operator. left: Box, @@ -656,12 +762,19 @@ pub struct NestedLoopJoinOperator { condition: Option>, /// Join type. join_type: JoinType, - /// Output schema. + /// Output schema. Only its right-side part is used: for the right side's + /// columns of a left join when the right side has no rows. output_schema: Vec, + /// The column types of the materialized right side. + right_types: ColumnTypes, /// Materialized right side chunks. right_chunks: Vec, /// Whether the right side is materialized. right_materialized: bool, + /// Whether the whole left side is read before the right side. + left_first: bool, + /// The left side's chunks, when it is read first. + left_chunks: Option>, /// Current left chunk. current_left_chunk: Option, /// Current row in the left chunk. @@ -686,6 +799,69 @@ pub trait JoinCondition: Send + Sync { ) -> bool; } +/// A condition given by a predicate over the joined row: the left row's +/// columns, then the right row's, numbered as the predicate's variable +/// columns number them. +pub struct JoinedRowCondition { + predicate: Box, + /// The joined row being checked, reused from pair to pair while the + /// column types stay the same. + scratch: parking_lot::Mutex>, +} + +impl JoinedRowCondition { + /// Creates the condition from a predicate over the joined row. + pub fn new(predicate: Box) -> Self { + Self { + predicate, + scratch: parking_lot::Mutex::new(None), + } + } +} + +impl JoinCondition for JoinedRowCondition { + fn evaluate( + &self, + left_chunk: &DataChunk, + left_row: usize, + right_chunk: &DataChunk, + right_row: usize, + ) -> bool { + let sources: Vec<(&ValueVector, usize)> = + [(left_chunk, left_row), (right_chunk, right_row)] + .into_iter() + .flat_map(|(chunk, row)| chunk.columns().iter().map(move |column| (column, row))) + .collect(); + let mut scratch = self.scratch.lock(); + let fits = scratch.as_ref().is_some_and(|joined| { + joined.column_count() == sources.len() + && joined + .columns() + .iter() + .zip(&sources) + .all(|(column, (source, _))| column.data_type() == source.data_type()) + }); + if !fits { + *scratch = Some(DataChunk::new( + sources + .iter() + .map(|(source, _)| ValueVector::with_capacity(source.data_type().clone(), 1)) + .collect(), + )); + } + let joined = scratch.as_mut().expect("the joined row was just built"); + for (index, (source, row)) in sources.iter().enumerate() { + let column = joined + .column_mut(index) + .expect("the joined row has a column per source column"); + column.clear(); + source.copy_row_to(*row, column); + } + joined.set_count(1); + self.predicate.evaluate(joined, 0) + } +} + /// A simple equality condition for nested loop joins. pub struct EqualityCondition { /// Column index on the left side. @@ -741,8 +917,11 @@ impl NestedLoopJoinOperator { condition, join_type, output_schema, + right_types: ColumnTypes::default(), right_chunks: Vec::new(), right_materialized: false, + left_first: false, + left_chunks: None, current_left_chunk: None, current_left_row: 0, current_right_chunk: 0, @@ -751,15 +930,52 @@ impl NestedLoopJoinOperator { } } + /// Reads the whole left side before the right side, so that the right + /// side sees what the left side wrote: in `INSERT (:N) WITH 1 AS x + /// MATCH (n:N)` the scan of `n` runs after the insert. + #[must_use] + pub fn with_left_first(mut self) -> Self { + self.left_first = true; + self + } + + /// The next left chunk, from the buffer when the left side was read first. + fn next_left(&mut self) -> OperatorResult { + match &mut self.left_chunks { + Some(chunks) => Ok(chunks.pop_front()), + None => self.left.next(), + } + } + /// Materializes the right side. fn materialize_right(&mut self) -> Result<(), OperatorError> { while let Some(chunk) = self.right.next()? { + self.right_types.add(&chunk); self.right_chunks.push(chunk); } self.right_materialized = true; Ok(()) } + /// The column types of the rows joined from `left_chunk`: its own, then + /// the right side's, so the copied values keep their types (see + /// [`ColumnTypes`]). The declared schema gives the right side's types + /// while it has no rows. + fn output_types(&self, left_chunk: &DataChunk) -> Vec { + let mut types = left_chunk.column_types(); + if self.right_chunks.is_empty() { + types.extend( + self.output_schema + .iter() + .skip(left_chunk.column_count()) + .cloned(), + ); + } else { + types.extend_from_slice(self.right_types.types()); + } + copied_column_types(&types, &self.output_schema) + } + /// Produces an output row. fn produce_row( &self, @@ -835,6 +1051,14 @@ impl NestedLoopJoinOperator { impl Operator for NestedLoopJoinOperator { fn next(&mut self) -> OperatorResult { + if self.left_first && self.left_chunks.is_none() { + let mut chunks = VecDeque::new(); + while let Some(chunk) = self.left.next()? { + chunks.push_back(chunk); + } + self.left_chunks = Some(chunks); + } + // Materialize right side if !self.right_materialized { self.materialize_right()?; @@ -845,23 +1069,17 @@ impl Operator for NestedLoopJoinOperator { return Ok(None); } - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, 2048); - loop { // Get current left chunk if self.current_left_chunk.is_none() { - self.current_left_chunk = self.left.next()?; + self.current_left_chunk = self.next_left()?; self.current_left_row = 0; self.current_right_chunk = 0; self.current_right_row = 0; if self.current_left_chunk.is_none() { // No more left data - return if builder.row_count() > 0 { - Ok(Some(builder.finish())) - } else { - Ok(None) - }; + return Ok(None); } } @@ -870,6 +1088,9 @@ impl Operator for NestedLoopJoinOperator { .as_ref() .expect("left chunk is Some: loaded in loop above"); let left_rows: Vec = left_chunk.selected_indices().collect(); + // Each returned chunk holds rows of one left chunk, built in its + // column types. + let mut builder = DataChunkBuilder::with_capacity(&self.output_types(left_chunk), 2048); // Calculate right column count for potential unmatched rows let right_col_count = if !self.right_chunks.is_empty() { @@ -965,8 +1186,10 @@ impl Operator for NestedLoopJoinOperator { fn reset(&mut self) { self.left.reset(); self.right.reset(); + self.right_types = ColumnTypes::default(); self.right_chunks.clear(); self.right_materialized = false; + self.left_chunks = None; self.current_left_chunk = None; self.current_left_row = 0; self.current_right_chunk = 0; @@ -1103,6 +1326,71 @@ mod tests { assert_eq!(results[2], (3, Some(3))); } + /// A residual condition decides which key matches count: a left row none + /// of whose pairs pass it keeps nulls (once), the others keep the pairs + /// that pass. The condition reads the joined row: left columns, then right. + #[test] + fn test_hash_join_left_outer_with_a_residual() { + struct LeftIsNot2; + impl super::super::filter::Predicate for LeftIsNot2 { + fn evaluate(&self, chunk: &DataChunk, row: usize) -> bool { + assert_eq!( + chunk.column_count(), + 2, + "the left column, then the right one" + ); + chunk.column(0).unwrap().get_int64(row) != Some(2) + } + } + + let left = MockOperator::new(vec![create_int_chunk(&[1, 2, 3])]); + let right = MockOperator::new(vec![create_int_chunk(&[1, 2, 2, 3])]); + let mut join = HashJoinOperator::new( + Box::new(left), + Box::new(right), + vec![0], + vec![0], + JoinType::Left, + vec![LogicalType::Int64, LogicalType::Int64], + ) + .with_residual(Box::new(JoinedRowCondition::new(Box::new(LeftIsNot2)))); + + let mut results = Vec::new(); + while let Some(chunk) = join.next().unwrap() { + for row in chunk.selected_indices() { + let left_val = chunk.column(0).unwrap().get_int64(row).unwrap(); + let right_val = chunk.column(1).unwrap().get_int64(row); + results.push((left_val, right_val)); + } + } + results.sort_by_key(|(l, _)| *l); + assert_eq!(results, [(1, Some(1)), (2, None), (3, Some(3))]); + + // A semi-join keeps the rows with a pair that passes it, an anti-join + // the others. + for (join_type, expected) in [(JoinType::Semi, vec![1, 3]), (JoinType::Anti, vec![2])] { + let left = MockOperator::new(vec![create_int_chunk(&[1, 2, 3])]); + let right = MockOperator::new(vec![create_int_chunk(&[1, 2, 2, 3])]); + let mut join = HashJoinOperator::new( + Box::new(left), + Box::new(right), + vec![0], + vec![0], + join_type, + vec![LogicalType::Int64], + ) + .with_residual(Box::new(JoinedRowCondition::new(Box::new(LeftIsNot2)))); + let mut kept = Vec::new(); + while let Some(chunk) = join.next().unwrap() { + for row in chunk.selected_indices() { + kept.push(chunk.column(0).unwrap().get_int64(row).unwrap()); + } + } + kept.sort_unstable(); + assert_eq!(kept, expected, "{join_type:?}"); + } + } + #[test] fn test_nested_loop_cross_join() { // Left: [1, 2] @@ -1134,6 +1422,173 @@ mod tests { assert_eq!(results, vec![(1, 10), (1, 20), (2, 10), (2, 20)]); } + /// A cross join keeps the column types of both sides, whatever schema it + /// declares: an edge stays an edge, so its properties are not read from + /// the node with the same ID. + #[test] + fn a_cross_join_keeps_the_column_types() { + use grafeo_common::types::{EdgeId, NodeId}; + + let mut left = DataChunkBuilder::new(&[LogicalType::Edge, LogicalType::Node]); + for id in [3_u64, 5] { + left.column_mut(0).unwrap().push_edge_id(EdgeId::new(id)); + left.column_mut(1) + .unwrap() + .push_node_id(NodeId::new(id + 100)); + left.advance_row(); + } + let mut right = DataChunkBuilder::new(&[LogicalType::Node]); + right.column_mut(0).unwrap().push_node_id(NodeId::new(7)); + right.advance_row(); + + let mut join = NestedLoopJoinOperator::new( + Box::new(MockOperator::new(vec![left.finish()])), + Box::new(MockOperator::new(vec![right.finish()])), + None, + JoinType::Cross, + vec![LogicalType::Any, LogicalType::Any, LogicalType::Node], + ); + + let chunk = join.next().unwrap().unwrap(); + assert_eq!( + chunk.column_types(), + [LogicalType::Edge, LogicalType::Node, LogicalType::Node] + ); + let rows: Vec<(u64, u64, u64)> = chunk + .selected_indices() + .map(|row| { + ( + chunk.column(0).unwrap().get_edge_id(row).unwrap().as_u64(), + chunk.column(1).unwrap().get_node_id(row).unwrap().as_u64(), + chunk.column(2).unwrap().get_node_id(row).unwrap().as_u64(), + ) + }) + .collect(); + assert_eq!(rows, [(3, 103, 7), (5, 105, 7)]); + assert!(join.next().unwrap().is_none()); + } + + /// Probe rows (node, key) for keys 1 and 2, nodes 101 and 102. + fn node_probe_chunk() -> DataChunk { + use grafeo_common::types::NodeId; + + let mut probe = DataChunkBuilder::new(&[LogicalType::Node, LogicalType::Int64]); + for key in [1_u64, 2] { + probe + .column_mut(0) + .unwrap() + .push_node_id(NodeId::new(key + 100)); + probe + .column_mut(1) + .unwrap() + .push_int64(i64::try_from(key).unwrap()); + probe.advance_row(); + } + probe.finish() + } + + /// A hash join keeps the column types of both sides (the declared schema + /// says `Any`): a node stays a node and an edge an edge. + #[test] + fn a_hash_join_keeps_the_column_types() { + use grafeo_common::types::EdgeId; + + let mut build = DataChunkBuilder::new(&[LogicalType::Int64, LogicalType::Edge]); + build.column_mut(0).unwrap().push_int64(2); + build.column_mut(1).unwrap().push_edge_id(EdgeId::new(7)); + build.advance_row(); + + let mut join = HashJoinOperator::new( + Box::new(MockOperator::new(vec![node_probe_chunk()])), + Box::new(MockOperator::new(vec![build.finish()])), + vec![1], + vec![0], + JoinType::Inner, + vec![LogicalType::Any; 4], + ); + + let chunk = join.next().unwrap().unwrap(); + assert_eq!( + chunk.column_types(), + [ + LogicalType::Node, + LogicalType::Int64, + LogicalType::Int64, + LogicalType::Edge + ] + ); + assert_eq!(chunk.row_count(), 1); + assert_eq!( + chunk.column(0).unwrap().get_node_id(0).unwrap().as_u64(), + 102 + ); + assert!(chunk.column(0).unwrap().get_edge_id(0).is_none()); + assert_eq!(chunk.column(3).unwrap().get_edge_id(0).unwrap().as_u64(), 7); + assert!(chunk.column(3).unwrap().get_node_id(0).is_none()); + assert!(join.next().unwrap().is_none()); + } + + /// A semi-join returns its probe rows in their own types. + #[test] + fn a_semi_join_keeps_the_probe_types() { + let mut build = DataChunkBuilder::new(&[LogicalType::Int64]); + build.column_mut(0).unwrap().push_int64(1); + build.advance_row(); + + let mut join = HashJoinOperator::new( + Box::new(MockOperator::new(vec![node_probe_chunk()])), + Box::new(MockOperator::new(vec![build.finish()])), + vec![1], + vec![0], + JoinType::Semi, + vec![LogicalType::Any; 2], + ); + + let chunk = join.next().unwrap().unwrap(); + assert_eq!( + chunk.column_types(), + [LogicalType::Node, LogicalType::Int64] + ); + assert_eq!(chunk.row_count(), 1); + assert_eq!( + chunk.column(0).unwrap().get_node_id(0).unwrap().as_u64(), + 101 + ); + } + + /// A left join with no build rows takes the build side's types from the + /// declared schema and the probe side's from its rows. + #[test] + fn a_left_join_without_build_rows_keeps_the_probe_types() { + let mut join = HashJoinOperator::new( + Box::new(MockOperator::new(vec![node_probe_chunk()])), + Box::new(MockOperator::new(vec![])), + vec![1], + vec![0], + JoinType::Left, + vec![ + LogicalType::Any, + LogicalType::Any, + LogicalType::Int64, + LogicalType::Edge, + ], + ); + + let chunk = join.next().unwrap().unwrap(); + assert_eq!( + chunk.column_types(), + [ + LogicalType::Node, + LogicalType::Int64, + LogicalType::Int64, + LogicalType::Edge + ] + ); + assert_eq!(chunk.row_count(), 2); + assert!(chunk.column(3).unwrap().is_null(0)); + assert!(chunk.column(3).unwrap().is_null(1)); + } + #[test] fn test_hash_join_semi() { // Left: [1, 2, 3, 4] diff --git a/crates/grafeo-core/src/execution/operators/leapfrog_join.rs b/crates/grafeo-core/src/execution/operators/leapfrog_join.rs index 0ebb013a6..2140a9cc6 100644 --- a/crates/grafeo-core/src/execution/operators/leapfrog_join.rs +++ b/crates/grafeo-core/src/execution/operators/leapfrog_join.rs @@ -10,7 +10,7 @@ use grafeo_common::types::{EdgeId, LogicalType, NodeId, Value}; use super::{Operator, OperatorError, OperatorResult}; use crate::execution::DataChunk; -use crate::execution::chunk::DataChunkBuilder; +use crate::execution::chunk::{ColumnTypes, DataChunkBuilder, copied_column_types}; use crate::index::trie::{LeapfrogJoin, TrieIndex}; /// Row identifier for reconstructing output: (input_index, chunk_index, row_index). @@ -44,6 +44,9 @@ pub struct LeapfrogJoinOperator { /// Materialized input chunks (built once during first next() call). materialized_inputs: Vec>, + /// The column types of each input's materialized chunks. + input_types: Vec, + /// TrieIndex structures built from materialized inputs. tries: Vec, @@ -85,6 +88,7 @@ impl LeapfrogJoinOperator { output_schema, output_column_mapping, materialized_inputs: Vec::new(), + input_types: Vec::new(), tries: Vec::new(), materialized: false, results: Vec::new(), @@ -94,15 +98,38 @@ impl LeapfrogJoinOperator { } } + /// The output column types: those of the input columns they are copied + /// from (a node or edge stays one, see [`ColumnTypes`]), declared for an + /// input without rows. + fn output_types(&self) -> Vec { + let input: Vec = self + .output_column_mapping + .iter() + .enumerate() + .map(|(out_col, &(input_idx, column))| { + self.input_types + .get(input_idx) + .and_then(|types| types.types().get(column)) + .or_else(|| self.output_schema.get(out_col)) + .cloned() + .unwrap_or(LogicalType::Any) + }) + .collect(); + copied_column_types(&input, &self.output_schema) + } + /// Materializes all inputs and builds trie indexes. fn materialize_inputs(&mut self) -> Result<(), OperatorError> { // Phase 1: Collect all chunks from each input for input in &mut self.inputs { let mut chunks = Vec::new(); + let mut types = ColumnTypes::default(); while let Some(chunk) = input.next()? { + types.add(&chunk); chunks.push(chunk); } self.materialized_inputs.push(chunks); + self.input_types.push(types); } // Phase 2: Build TrieIndex for each input @@ -336,7 +363,7 @@ impl Operator for LeapfrogJoinOperator { return Ok(None); } - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, 2048); + let mut builder = DataChunkBuilder::with_capacity(&self.output_types(), 2048); while !builder.is_full() { self.build_output_row(&mut builder)?; @@ -359,6 +386,7 @@ impl Operator for LeapfrogJoinOperator { input.reset(); } self.materialized_inputs.clear(); + self.input_types.clear(); self.tries.clear(); self.materialized = false; self.results.clear(); @@ -427,6 +455,38 @@ mod tests { DataChunk::new(vec![col]) } + /// The output columns keep the types of the input columns they come from. + #[test] + fn output_columns_keep_their_input_types() { + let mut keys = ValueVector::with_type(LogicalType::Int64); + let mut nodes = ValueVector::with_type(LogicalType::Node); + for key in [1_u64, 2] { + keys.push_int64(i64::try_from(key).unwrap()); + nodes.push_node_id(NodeId::new(100 + key)); + } + let inputs: Vec> = vec![ + Box::new(MockScanOperator::new(DataChunk::new(vec![keys, nodes]))), + Box::new(MockScanOperator::new(create_node_chunk(&[2]))), + ]; + let mut leapfrog = LeapfrogJoinOperator::new( + inputs, + vec![vec![0], vec![0]], + vec![LogicalType::Any; 3], + vec![(0, 0), (0, 1), (1, 0)], + ); + + let chunk = leapfrog.next().unwrap().unwrap(); + assert_eq!( + chunk.column_types(), + [LogicalType::Int64, LogicalType::Node, LogicalType::Int64] + ); + assert_eq!(chunk.row_count(), 1); + assert_eq!( + chunk.column(1).unwrap().get_node_id(0).unwrap().as_u64(), + 102 + ); + } + #[test] fn test_leapfrog_binary_intersection() { // Input 1: nodes [1, 2, 3, 5] diff --git a/crates/grafeo-core/src/execution/operators/limit.rs b/crates/grafeo-core/src/execution/operators/limit.rs index 5816076c9..80ec1ae88 100644 --- a/crates/grafeo-core/src/execution/operators/limit.rs +++ b/crates/grafeo-core/src/execution/operators/limit.rs @@ -5,32 +5,27 @@ //! - `SkipOperator`: Skips a number of input rows //! - `LimitSkipOperator`: Combined LIMIT and OFFSET/SKIP -use grafeo_common::types::{LogicalType, Value}; - use super::{Operator, OperatorResult}; -use crate::execution::chunk::DataChunkBuilder; /// Limit operator. /// -/// Returns at most `limit` rows from the input. +/// Returns at most `limit` rows from the input. A chunk it cuts short keeps +/// its columns and values as they are. pub struct LimitOperator { /// Child operator. child: Box, /// Maximum number of rows to return. limit: usize, - /// Output schema. - output_schema: Vec, /// Number of rows returned so far. returned: usize, } impl LimitOperator { /// Creates a new limit operator. - pub fn new(child: Box, limit: usize, output_schema: Vec) -> Self { + pub fn new(child: Box, limit: usize) -> Self { Self { child, limit, - output_schema, returned: 0, } } @@ -65,32 +60,11 @@ impl Operator for LimitOperator { return Ok(Some(chunk)); } - // Return partial chunk - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, remaining); - - let mut count = 0; - for row in chunk.selected_indices() { - if count >= remaining { - break; - } - - for col_idx in 0..chunk.column_count() { - if let (Some(src_col), Some(dst_col)) = - (chunk.column(col_idx), builder.column_mut(col_idx)) - { - if let Some(value) = src_col.get_value(row) { - dst_col.push_value(value); - } else { - dst_col.push_value(Value::Null); - } - } - } - builder.advance_row(); - count += 1; - } - - self.returned += count; - return Ok(Some(builder.finish())); + // The first rows of the chunk, copied as they are (a column + // rebuilt by a declared type would turn values of another type + // into that type's default). + self.returned += remaining; + return Ok(Some(chunk.slice(0, remaining))); } } @@ -110,25 +84,23 @@ impl Operator for LimitOperator { /// Skip operator. /// -/// Skips the first `skip` rows from the input. +/// Skips the first `skip` rows from the input. A chunk it cuts short keeps +/// its columns and values as they are. pub struct SkipOperator { /// Child operator. child: Box, /// Number of rows to skip. skip: usize, - /// Output schema. - output_schema: Vec, /// Number of rows skipped so far. skipped: usize, } impl SkipOperator { /// Creates a new skip operator. - pub fn new(child: Box, skip: usize, output_schema: Vec) -> Self { + pub fn new(child: Box, skip: usize) -> Self { Self { child, skip, - output_schema, skipped: 0, } } @@ -154,26 +126,7 @@ impl Operator for SkipOperator { // Skip partial chunk self.skipped = self.skip; - let mut builder = - DataChunkBuilder::with_capacity(&self.output_schema, row_count - to_skip); - - let rows: Vec = chunk.selected_indices().collect(); - for &row in rows.iter().skip(to_skip) { - for col_idx in 0..chunk.column_count() { - if let (Some(src_col), Some(dst_col)) = - (chunk.column(col_idx), builder.column_mut(col_idx)) - { - if let Some(value) = src_col.get_value(row) { - dst_col.push_value(value); - } else { - dst_col.push_value(Value::Null); - } - } - } - builder.advance_row(); - } - - return Ok(Some(builder.finish())); + return Ok(Some(chunk.slice(to_skip, row_count - to_skip))); } // After skipping, just pass through @@ -196,7 +149,8 @@ impl Operator for SkipOperator { /// Combined Limit and Skip operator. /// -/// Equivalent to OFFSET skip LIMIT limit. +/// Equivalent to OFFSET skip LIMIT limit. A chunk it cuts short keeps its +/// columns and values as they are. pub struct LimitSkipOperator { /// Child operator. child: Box, @@ -204,8 +158,6 @@ pub struct LimitSkipOperator { skip: usize, /// Maximum number of rows to return. limit: usize, - /// Output schema. - output_schema: Vec, /// Number of rows skipped so far. skipped: usize, /// Number of rows returned so far. @@ -214,17 +166,11 @@ pub struct LimitSkipOperator { impl LimitSkipOperator { /// Creates a new limit/skip operator. - pub fn new( - child: Box, - skip: usize, - limit: usize, - output_schema: Vec, - ) -> Self { + pub fn new(child: Box, skip: usize, limit: usize) -> Self { Self { child, skip, limit, - output_schema, skipped: 0, returned: 0, } @@ -248,7 +194,6 @@ impl Operator for LimitSkipOperator { continue; } - let rows: Vec = chunk.selected_indices().collect(); let mut start_idx = 0; // Skip rows if needed @@ -271,25 +216,11 @@ impl Operator for LimitSkipOperator { return Ok(None); } - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, to_return); - - for &row in rows.iter().skip(start_idx).take(to_return) { - for col_idx in 0..chunk.column_count() { - if let (Some(src_col), Some(dst_col)) = - (chunk.column(col_idx), builder.column_mut(col_idx)) - { - if let Some(value) = src_col.get_value(row) { - dst_col.push_value(value); - } else { - dst_col.push_value(Value::Null); - } - } - } - builder.advance_row(); - } - self.returned += to_return; - return Ok(Some(builder.finish())); + if start_idx == 0 && to_return == row_count { + return Ok(Some(chunk)); + } + return Ok(Some(chunk.slice(start_idx, to_return))); } } @@ -313,6 +244,7 @@ mod tests { use super::*; use crate::execution::DataChunk; use crate::execution::chunk::DataChunkBuilder; + use grafeo_common::types::{LogicalType, Value}; struct MockOperator { chunks: Vec, @@ -365,7 +297,7 @@ mod tests { fn test_limit() { let mock = MockOperator::new(vec![create_numbered_chunk(&[1, 2, 3, 4, 5])]); - let mut limit = LimitOperator::new(Box::new(mock), 3, vec![LogicalType::Int64]); + let mut limit = LimitOperator::new(Box::new(mock), 3); let mut results = Vec::new(); while let Some(chunk) = limit.next().unwrap() { @@ -382,7 +314,7 @@ mod tests { fn test_limit_larger_than_input() { let mock = MockOperator::new(vec![create_numbered_chunk(&[1, 2, 3])]); - let mut limit = LimitOperator::new(Box::new(mock), 10, vec![LogicalType::Int64]); + let mut limit = LimitOperator::new(Box::new(mock), 10); let mut results = Vec::new(); while let Some(chunk) = limit.next().unwrap() { @@ -399,7 +331,7 @@ mod tests { fn test_skip() { let mock = MockOperator::new(vec![create_numbered_chunk(&[1, 2, 3, 4, 5])]); - let mut skip = SkipOperator::new(Box::new(mock), 2, vec![LogicalType::Int64]); + let mut skip = SkipOperator::new(Box::new(mock), 2); let mut results = Vec::new(); while let Some(chunk) = skip.next().unwrap() { @@ -416,7 +348,7 @@ mod tests { fn test_skip_all() { let mock = MockOperator::new(vec![create_numbered_chunk(&[1, 2, 3])]); - let mut skip = SkipOperator::new(Box::new(mock), 5, vec![LogicalType::Int64]); + let mut skip = SkipOperator::new(Box::new(mock), 5); let result = skip.next().unwrap(); assert!(result.is_none()); @@ -431,8 +363,7 @@ mod tests { let mut op = LimitSkipOperator::new( Box::new(mock), 3, // Skip first 3 - 4, // Take next 4 - vec![LogicalType::Int64], + 4, ); let mut results = Vec::new(); @@ -454,7 +385,7 @@ mod tests { create_numbered_chunk(&[5, 6]), ]); - let mut limit = LimitOperator::new(Box::new(mock), 5, vec![LogicalType::Int64]); + let mut limit = LimitOperator::new(Box::new(mock), 5); let mut results = Vec::new(); while let Some(chunk) = limit.next().unwrap() { @@ -475,7 +406,7 @@ mod tests { create_numbered_chunk(&[5, 6]), ]); - let mut skip = SkipOperator::new(Box::new(mock), 3, vec![LogicalType::Int64]); + let mut skip = SkipOperator::new(Box::new(mock), 3); let mut results = Vec::new(); while let Some(chunk) = skip.next().unwrap() { @@ -491,7 +422,7 @@ mod tests { #[test] fn test_limit_into_parts() { let child = Box::new(MockOperator::new(vec![])); - let limit = LimitOperator::new(child, 42, vec![LogicalType::Int64]); + let limit = LimitOperator::new(child, 42); let (_, limit_value) = limit.into_parts(); assert_eq!(limit_value, 42); } @@ -499,8 +430,84 @@ mod tests { #[test] fn test_limit_into_any() { let child = Box::new(MockOperator::new(vec![])); - let limit: Box = Box::new(LimitOperator::new(child, 10, vec![])); + let limit: Box = Box::new(LimitOperator::new(child, 10)); let any = limit.into_any(); assert!(any.downcast::().is_ok()); } + + /// A chunk of `values` in an untyped column, with its first row filtered + /// out: the operators must also respect the selection. + fn mixed_chunk(values: &[Value]) -> DataChunk { + let mut chunk = DataChunk::new(vec![crate::execution::ValueVector::from_values(values)]); + chunk.set_selection(crate::execution::SelectionVector::from_predicate( + values.len(), + |row| row > 0, + )); + chunk + } + + /// The values the operator returns, in order, and the type of the column + /// of each chunk it returns. + fn output_of(mut op: impl Operator) -> (Vec, Vec) { + let (mut values, mut types) = (Vec::new(), Vec::new()); + while let Some(chunk) = op.next().unwrap() { + let column = chunk.column(0).unwrap(); + types.push(column.data_type().clone()); + values.extend( + chunk + .selected_indices() + .map(|row| column.get_value(row).unwrap()), + ); + } + (values, types) + } + + /// #482: cutting a chunk short keeps its values as they are, a mix of + /// types and a null in an untyped column included. Before, the cut rows + /// were rebuilt in columns of a declared type, which turned every value + /// of another type into that type's default. + #[test] + fn a_partial_chunk_keeps_its_values() { + let values = [ + Value::Int64(0), + Value::Int64(7), + Value::from("eight"), + Value::Null, + Value::Int64(9), + ]; + let child = || Box::new(MockOperator::new(vec![mixed_chunk(&values)])); + + assert_eq!(output_of(LimitOperator::new(child(), 3)).0, values[1..4]); + assert_eq!(output_of(SkipOperator::new(child(), 2)).0, values[3..]); + assert_eq!( + output_of(LimitSkipOperator::new(child(), 1, 2)).0, + values[2..4] + ); + } + + /// A cut chunk keeps its column's type: here node IDs stay node IDs. + #[test] + fn a_partial_chunk_keeps_its_column_type() { + let child = || { + let mut builder = DataChunkBuilder::new(&[LogicalType::Node]); + for id in 1..=4 { + builder + .column_mut(0) + .unwrap() + .push_node_id(grafeo_common::types::NodeId::new(id)); + builder.advance_row(); + } + Box::new(MockOperator::new(vec![builder.finish()])) + }; + + let (values, types) = output_of(LimitOperator::new(child(), 2)); + assert_eq!(values, [Value::Int64(1), Value::Int64(2)]); + assert_eq!(types, [LogicalType::Node]); + let (values, types) = output_of(SkipOperator::new(child(), 3)); + assert_eq!(values, [Value::Int64(4)]); + assert_eq!(types, [LogicalType::Node]); + let (values, types) = output_of(LimitSkipOperator::new(child(), 1, 2)); + assert_eq!(values, [Value::Int64(2), Value::Int64(3)]); + assert_eq!(types, [LogicalType::Node]); + } } diff --git a/crates/grafeo-core/src/execution/operators/merge.rs b/crates/grafeo-core/src/execution/operators/merge.rs index 9bfda31ae..671fba075 100644 --- a/crates/grafeo-core/src/execution/operators/merge.rs +++ b/crates/grafeo-core/src/execution/operators/merge.rs @@ -9,7 +9,7 @@ use super::{ ExpressionPredicate, GraphWriter, Operator, OperatorError, OperatorResult, PropertySource, SessionContext, }; -use crate::execution::chunk::{DataChunk, DataChunkBuilder}; +use crate::execution::chunk::{DataChunk, DataChunkBuilder, copied_column_types}; use crate::graph::{GraphStore, GraphStoreSearch}; use grafeo_common::types::{ EdgeId, EpochId, LogicalType, NodeId, PropertyKey, TransactionId, Value, @@ -38,6 +38,26 @@ pub struct MergeConfig { pub bound_variable_column: Option, } +/// The column types of output rows for `chunk` (none for a standalone MERGE), +/// as many as the declared ones: its own for the input columns, which are +/// copied (a node or edge stays one, see `ColumnTypes`), then the declared +/// ones, with `entity` for the column of the merged node or edge. +fn output_types( + chunk: Option<&DataChunk>, + declared: &[LogicalType], + entity_column: usize, + entity: LogicalType, +) -> Vec { + let mut input = chunk.map(DataChunk::column_types).unwrap_or_default(); + input.truncate(declared.len()); + let mut types = copied_column_types(&input, declared); + types.extend(declared.iter().skip(input.len()).cloned()); + if let Some(column_type) = types.get_mut(entity_column) { + *column_type = entity; + } + types +} + /// Merge operator for MERGE clause. /// /// Tries to match a node with the given labels and properties. @@ -150,7 +170,13 @@ impl MergeOperator { row: usize, merged_node: NodeId, ) -> DataChunk { - let mut builder = DataChunkBuilder::with_capacity(&self.config.output_schema, 1); + let types = output_types( + chunk, + &self.config.output_schema, + self.config.output_column, + LogicalType::Node, + ); + let mut builder = DataChunkBuilder::with_capacity(&types, 1); if let Some(input) = chunk { for col_idx in 0..input.column_count() { let val = input @@ -226,8 +252,9 @@ impl MergeOperator { Ok(out) } - /// Tries to find a matching node with the given resolved properties. - fn find_matching_node(&self, resolved_match_props: &[(String, Value)]) -> Option { + /// The nodes that match the given resolved properties (every one, as in + /// openCypher, where MERGE binds each match). + fn find_matching_nodes(&self, resolved_match_props: &[(String, Value)]) -> Vec { // Use a property index when available to avoid a full label scan. // Null conditions are excluded from the index query and verified in the loop. let use_index = resolved_match_props @@ -247,6 +274,7 @@ impl MergeOperator { self.writer.store().node_ids() }; + let mut matches = Vec::new(); for node_id in candidates { // Transactional creates write their version at `EpochId::PENDING`, // so the unversioned `get_node` (which checks visibility against @@ -278,11 +306,11 @@ impl MergeOperator { }); if has_all_props { - return Some(node_id); + matches.push(node_id); } } - None + matches } /// Merges match and ON CREATE property lists, with ON CREATE values @@ -302,12 +330,13 @@ impl MergeOperator { merged } - /// Finds or creates a matching node for a single row, applying ON MATCH/ON CREATE. + /// Finds the matching nodes for a single row and applies ON MATCH to each, + /// or creates one and applies ON CREATE. fn merge_node_for_row( &mut self, chunk: Option<&DataChunk>, row: usize, - ) -> Result { + ) -> Result, super::OperatorError> { let store_ref: &dyn GraphStore = self.writer.store().as_ref(); // Match properties cannot reference the MERGE variable (ISO §15.5), // so they resolve against the input chunk directly. @@ -325,18 +354,21 @@ impl MergeOperator { }, )?; - if let Some(existing_id) = self.find_matching_node(&resolved_match) { - // Resolve ON MATCH SET against an augmented row containing the - // matched node id, so `coalesce(n.x, 0)` can read the live value. - let resolved_on_match = self.resolve_action_properties( - &self.config.on_match_properties, - chunk, - row, - existing_id, - )?; - self.writer - .set_node_properties(existing_id, &resolved_on_match, false)?; - Ok(existing_id) + let matches = self.find_matching_nodes(&resolved_match); + if !matches.is_empty() { + for &existing_id in &matches { + // Resolve ON MATCH SET against an augmented row containing the + // matched node id, so `coalesce(n.x, 0)` can read the live value. + let resolved_on_match = self.resolve_action_properties( + &self.config.on_match_properties, + chunk, + row, + existing_id, + )?; + self.writer + .set_node_properties(existing_id, &resolved_on_match, false)?; + } + Ok(matches) } else if Self::has_expression_source(&self.config.on_create_properties) { // ON CREATE expressions read the new node, so it is created from // the match properties first; the whole property set is checked @@ -350,14 +382,17 @@ impl MergeOperator { new_id, ) }) + .map(|created| vec![created]) } else { // No runtime expressions: create with all properties at once. let resolved_on_create = Self::resolve_properties(&self.config.on_create_properties, chunk, row, store_ref); - self.writer.create_node( - &self.config.labels, - Self::merge_node_props(&resolved_match, &resolved_on_create), - ) + self.writer + .create_node( + &self.config.labels, + Self::merge_node_props(&resolved_match, &resolved_on_create), + ) + .map(|created| vec![created]) } } } @@ -368,9 +403,14 @@ impl Operator for MergeOperator { // merged node ID appended (used for chained inline MERGE patterns). if let Some(ref mut input) = self.input { if let Some(chunk) = input.next()? { - let mut builder = - DataChunkBuilder::with_capacity(&self.config.output_schema, chunk.row_count()); - + let types = output_types( + Some(&chunk), + &self.config.output_schema, + self.config.output_column, + LogicalType::Node, + ); + // A row comes out once per node it merges (every match). + let mut merged = Vec::with_capacity(chunk.row_count()); for row in chunk.selected_indices() { // Reject NULL bound variables (e.g., from unmatched OPTIONAL MATCH) if let Some(bound_col) = self.config.bound_variable_column { @@ -387,8 +427,13 @@ impl Operator for MergeOperator { } // Merge the node per-row: resolve properties from this row - let node_id = self.merge_node_for_row(Some(&chunk), row)?; + for node_id in self.merge_node_for_row(Some(&chunk), row)? { + merged.push((row, node_id)); + } + } + let mut builder = DataChunkBuilder::with_capacity(&types, merged.len().max(1)); + for (row, node_id) in merged { // Copy input columns to output for col_idx in 0..chunk.column_count() { if let (Some(src), Some(dst)) = @@ -421,13 +466,21 @@ impl Operator for MergeOperator { } self.executed = true; - let node_id = self.merge_node_for_row(None, 0)?; + let node_ids = self.merge_node_for_row(None, 0)?; - let mut builder = DataChunkBuilder::new(&self.config.output_schema); - if let Some(dst) = builder.column_mut(self.config.output_column) { - dst.push_node_id(node_id); + let types = output_types( + None, + &self.config.output_schema, + self.config.output_column, + LogicalType::Node, + ); + let mut builder = DataChunkBuilder::with_capacity(&types, node_ids.len().max(1)); + for node_id in node_ids { + if let Some(dst) = builder.column_mut(self.config.output_column) { + dst.push_node_id(node_id); + } + builder.advance_row(); } - builder.advance_row(); Ok(Some(builder.finish())) } @@ -532,7 +585,13 @@ impl MergeRelationshipOperator { row: usize, merged_edge: EdgeId, ) -> DataChunk { - let mut builder = DataChunkBuilder::with_capacity(&self.config.output_schema, 1); + let types = output_types( + Some(chunk), + &self.config.output_schema, + self.config.edge_output_column, + LogicalType::Edge, + ); + let mut builder = DataChunkBuilder::with_capacity(&types, 1); for col_idx in 0..chunk.column_count() { let val = chunk .column(col_idx) @@ -601,15 +660,17 @@ impl MergeRelationshipOperator { Ok(out) } - /// Tries to find a matching relationship between source and target. - fn find_matching_edge( + /// The relationships between source and target that match (every one, + /// as in openCypher, where MERGE binds each match). + fn find_matching_edges( &self, src: NodeId, dst: NodeId, resolved_match_props: &[(String, Value)], - ) -> Option { + ) -> Vec { use crate::graph::Direction; + let mut matches = Vec::new(); for (target, edge_id) in self.writer.store().edges_from(src, Direction::Outgoing) { if target != dst { continue; @@ -641,12 +702,12 @@ impl MergeRelationshipOperator { }); if has_all_props { - return Some(edge_id); + matches.push(edge_id); } } } - None + matches } } @@ -655,9 +716,14 @@ impl Operator for MergeRelationshipOperator { use super::OperatorError; if let Some(chunk) = self.input.next()? { - let mut builder = - DataChunkBuilder::with_capacity(&self.config.output_schema, chunk.row_count()); - + let types = output_types( + Some(&chunk), + &self.config.output_schema, + self.config.edge_output_column, + LogicalType::Edge, + ); + // A row comes out once per relationship it merges (every match). + let mut merged = Vec::with_capacity(chunk.row_count()); for row in chunk.selected_indices() { let src_val = chunk .column(self.config.source_column) @@ -696,49 +762,57 @@ impl Operator for MergeRelationshipOperator { }, )?; - let edge_id = if let Some(existing) = - self.find_matching_edge(src_val, dst_val, &resolved_match) - { - let resolved_on_match = self.resolve_action_properties( - &self.config.on_match_properties, - &chunk, - row, - existing, - )?; - self.writer - .set_edge_properties(existing, &resolved_on_match, false)?; - existing - } else if MergeOperator::has_expression_source(&self.config.on_create_properties) { - // ON CREATE expressions read the new edge: see MergeOperator. - self.writer.create_edge_with( - src_val, - dst_val, - &self.config.edge_type, - resolved_match, - |new_id| { - self.resolve_action_properties( - &self.config.on_create_properties, - &chunk, - row, - new_id, - ) - }, - )? - } else { - let resolved_on_create = MergeOperator::resolve_properties( - &self.config.on_create_properties, - Some(&chunk), - row, - store_ref, - ); - self.writer.create_edge( - src_val, - dst_val, - &self.config.edge_type, - MergeOperator::merge_node_props(&resolved_match, &resolved_on_create), - )? - }; + let matches = self.find_matching_edges(src_val, dst_val, &resolved_match); + if !matches.is_empty() { + for &existing in &matches { + let resolved_on_match = self.resolve_action_properties( + &self.config.on_match_properties, + &chunk, + row, + existing, + )?; + self.writer + .set_edge_properties(existing, &resolved_on_match, false)?; + merged.push((row, existing)); + } + continue; + } + let edge_id = + if MergeOperator::has_expression_source(&self.config.on_create_properties) { + // ON CREATE expressions read the new edge: see MergeOperator. + self.writer.create_edge_with( + src_val, + dst_val, + &self.config.edge_type, + resolved_match, + |new_id| { + self.resolve_action_properties( + &self.config.on_create_properties, + &chunk, + row, + new_id, + ) + }, + )? + } else { + let resolved_on_create = MergeOperator::resolve_properties( + &self.config.on_create_properties, + Some(&chunk), + row, + store_ref, + ); + self.writer.create_edge( + src_val, + dst_val, + &self.config.edge_type, + MergeOperator::merge_node_props(&resolved_match, &resolved_on_create), + )? + }; + merged.push((row, edge_id)); + } + let mut builder = DataChunkBuilder::with_capacity(&types, merged.len().max(1)); + for (row, edge_id) in merged { // Copy input columns to output, then add the edge column for col_idx in 0..self.config.output_schema.len() { if col_idx == self.config.edge_output_column { diff --git a/crates/grafeo-core/src/execution/operators/mod.rs b/crates/grafeo-core/src/execution/operators/mod.rs index ffb1c3a89..2179a885e 100644 --- a/crates/grafeo-core/src/execution/operators/mod.rs +++ b/crates/grafeo-core/src/execution/operators/mod.rs @@ -81,7 +81,8 @@ pub use filter::{ }; pub use horizontal_aggregate::{EntityKind, HorizontalAggregateOperator}; pub use join::{ - EqualityCondition, HashJoinOperator, HashKey, JoinCondition, JoinType, NestedLoopJoinOperator, + EqualityCondition, HashJoinOperator, HashKey, JoinCondition, JoinType, JoinedRowCondition, + NestedLoopJoinOperator, }; pub use leapfrog_join::LeapfrogJoinOperator; pub use limit::{LimitOperator, LimitSkipOperator, SkipOperator}; diff --git a/crates/grafeo-core/src/execution/operators/mutation.rs b/crates/grafeo-core/src/execution/operators/mutation.rs index 61124e9ce..ffeff3c19 100644 --- a/crates/grafeo-core/src/execution/operators/mutation.rs +++ b/crates/grafeo-core/src/execution/operators/mutation.rs @@ -15,7 +15,7 @@ use grafeo_common::types::{ use super::filter::{ExpressionPredicate, FilterExpression}; use super::{GraphWriter, Operator, OperatorError, OperatorResult, SessionContext}; -use crate::execution::chunk::{DataChunk, DataChunkBuilder}; +use crate::execution::chunk::{DataChunk, DataChunkBuilder, copied_column_types}; use crate::graph::{GraphStore, GraphStoreSearch}; /// Trait for validating schema constraints during mutation operations. @@ -399,8 +399,11 @@ impl Operator for CreateNodeOperator { let Some(chunk) = input.next()? else { return Ok(None); }; - let mut builder = - DataChunkBuilder::with_capacity(&self.output_schema, chunk.row_count()); + let mut types = output_types(&chunk, self.output_column, &self.output_schema); + if let Some(node_type) = types.get_mut(self.output_column) { + *node_type = LogicalType::Node; + } + let mut builder = DataChunkBuilder::with_capacity(&types, chunk.row_count()); for row in chunk.selected_indices() { let properties = self.expressions.resolve_row( @@ -445,7 +448,11 @@ impl Operator for CreateNodeOperator { .collect(); let node_id = self.writer.create_node(&self.labels, properties)?; - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, 1); + let mut types = self.output_schema.clone(); + if let Some(node_type) = types.get_mut(self.output_column) { + *node_type = LogicalType::Node; + } + let mut builder = DataChunkBuilder::with_capacity(&types, 1); if let Some(dst) = builder.column_mut(self.output_column) { dst.push_value(id_value(node_id.0)); } @@ -493,6 +500,18 @@ fn id_at( } } +/// The column types of the output chunk for `chunk`, as many as the declared +/// ones: its own for the first `copied` columns, copied from it (a node or edge +/// stays one, see `ColumnTypes`), then the declared ones. Input columns past the +/// declared ones (the planner's hidden expression columns) are not passed on. +fn output_types(chunk: &DataChunk, copied: usize, declared: &[LogicalType]) -> Vec { + let copied = copied.min(chunk.column_count()).min(declared.len()); + let input: Vec = chunk.column_types().into_iter().take(copied).collect(); + let mut types = copied_column_types(&input, declared); + types.extend(declared.iter().skip(copied).cloned()); + types +} + /// Copies the first `columns` input columns of `row` to the output row. fn copy_columns(chunk: &DataChunk, row: usize, builder: &mut DataChunkBuilder, columns: usize) { for col_idx in 0..columns.min(chunk.column_count()) { @@ -589,7 +608,11 @@ impl Operator for CreateEdgeOperator { let Some(chunk) = self.input.next()? else { return Ok(None); }; - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, chunk.row_count()); + let mut types = output_types(&chunk, chunk.column_count(), &self.output_schema); + if let Some(edge_type) = self.output_column.and_then(|column| types.get_mut(column)) { + *edge_type = LogicalType::Edge; + } + let mut builder = DataChunkBuilder::with_capacity(&types, chunk.row_count()); for row in chunk.selected_indices() { let from = NodeId(id_at(&chunk, self.from_column, row, "from", "node")?); @@ -669,7 +692,8 @@ impl Operator for DeleteNodeOperator { let Some(chunk) = self.input.next()? else { return Ok(None); }; - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, chunk.row_count()); + let types = output_types(&chunk, chunk.column_count(), &self.output_schema); + let mut builder = DataChunkBuilder::with_capacity(&types, chunk.row_count()); for row in chunk.selected_indices() { let node_id = NodeId(id_at(&chunk, self.node_column, row, "node", "node")?); @@ -731,7 +755,8 @@ impl Operator for DeleteEdgeOperator { let Some(chunk) = self.input.next()? else { return Ok(None); }; - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, chunk.row_count()); + let types = output_types(&chunk, chunk.column_count(), &self.output_schema); + let mut builder = DataChunkBuilder::with_capacity(&types, chunk.row_count()); for row in chunk.selected_indices() { let edge_id = EdgeId(id_at(&chunk, self.edge_column, row, "edge", "edge")?); @@ -798,7 +823,8 @@ impl Operator for AddLabelOperator { let Some(chunk) = self.input.next()? else { return Ok(None); }; - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, chunk.row_count()); + let types = output_types(&chunk, chunk.column_count(), &self.output_schema); + let mut builder = DataChunkBuilder::with_capacity(&types, chunk.row_count()); for row in chunk.selected_indices() { let node_id = NodeId(id_at(&chunk, self.node_column, row, "node", "node")?); @@ -869,7 +895,8 @@ impl Operator for RemoveLabelOperator { let Some(chunk) = self.input.next()? else { return Ok(None); }; - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, chunk.row_count()); + let types = output_types(&chunk, chunk.column_count(), &self.output_schema); + let mut builder = DataChunkBuilder::with_capacity(&types, chunk.row_count()); for row in chunk.selected_indices() { let node_id = NodeId(id_at(&chunk, self.node_column, row, "node", "node")?); @@ -970,7 +997,8 @@ impl Operator for SetPropertyOperator { let Some(chunk) = self.input.next()? else { return Ok(None); }; - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, chunk.row_count()); + let types = output_types(&chunk, chunk.column_count(), &self.output_schema); + let mut builder = DataChunkBuilder::with_capacity(&types, chunk.row_count()); for row in chunk.selected_indices() { let entity_id = id_at(&chunk, self.entity_column, row, "entity", "entity")?; @@ -1604,12 +1632,13 @@ mod tests { assert_eq!(chunk.row_count(), 1); assert_eq!(store.edge_count(), 1); - // Verify the output chunk contains the edge ID in column 2 - let edge_id_raw = chunk + // Verify the output chunk contains the edge in column 2, as an edge + // (whatever type the planner declared for it) + assert_eq!(chunk.column_types()[2], LogicalType::Edge); + let edge_id = chunk .column(2) - .and_then(|c| c.get_int64(0)) + .and_then(|c| c.get_edge_id(0)) .expect("edge ID should be in output column 2"); - let edge_id = EdgeId(edge_id_raw as u64); // Verify the edge has the property let edge = store.get_edge(edge_id).expect("edge should exist"); diff --git a/crates/grafeo-core/src/execution/operators/parameter_scan.rs b/crates/grafeo-core/src/execution/operators/parameter_scan.rs index 4d2c4b7da..51eaa28e1 100644 --- a/crates/grafeo-core/src/execution/operators/parameter_scan.rs +++ b/crates/grafeo-core/src/execution/operators/parameter_scan.rs @@ -21,8 +21,9 @@ use grafeo_common::types::Value; pub struct ParameterState { /// Column names for the injected parameters. pub columns: Vec, - /// Current row values (set by Apply before each inner execution). - values: Mutex>>, + /// Current row values and the types of the columns they come from (set by + /// Apply before each inner execution; no types: any). + values: Mutex, Vec)>>, } impl ParameterState { @@ -35,9 +36,17 @@ impl ParameterState { } } - /// Sets the current parameter values (called by the Apply operator). + /// Sets the current parameter values (called by the Apply operator), in + /// columns of any type. pub fn set_values(&self, values: Vec) { - *self.values.lock() = Some(values); + *self.values.lock() = Some((values, Vec::new())); + } + + /// Sets the current parameter values with the types of the columns they + /// come from, so a node or edge of the outer row stays one in the inner + /// plan (in a column of any type an ID reads whichever entity has it). + pub fn set_typed_values(&self, values: Vec, types: Vec) { + *self.values.lock() = Some((values, types)); } /// Clears the current parameter values. @@ -45,9 +54,11 @@ impl ParameterState { *self.values.lock() = None; } - /// Takes the current parameter values. - fn take_values(&self) -> Option> { - self.values.lock().take() + /// The current parameter values and their column types. They stay set: + /// every scan of the state reads them (each branch of a `UNION` in the + /// subquery starts from one, and a rescan after `reset` reads them again). + fn current_values(&self) -> Option<(Vec, Vec)> { + self.values.lock().clone() } } @@ -85,15 +96,18 @@ impl Operator for ParameterScanOperator { } self.emitted = true; - let Some(values) = self.state.take_values() else { + let Some((values, types)) = self.state.current_values() else { return Ok(None); }; - // Build a single-row DataChunk with one column per parameter + // Build a single-row DataChunk with one column per parameter, in the + // type of the column it comes from. let columns: Vec = values .into_iter() - .map(|val| { - let mut col = ValueVector::with_capacity(LogicalType::Any, 1); + .enumerate() + .map(|(i, val)| { + let column_type = types.get(i).cloned().unwrap_or(LogicalType::Any); + let mut col = ValueVector::with_capacity(column_type, 1); col.push_value(val); col }) @@ -160,6 +174,42 @@ mod tests { assert_eq!(chunk.column(0).unwrap().get_value(0), Some(Value::Int64(2))); } + /// Typed values come out in the types of the columns they came from. + #[test] + fn typed_values_keep_their_column_types() { + let state = Arc::new(ParameterState::new(vec!["a".to_string(), "r".to_string()])); + let mut op = ParameterScanOperator::new(Arc::clone(&state)); + + state.set_typed_values( + vec![Value::Int64(3), Value::Int64(3)], + vec![LogicalType::Node, LogicalType::Edge], + ); + let chunk = op.next().unwrap().expect("should emit a chunk"); + assert_eq!(chunk.column_types(), [LogicalType::Node, LogicalType::Edge]); + assert_eq!(chunk.column(0).unwrap().get_node_id(0).unwrap().as_u64(), 3); + assert!(chunk.column(0).unwrap().get_edge_id(0).is_none()); + assert_eq!(chunk.column(1).unwrap().get_edge_id(0).unwrap().as_u64(), 3); + assert!(chunk.column(1).unwrap().get_node_id(0).is_none()); + } + + /// Two scans of one state (the branches of a UNION) both read the row, + /// and so does a rescan after `reset` without new values. + #[test] + fn every_scan_of_the_state_reads_the_values() { + let state = Arc::new(ParameterState::new(vec!["x".to_string()])); + let mut first = ParameterScanOperator::new(Arc::clone(&state)); + let mut second = ParameterScanOperator::new(Arc::clone(&state)); + state.set_values(vec![Value::Int64(7)]); + for op in [&mut first, &mut second] { + let chunk = op.next().unwrap().expect("each scan emits the row"); + assert_eq!(chunk.column(0).unwrap().get_value(0), Some(Value::Int64(7))); + assert!(op.next().unwrap().is_none()); + } + first.reset(); + let chunk = first.next().unwrap().expect("a rescan emits the row again"); + assert_eq!(chunk.column(0).unwrap().get_value(0), Some(Value::Int64(7))); + } + #[test] fn test_parameter_scan_no_values() { let state = Arc::new(ParameterState::new(vec!["x".to_string()])); diff --git a/crates/grafeo-core/src/execution/operators/project.rs b/crates/grafeo-core/src/execution/operators/project.rs index c4d5b5b62..b26e2bc7c 100644 --- a/crates/grafeo-core/src/execution/operators/project.rs +++ b/crates/grafeo-core/src/execution/operators/project.rs @@ -2,7 +2,7 @@ use super::filter::{ExpressionPredicate, FilterExpression, SessionContext}; use super::{Operator, OperatorError, OperatorResult}; -use crate::execution::DataChunk; +use crate::execution::{DataChunk, ValueVector}; use crate::graph::GraphStoreSearch; use crate::graph::lpg::{Edge, Node}; use grafeo_common::types::{ @@ -203,6 +203,17 @@ impl Operator for ProjectOperator { .column_mut(i) .expect("column exists: index matches projection schema"); + // A copy the planner declared `Any` keeps the input's + // type: node and edge IDs stay nodes and edges (in an + // `Any` column an ID reads the properties of whichever + // entity has it), and every value is copied as it is. + if self.output_types[i] == LogicalType::Any { + *output_col = ValueVector::with_capacity( + input_col.data_type().clone(), + input.row_count(), + ); + } + // Copy selected rows for row in input.selected_indices() { if let Some(value) = input_col.get_value(row) { @@ -342,12 +353,20 @@ impl Operator for ProjectOperator { evaluator = evaluator.with_transaction_context(ep, tx_id); } + // A node or edge column holds ids: it takes a node or edge + // map (the items of `nodes(p)`) by its id, and anything + // that names no entity as null, never as entity 0. + let entities = matches!( + output_col.data_type(), + LogicalType::Node | LogicalType::Edge + ); for row in input.selected_indices() { let value = evaluator.eval_at(&input, row).unwrap_or(Value::Null); + let value = if entities { entity_id(value) } else { value }; output_col.push_value(value); } } - ProjectExpr::NodeResolve { column } => { + ProjectExpr::NodeResolve { column } | ProjectExpr::EdgeResolve { column } => { let input_col = input .column(*column) .ok_or_else(|| OperatorError::ColumnNotFound(format!("Column {column}")))?; @@ -357,54 +376,43 @@ impl Operator for ProjectOperator { .expect("column exists: index matches projection schema"); let store = self.store.as_ref().ok_or_else(|| { - OperatorError::Execution("Store required for node resolution".to_string()) + OperatorError::Execution("Store required for entity resolution".to_string()) })?; + // The planner says by name whether the column holds nodes + // or edges; a column typed by its rows says it per chunk, + // which wins: the branches of a set operation may bind one + // name to nodes in one branch and to edges in another. + let edges = match input_col.data_type() { + LogicalType::Edge => true, + LogicalType::Node => false, + _ => matches!(proj, ProjectExpr::EdgeResolve { .. }), + }; let epoch = self.viewing_epoch; let tx_id = self.transaction_id; for row in input.selected_indices() { - let value = if let Some(node_id) = input_col.get_node_id(row) { - let node = if let (Some(ep), Some(tx)) = (epoch, tx_id) { - store.get_node_versioned(node_id, ep, tx) - } else if let Some(ep) = epoch { - store.get_node_at_epoch(node_id, ep) - } else { - store.get_node(node_id) - }; - node.map_or(Value::Null, |n| node_to_map(&n)) - } else { - Value::Null - }; - output_col.push_value(value); - } - } - ProjectExpr::EdgeResolve { column } => { - let input_col = input - .column(*column) - .ok_or_else(|| OperatorError::ColumnNotFound(format!("Column {column}")))?; - - let output_col = output - .column_mut(i) - .expect("column exists: index matches projection schema"); - - let store = self.store.as_ref().ok_or_else(|| { - OperatorError::Execution("Store required for edge resolution".to_string()) - })?; - - let epoch = self.viewing_epoch; - let tx_id = self.transaction_id; - for row in input.selected_indices() { - let value = if let Some(edge_id) = input_col.get_edge_id(row) { - let edge = if let (Some(ep), Some(tx)) = (epoch, tx_id) { - store.get_edge_versioned(edge_id, ep, tx) - } else if let Some(ep) = epoch { - store.get_edge_at_epoch(edge_id, ep) - } else { - store.get_edge(edge_id) - }; - edge.map_or(Value::Null, |e| edge_to_map(&e)) + let value = if edges { + input_col.get_edge_id(row).map_or(Value::Null, |edge_id| { + let edge = if let (Some(ep), Some(tx)) = (epoch, tx_id) { + store.get_edge_versioned(edge_id, ep, tx) + } else if let Some(ep) = epoch { + store.get_edge_at_epoch(edge_id, ep) + } else { + store.get_edge(edge_id) + }; + edge.map_or(Value::Null, |e| edge_to_map(&e)) + }) } else { - Value::Null + input_col.get_node_id(row).map_or(Value::Null, |node_id| { + let node = if let (Some(ep), Some(tx)) = (epoch, tx_id) { + store.get_node_versioned(node_id, ep, tx) + } else if let Some(ep) = epoch { + store.get_node_at_epoch(node_id, ep) + } else { + store.get_node(node_id) + }; + node.map_or(Value::Null, |n| node_to_map(&n)) + }) }; output_col.push_value(value); } @@ -587,6 +595,19 @@ fn edge_to_map(edge: &Edge) -> Value { Value::Map(Arc::new(map)) } +/// The id a value gives a node or edge column: an id, the `_id` of an entity +/// map, or null. +fn entity_id(value: Value) -> Value { + match value { + Value::Int64(_) | Value::Null => value, + Value::Map(map) => match map.get(&PropertyKey::new("_id")) { + Some(id @ Value::Int64(_)) => id.clone(), + _ => Value::Null, + }, + _ => Value::Null, + } +} + #[cfg(all(test, feature = "lpg"))] mod tests { use super::*; @@ -663,6 +684,49 @@ mod tests { assert_eq!(result.column(1).unwrap().get_int64(0), Some(1)); } + /// A copy declared `Any` keeps the input column's type, so node IDs stay + /// node IDs (in an `Any` column an ID reads whichever entity has it); a + /// declared type still applies. + #[test] + fn a_copy_declared_any_keeps_the_input_type() { + let mut builder = DataChunkBuilder::new(&[LogicalType::Node, LogicalType::Int64]); + for id in 1..=3_u64 { + builder + .column_mut(0) + .unwrap() + .push_node_id(grafeo_common::types::NodeId::new(id)); + builder + .column_mut(1) + .unwrap() + .push_int64(i64::try_from(id).unwrap() * 10); + builder.advance_row(); + } + let mock_scan = MockScanOperator { + chunks: vec![builder.finish()], + position: 0, + }; + let mut project = ProjectOperator::new( + Box::new(mock_scan), + vec![ + ProjectExpr::Column(0), + ProjectExpr::Column(1), + ProjectExpr::Column(1), + ], + vec![LogicalType::Any, LogicalType::Any, LogicalType::Float64], + ); + + let result = project.next().unwrap().unwrap(); + let ids = result.column(0).unwrap(); + assert_eq!(ids.data_type(), &LogicalType::Node); + assert_eq!( + ids.get_node_id(2), + Some(grafeo_common::types::NodeId::new(3)) + ); + assert_eq!(result.column(1).unwrap().data_type(), &LogicalType::Int64); + assert_eq!(result.column(1).unwrap().get_int64(0), Some(10)); + assert_eq!(result.column(2).unwrap().data_type(), &LogicalType::Float64); + } + #[test] fn test_project_constant() { let mut builder = DataChunkBuilder::new(&[LogicalType::Int64]); @@ -799,6 +863,49 @@ mod tests { assert_eq!(project.name(), "Project"); } + /// A node or edge column takes an entity map by its `_id`; a value that + /// names no entity is null there, never entity 0. + #[test] + fn test_project_entity_values_into_entity_columns() { + let mut builder = DataChunkBuilder::new(&[LogicalType::Int64]); + builder.column_mut(0).unwrap().push_int64(1); + builder.advance_row(); + let mock_scan = MockScanOperator { + chunks: vec![builder.finish()], + position: 0, + }; + let map = |entries: &[(&str, Value)]| { + Value::Map(Arc::new( + entries + .iter() + .map(|(key, value)| (PropertyKey::new(*key), value.clone())) + .collect(), + )) + }; + let expression = |value: Value| ProjectExpr::Expression { + expr: FilterExpression::Literal(value), + variable_columns: HashMap::new(), + }; + let mut project = ProjectOperator::with_store( + Box::new(mock_scan), + vec![ + expression(map(&[("_id", Value::Int64(7))])), + expression(map(&[("name", Value::from("Alix"))])), + expression(Value::from("Alix")), + ], + vec![LogicalType::Node, LogicalType::Node, LogicalType::Edge], + Arc::new(LpgStore::new().unwrap()), + ); + + let result = project.next().unwrap().unwrap(); + assert_eq!( + result.column(0).unwrap().get_node_id(0), + Some(NodeId::new(7)) + ); + assert_eq!(result.column(1).unwrap().get_node_id(0), None); + assert_eq!(result.column(2).unwrap().get_edge_id(0), None); + } + #[test] // reason: test IDs are small sequential counters #[allow(clippy::cast_possible_wrap)] diff --git a/crates/grafeo-core/src/execution/operators/push/aggregate.rs b/crates/grafeo-core/src/execution/operators/push/aggregate.rs index 41695ad74..7cbf9bd4e 100644 --- a/crates/grafeo-core/src/execution/operators/push/aggregate.rs +++ b/crates/grafeo-core/src/execution/operators/push/aggregate.rs @@ -1,6 +1,6 @@ //! Push-based aggregate operator (pipeline breaker). -use crate::execution::chunk::DataChunk; +use crate::execution::chunk::{ColumnTypes, DataChunk}; use crate::execution::operators::OperatorError; use crate::execution::operators::accumulator::{AggregateExpr, AggregateFunction, AggregateState}; use crate::execution::pipeline::{ChunkSizeHint, PushOperator, Sink}; @@ -191,6 +191,23 @@ fn hash_value(value: &Value) -> u64 { hasher.finish() } +/// The output columns: the group keys in their input columns' types (a node +/// or edge stays one, see `ColumnTypes`), then one per aggregate. +fn output_columns( + group_by: &[usize], + aggregates: usize, + input_types: &ColumnTypes, +) -> Vec { + group_by + .iter() + .map(|&column| match input_types.types().get(column) { + Some(column_type) => ValueVector::with_capacity(column_type.clone(), 0), + None => ValueVector::new(), + }) + .chain((0..aggregates).map(|_| ValueVector::new())) + .collect() +} + /// Group state with key values and accumulators. #[derive(Clone)] struct GroupState { @@ -211,6 +228,8 @@ pub struct AggregatePushOperator { groups: HashMap, /// Global accumulator (for no GROUP BY). global_state: Option>, + /// The column types of the input chunks. + input_types: ColumnTypes, } impl AggregatePushOperator { @@ -227,6 +246,7 @@ impl AggregatePushOperator { aggregates, groups: HashMap::new(), global_state, + input_types: ColumnTypes::default(), } } @@ -241,6 +261,7 @@ impl PushOperator for AggregatePushOperator { if chunk.is_empty() { return Ok(true); } + self.input_types.add(&chunk); for row in chunk.selected_indices() { if self.group_by.is_empty() { @@ -282,9 +303,7 @@ impl PushOperator for AggregatePushOperator { } fn finalize(&mut self, sink: &mut dyn Sink) -> Result<(), OperatorError> { - let num_output_cols = self.group_by.len() + self.aggregates.len(); - let mut columns: Vec = - (0..num_output_cols).map(|_| ValueVector::new()).collect(); + let mut columns = output_columns(&self.group_by, self.aggregates.len(), &self.input_types); if self.group_by.is_empty() { // Global aggregation - single row output @@ -603,6 +622,8 @@ pub struct SpillableAggregatePushOperator { spill_state: Option>, /// Running total of estimated group memory in bytes (incremental tracking). estimated_bytes: usize, + /// The column types of the input chunks. + input_types: ColumnTypes, } #[cfg(feature = "spill")] @@ -627,6 +648,7 @@ impl SpillableAggregatePushOperator { memory_ctx: None, spill_state: None, estimated_bytes: 0, + input_types: ColumnTypes::default(), } } @@ -662,6 +684,7 @@ impl SpillableAggregatePushOperator { memory_ctx: None, spill_state: None, estimated_bytes: 0, + input_types: ColumnTypes::default(), } } @@ -709,6 +732,7 @@ impl SpillableAggregatePushOperator { memory_ctx: Some(ctx), spill_state: Some(state), estimated_bytes: 0, + input_types: ColumnTypes::default(), } } @@ -816,6 +840,7 @@ impl PushOperator for SpillableAggregatePushOperator { if chunk.is_empty() { return Ok(true); } + self.input_types.add(&chunk); for row in chunk.selected_indices() { if self.group_by.is_empty() { @@ -903,9 +928,7 @@ impl PushOperator for SpillableAggregatePushOperator { } fn finalize(&mut self, sink: &mut dyn Sink) -> Result<(), OperatorError> { - let num_output_cols = self.group_by.len() + self.aggregates.len(); - let mut columns: Vec = - (0..num_output_cols).map(|_| ValueVector::new()).collect(); + let mut columns = output_columns(&self.group_by, self.aggregates.len(), &self.input_types); if self.group_by.is_empty() { // Global aggregation - single row output @@ -989,6 +1012,45 @@ mod tests { ]) } + /// A node group key keeps its column type. + #[test] + fn group_keys_keep_their_column_types() { + use crate::execution::chunk::DataChunkBuilder; + use grafeo_common::types::{LogicalType, NodeId}; + + let mut builder = DataChunkBuilder::new(&[LogicalType::Node]); + for node in [101, 101, 102] { + builder + .column_mut(0) + .unwrap() + .push_node_id(NodeId::new(node)); + builder.advance_row(); + } + let mut agg = AggregatePushOperator::new(vec![0], vec![AggregateExpr::count_star()]); + let mut sink = CollectorSink::new(); + agg.push(builder.finish(), &mut sink).unwrap(); + agg.finalize(&mut sink).unwrap(); + + let chunks = sink.into_chunks(); + assert_eq!(chunks.len(), 1); + let chunk = &chunks[0]; + assert_eq!(chunk.column_types()[0], LogicalType::Node); + let mut groups: Vec<(u64, Option)> = chunk + .selected_indices() + .map(|row| { + ( + chunk.column(0).unwrap().get_node_id(row).unwrap().as_u64(), + chunk.column(1).unwrap().get_value(row), + ) + }) + .collect(); + groups.sort_by_key(|(node, _)| *node); + assert_eq!( + groups, + [(101, Some(Value::Int64(2))), (102, Some(Value::Int64(1)))] + ); + } + #[test] fn test_global_count() { let mut agg = AggregatePushOperator::global(vec![AggregateExpr::count_star()]); diff --git a/crates/grafeo-core/src/execution/operators/push/sort.rs b/crates/grafeo-core/src/execution/operators/push/sort.rs index 10ede1707..f5e5b8865 100644 --- a/crates/grafeo-core/src/execution/operators/push/sort.rs +++ b/crates/grafeo-core/src/execution/operators/push/sort.rs @@ -1,13 +1,12 @@ //! Push-based sort operator (pipeline breaker). -use crate::execution::chunk::DataChunk; +use crate::execution::chunk::{ColumnTypes, DataChunk, DataChunkBuilder}; use crate::execution::operators::OperatorError; -use crate::execution::operators::value_utils::compare_values_total; +use crate::execution::operators::value_utils::order_by; use crate::execution::pipeline::{ChunkSizeHint, PushOperator, Sink}; #[cfg(feature = "spill")] use crate::execution::spill::{ExternalSort, SpillManager}; -use crate::execution::vector::ValueVector; -use grafeo_common::types::Value; +use grafeo_common::types::{LogicalType, Value}; use std::cmp::Ordering; #[cfg(feature = "spill")] use std::sync::Arc; @@ -22,7 +21,7 @@ pub enum SortDirection { Descending, } -/// Null handling in sort. +/// Where nulls go, in either sort direction. #[derive(Debug, Clone, Copy, PartialEq, Eq)] #[non_exhaustive] pub enum NullOrder { @@ -74,6 +73,10 @@ pub struct SortPushOperator { buffer: Vec>, /// Number of columns per row. num_columns: Option, + /// How many leading columns each output row keeps (all when `None`). + output_width: Option, + /// The input's column types: a node or edge column stays one. + column_types: ColumnTypes, } impl SortPushOperator { @@ -83,6 +86,8 @@ impl SortPushOperator { keys, buffer: Vec::new(), num_columns: None, + output_width: None, + column_types: ColumnTypes::default(), } } @@ -95,33 +100,26 @@ impl SortPushOperator { pub fn descending(column: usize) -> Self { Self::new(vec![SortKey::descending(column)]) } + + /// Returns only the first `width` columns of each row: the columns after + /// them hold sort keys the planner added to sort by (see + /// [`SortOperator::with_output_width`](crate::execution::operators::SortOperator::with_output_width)). + #[must_use] + pub fn with_output_width(mut self, width: usize) -> Self { + self.output_width = Some(width); + self + } } /// Compare two rows by sort keys. fn compare_rows(a: &[Value], b: &[Value], keys: &[SortKey]) -> Ordering { for key in keys { - let a_val = a.get(key.column); - let b_val = b.get(key.column); - - let ordering = match (a_val, b_val) { - (Some(Value::Null), Some(Value::Null)) => Ordering::Equal, - (Some(Value::Null), _) => match key.null_order { - NullOrder::First => Ordering::Less, - NullOrder::Last => Ordering::Greater, - }, - (_, Some(Value::Null)) => match key.null_order { - NullOrder::First => Ordering::Greater, - NullOrder::Last => Ordering::Less, - }, - (Some(a), Some(b)) => compare_values_total(a, b), - _ => Ordering::Equal, - }; - - let ordering = match key.direction { - SortDirection::Ascending => ordering, - SortDirection::Descending => ordering.reverse(), - }; - + let ordering = order_by( + a.get(key.column), + b.get(key.column), + key.direction == SortDirection::Descending, + key.null_order == NullOrder::First, + ); if ordering != Ordering::Equal { return ordering; } @@ -130,6 +128,36 @@ fn compare_rows(a: &[Value], b: &[Value], keys: &[SortKey]) -> Ordering { Ordering::Equal } +/// How many columns the output rows have: `width` when the sort drops +/// trailing sort-key columns, otherwise all `num_cols`. +fn output_width(num_cols: usize, width: Option) -> usize { + width.map_or(num_cols, |width| width.min(num_cols)) +} + +/// One chunk with the first `width` values of each sorted row, in columns of +/// the input's types (a node or edge column stays one, see `ColumnTypes`). +fn output_chunk(rows: &[Vec], column_types: &ColumnTypes, width: usize) -> DataChunk { + let types: Vec = (0..width) + .map(|column| { + column_types + .types() + .get(column) + .cloned() + .unwrap_or(LogicalType::Any) + }) + .collect(); + let mut builder = DataChunkBuilder::with_capacity(&types, rows.len()); + for row in rows { + for column in 0..width { + if let Some(out) = builder.column_mut(column) { + out.push_value(row.get(column).cloned().unwrap_or(Value::Null)); + } + } + builder.advance_row(); + } + builder.finish() +} + impl PushOperator for SortPushOperator { fn push(&mut self, chunk: DataChunk, _sink: &mut dyn Sink) -> Result { if chunk.is_empty() { @@ -140,6 +168,7 @@ impl PushOperator for SortPushOperator { if self.num_columns.is_none() { self.num_columns = Some(chunk.column_count()); } + self.column_types.add(&chunk); let num_cols = chunk.column_count(); @@ -168,23 +197,15 @@ impl PushOperator for SortPushOperator { let keys = &self.keys; self.buffer.sort_by(|a, b| compare_rows(a, b, keys)); - // Emit sorted rows in chunks let num_cols = self.num_columns.unwrap_or(0); if num_cols == 0 { return Ok(()); } - - // Build output chunk from sorted rows - let mut columns: Vec = (0..num_cols).map(|_| ValueVector::new()).collect(); - - for row in &self.buffer { - for (col_idx, col) in columns.iter_mut().enumerate() { - let val = row.get(col_idx).cloned().unwrap_or(Value::Null); - col.push(val); - } - } - - let chunk = DataChunk::new(columns); + let chunk = output_chunk( + &self.buffer, + &self.column_types, + output_width(num_cols, self.output_width), + ); sink.consume(chunk)?; Ok(()) @@ -233,6 +254,10 @@ pub struct SpillableSortPushOperator { buffer: Vec>, /// Number of columns per row. num_columns: Option, + /// How many leading columns each output row keeps (all when `None`). + output_width: Option, + /// The input's column types: a node or edge column stays one. + column_types: ColumnTypes, /// Spill manager for file creation (used by row-count fallback mode). spill_manager: Option>, /// External sort state (created when first spill occurs). @@ -257,6 +282,8 @@ impl SpillableSortPushOperator { keys, buffer: Vec::new(), num_columns: None, + output_width: None, + column_types: ColumnTypes::default(), spill_manager: None, external_sort: None, spill_threshold: DEFAULT_SPILL_THRESHOLD, @@ -272,6 +299,8 @@ impl SpillableSortPushOperator { keys, buffer: Vec::new(), num_columns: None, + output_width: None, + column_types: ColumnTypes::default(), spill_manager: Some(manager), external_sort: None, spill_threshold: threshold, @@ -300,6 +329,8 @@ impl SpillableSortPushOperator { keys, buffer: Vec::new(), num_columns: None, + output_width: None, + column_types: ColumnTypes::default(), spill_manager: None, external_sort: None, spill_threshold: DEFAULT_SPILL_THRESHOLD, @@ -333,6 +364,15 @@ impl SpillableSortPushOperator { self } + /// Returns only the first `width` columns of each row: the columns after + /// them hold sort keys the planner added to sort by (see + /// [`SortOperator::with_output_width`](crate::execution::operators::SortOperator::with_output_width)). + #[must_use] + pub fn with_output_width(mut self, width: usize) -> Self { + self.output_width = Some(width); + self + } + /// Checks whether spilling should occur and performs it if needed. fn maybe_spill(&mut self) -> Result<(), OperatorError> { let should_spill = if let Some(ref state) = self.spill_state { @@ -427,6 +467,7 @@ impl PushOperator for SpillableSortPushOperator { if self.num_columns.is_none() { self.num_columns = Some(chunk.column_count()); } + self.column_types.add(&chunk); let num_cols = chunk.column_count(); @@ -483,18 +524,11 @@ impl PushOperator for SpillableSortPushOperator { if sorted_rows.is_empty() { return Ok(()); } - - // Build output chunk from sorted rows - let mut columns: Vec = (0..num_cols).map(|_| ValueVector::new()).collect(); - - for row in &sorted_rows { - for (col_idx, col) in columns.iter_mut().enumerate() { - let val = row.get(col_idx).cloned().unwrap_or(Value::Null); - col.push(val); - } - } - - let chunk = DataChunk::new(columns); + let chunk = output_chunk( + &sorted_rows, + &self.column_types, + output_width(num_cols, self.output_width), + ); sink.consume(chunk)?; Ok(()) @@ -514,6 +548,7 @@ impl PushOperator for SpillableSortPushOperator { mod tests { use super::*; use crate::execution::sink::CollectorSink; + use crate::execution::vector::ValueVector; fn create_test_chunk(values: &[i64]) -> DataChunk { let v: Vec = values.iter().map(|&i| Value::Int64(i)).collect(); @@ -556,6 +591,167 @@ mod tests { assert_eq!(col.get_value(2), Some(Value::Int64(3))); } + /// Two chunks of mixed values with nulls. + fn mixed_chunks() -> Vec { + let first = [ + Value::Int64(3), + Value::String("a".into()), + Value::Null, + Value::Float64(2.5), + ]; + let second = [ + Value::Bool(true), + Value::Null, + Value::Int64(1), + Value::List(vec![Value::Int64(1)].into()), + ]; + vec![ + DataChunk::new(vec![ValueVector::from_values(&first)]), + DataChunk::new(vec![ValueVector::from_values(&second)]), + ] + } + + /// `ORDER BY x DESC NULLS LAST` over [`mixed_chunks`]. + const MIXED_DESC_NULLS_LAST: [&str; 8] = + ["3", "2.5", "1", "true", "\"a\"", "[1]", "NULL", "NULL"]; + + fn first_column_texts(sink: CollectorSink) -> Vec { + sink.into_chunks() + .iter() + .flat_map(|chunk| { + let column = chunk.column(0).unwrap(); + (0..chunk.len()) + .map(|row| column.get_value(row).unwrap().to_string()) + .collect::>() + }) + .collect() + } + + /// Values of different types sort in one order, and `NULLS LAST` holds + /// when descending. + #[test] + fn mixed_values_sort_in_one_order_with_nulls_last_descending() { + let key = SortKey { + column: 0, + direction: SortDirection::Descending, + null_order: NullOrder::Last, + }; + let mut sort = SortPushOperator::new(vec![key]); + let mut sink = CollectorSink::new(); + for chunk in mixed_chunks() { + sort.push(chunk, &mut sink).unwrap(); + } + sort.finalize(&mut sink).unwrap(); + + assert_eq!(first_column_texts(sink), MIXED_DESC_NULLS_LAST); + } + + /// Nodes 7, 8 and 9 with the sort keys 3, 1 and 2, in two chunks. + fn nodes_with_keys() -> Vec { + use crate::execution::chunk::DataChunkBuilder; + use grafeo_common::types::{LogicalType, NodeId}; + + [vec![(7_u64, 3_i64), (8, 1)], vec![(9, 2)]] + .into_iter() + .map(|rows| { + let mut builder = DataChunkBuilder::new(&[LogicalType::Node, LogicalType::Int64]); + for (id, key) in rows { + builder.column_mut(0).unwrap().push_node_id(NodeId::new(id)); + builder.column_mut(1).unwrap().push_int64(key); + builder.advance_row(); + } + builder.finish() + }) + .collect() + } + + /// The node IDs in `sink`, after checking that each chunk holds one + /// column and that it is a node column. + fn node_ids(sink: CollectorSink) -> Vec { + use grafeo_common::types::LogicalType; + + let mut ids = Vec::new(); + for chunk in sink.into_chunks() { + assert_eq!(chunk.column_count(), 1); + let nodes = chunk.column(0).unwrap(); + assert_eq!(nodes.data_type(), &LogicalType::Node); + ids.extend( + chunk + .selected_indices() + .map(|row| nodes.get_node_id(row).unwrap().as_u64()), + ); + } + ids + } + + /// With an output width the rows keep only their leading columns, with + /// the types they came in with: the columns after them were sort keys. + #[test] + fn output_width_drops_the_sort_key_columns() { + let mut sort = SortPushOperator::new(vec![SortKey::ascending(1)]).with_output_width(1); + let mut sink = CollectorSink::new(); + for chunk in nodes_with_keys() { + sort.push(chunk, &mut sink).unwrap(); + } + sort.finalize(&mut sink).unwrap(); + + assert_eq!(node_ids(sink), [8, 9, 7]); + } + + /// The same holds for the spillable sort, in memory and after its spilled + /// runs are merged. + #[test] + #[cfg(feature = "spill")] + fn spillable_sort_keeps_the_output_width_and_types() { + use tempfile::TempDir; + + let mut in_memory = + SpillableSortPushOperator::new(vec![SortKey::ascending(1)]).with_output_width(1); + let mut sink = CollectorSink::new(); + for chunk in nodes_with_keys() { + in_memory.push(chunk, &mut sink).unwrap(); + } + in_memory.finalize(&mut sink).unwrap(); + assert_eq!(node_ids(sink), [8, 9, 7]); + + let temp_dir = TempDir::new().unwrap(); + let manager = Arc::new(SpillManager::new(temp_dir.path()).unwrap()); + // A threshold of one row spills every chunk. + let mut spilled = + SpillableSortPushOperator::with_spilling(vec![SortKey::ascending(1)], manager, 1) + .with_output_width(1); + let mut sink = CollectorSink::new(); + for chunk in nodes_with_keys() { + spilled.push(chunk, &mut sink).unwrap(); + } + assert!(spilled.external_sort.is_some(), "the rows were spilled"); + spilled.finalize(&mut sink).unwrap(); + assert_eq!(node_ids(sink), [8, 9, 7]); + } + + /// Spilled runs merge in the same order as an in-memory sort. + #[test] + #[cfg(feature = "spill")] + fn spilled_runs_merge_mixed_values_in_the_same_order() { + use tempfile::TempDir; + + let temp_dir = TempDir::new().unwrap(); + let manager = Arc::new(SpillManager::new(temp_dir.path()).unwrap()); + let key = SortKey { + column: 0, + direction: SortDirection::Descending, + null_order: NullOrder::Last, + }; + let mut sort = SpillableSortPushOperator::with_spilling(vec![key], manager, 3); + let mut sink = CollectorSink::new(); + for chunk in mixed_chunks() { + sort.push(chunk, &mut sink).unwrap(); + } + sort.finalize(&mut sink).unwrap(); + + assert_eq!(first_column_texts(sink), MIXED_DESC_NULLS_LAST); + } + #[test] fn test_sort_multiple_chunks() { let mut sort = SortPushOperator::ascending(0); diff --git a/crates/grafeo-core/src/execution/operators/set_ops.rs b/crates/grafeo-core/src/execution/operators/set_ops.rs index 69a5f7bb2..1a2439f7c 100644 --- a/crates/grafeo-core/src/execution/operators/set_ops.rs +++ b/crates/grafeo-core/src/execution/operators/set_ops.rs @@ -8,7 +8,7 @@ use std::collections::HashSet; use grafeo_common::types::{HashableValue, LogicalType, Value}; use super::{DataChunk, Operator, OperatorError, OperatorResult}; -use crate::execution::chunk::DataChunkBuilder; +use crate::execution::chunk::{ColumnTypes, DataChunkBuilder}; /// A hashable row key: one `HashableValue` per column. type RowKey = Vec; @@ -31,18 +31,22 @@ fn row_values(key: &RowKey) -> Vec { key.iter().map(|hv| hv.0.clone()).collect() } -/// Materializes all rows from an operator into a vector of row keys. -fn materialize(op: &mut dyn Operator) -> Result, OperatorError> { +/// Materializes all rows from an operator into a vector of row keys, with +/// the column types of its chunks. +fn materialize(op: &mut dyn Operator) -> Result<(Vec, Vec), OperatorError> { let mut rows = Vec::new(); + let mut column_types = ColumnTypes::default(); while let Some(chunk) = op.next()? { + column_types.add(&chunk); for row in chunk.selected_indices() { rows.push(row_key(&chunk, row)); } } - Ok(rows) + Ok((rows, column_types.types().to_vec())) } -/// Rebuilds a `DataChunk` from a set of row keys. +/// Rebuilds a `DataChunk` from a set of row keys, in columns of the types +/// the rows came in. fn rows_to_chunk(rows: &[RowKey], schema: &[LogicalType]) -> DataChunk { if rows.is_empty() { return DataChunk::empty(); @@ -65,32 +69,29 @@ pub struct ExceptOperator { left: Box, right: Box, all: bool, - output_schema: Vec, + /// The column types of the left input, which the result rows come from. + column_types: Vec, result: Option>, position: usize, } impl ExceptOperator { /// Creates a new EXCEPT operator. - pub fn new( - left: Box, - right: Box, - all: bool, - output_schema: Vec, - ) -> Self { + pub fn new(left: Box, right: Box, all: bool) -> Self { Self { left, right, all, - output_schema, + column_types: Vec::new(), result: None, position: 0, } } fn compute(&mut self) -> Result<(), OperatorError> { - let left_rows = materialize(self.left.as_mut())?; - let right_rows = materialize(self.right.as_mut())?; + let (left_rows, column_types) = materialize(self.left.as_mut())?; + let (right_rows, _) = materialize(self.right.as_mut())?; + self.column_types = column_types; if self.all { // EXCEPT ALL: for each right row, remove one matching left row @@ -134,7 +135,7 @@ impl Operator for ExceptOperator { if batch.is_empty() { Ok(None) } else { - Ok(Some(rows_to_chunk(batch, &self.output_schema))) + Ok(Some(rows_to_chunk(batch, &self.column_types))) } } @@ -159,32 +160,29 @@ pub struct IntersectOperator { left: Box, right: Box, all: bool, - output_schema: Vec, + /// The column types of the left input, which the result rows come from. + column_types: Vec, result: Option>, position: usize, } impl IntersectOperator { /// Creates a new INTERSECT operator. - pub fn new( - left: Box, - right: Box, - all: bool, - output_schema: Vec, - ) -> Self { + pub fn new(left: Box, right: Box, all: bool) -> Self { Self { left, right, all, - output_schema, + column_types: Vec::new(), result: None, position: 0, } } fn compute(&mut self) -> Result<(), OperatorError> { - let left_rows = materialize(self.left.as_mut())?; - let right_rows = materialize(self.right.as_mut())?; + let (left_rows, column_types) = materialize(self.left.as_mut())?; + let (right_rows, _) = materialize(self.right.as_mut())?; + self.column_types = column_types; if self.all { // INTERSECT ALL: each right row matches at most one left row @@ -229,7 +227,7 @@ impl Operator for IntersectOperator { if batch.is_empty() { Ok(None) } else { - Ok(Some(rows_to_chunk(batch, &self.output_schema))) + Ok(Some(rows_to_chunk(batch, &self.column_types))) } } @@ -400,12 +398,7 @@ mod tests { fn test_except_distinct() { let left = MockOperator::new(vec![create_int_chunk(&[1, 2, 3, 2])]); let right = MockOperator::new(vec![create_int_chunk(&[2, 4])]); - let mut op = ExceptOperator::new( - Box::new(left), - Box::new(right), - false, - vec![LogicalType::Int64], - ); + let mut op = ExceptOperator::new(Box::new(left), Box::new(right), false); let mut result = collect_ints(&mut op); result.sort_unstable(); @@ -416,12 +409,7 @@ mod tests { fn test_except_all() { let left = MockOperator::new(vec![create_int_chunk(&[1, 2, 2, 3])]); let right = MockOperator::new(vec![create_int_chunk(&[2])]); - let mut op = ExceptOperator::new( - Box::new(left), - Box::new(right), - true, - vec![LogicalType::Int64], - ); + let mut op = ExceptOperator::new(Box::new(left), Box::new(right), true); let mut result = collect_ints(&mut op); result.sort_unstable(); @@ -433,12 +421,7 @@ mod tests { fn test_except_empty_right() { let left = MockOperator::new(vec![create_int_chunk(&[1, 2])]); let right = MockOperator::new(vec![]); - let mut op = ExceptOperator::new( - Box::new(left), - Box::new(right), - false, - vec![LogicalType::Int64], - ); + let mut op = ExceptOperator::new(Box::new(left), Box::new(right), false); let mut result = collect_ints(&mut op); result.sort_unstable(); @@ -449,12 +432,7 @@ mod tests { fn test_intersect_distinct() { let left = MockOperator::new(vec![create_int_chunk(&[1, 2, 3, 2])]); let right = MockOperator::new(vec![create_int_chunk(&[2, 3, 4])]); - let mut op = IntersectOperator::new( - Box::new(left), - Box::new(right), - false, - vec![LogicalType::Int64], - ); + let mut op = IntersectOperator::new(Box::new(left), Box::new(right), false); let mut result = collect_ints(&mut op); result.sort_unstable(); @@ -465,12 +443,7 @@ mod tests { fn test_intersect_all() { let left = MockOperator::new(vec![create_int_chunk(&[1, 2, 2, 3])]); let right = MockOperator::new(vec![create_int_chunk(&[2, 2, 4])]); - let mut op = IntersectOperator::new( - Box::new(left), - Box::new(right), - true, - vec![LogicalType::Int64], - ); + let mut op = IntersectOperator::new(Box::new(left), Box::new(right), true); let mut result = collect_ints(&mut op); result.sort_unstable(); @@ -481,12 +454,7 @@ mod tests { fn test_intersect_no_overlap() { let left = MockOperator::new(vec![create_int_chunk(&[1, 2])]); let right = MockOperator::new(vec![create_int_chunk(&[3, 4])]); - let mut op = IntersectOperator::new( - Box::new(left), - Box::new(right), - false, - vec![LogicalType::Int64], - ); + let mut op = IntersectOperator::new(Box::new(left), Box::new(right), false); let result = collect_ints(&mut op); assert!(result.is_empty()); @@ -526,10 +494,10 @@ mod tests { fn test_operator_names() { let empty = || MockOperator::new(vec![]); - let op = ExceptOperator::new(Box::new(empty()), Box::new(empty()), false, vec![]); + let op = ExceptOperator::new(Box::new(empty()), Box::new(empty()), false); assert_eq!(op.name(), "Except"); - let op = IntersectOperator::new(Box::new(empty()), Box::new(empty()), false, vec![]); + let op = IntersectOperator::new(Box::new(empty()), Box::new(empty()), false); assert_eq!(op.name(), "Intersect"); let op = OtherwiseOperator::new(Box::new(empty()), Box::new(empty())); @@ -544,7 +512,6 @@ mod tests { Box::new(empty()), Box::new(empty()), false, - vec![], )); assert!(op.into_any().downcast::().is_ok()); @@ -552,7 +519,6 @@ mod tests { Box::new(empty()), Box::new(empty()), false, - vec![], )); assert!(op.into_any().downcast::().is_ok()); diff --git a/crates/grafeo-core/src/execution/operators/shuffle.rs b/crates/grafeo-core/src/execution/operators/shuffle.rs index ff93169ad..9aa9e36f1 100644 --- a/crates/grafeo-core/src/execution/operators/shuffle.rs +++ b/crates/grafeo-core/src/execution/operators/shuffle.rs @@ -1,22 +1,26 @@ //! Shuffle operator: returns its input's rows in random order. -use grafeo_common::types::{LogicalType, Value}; +use grafeo_common::types::Value; use super::{Operator, OperatorError, OperatorResult}; use crate::execution::DataChunk; -use crate::execution::chunk::DataChunkBuilder; +use crate::execution::chunk::{ColumnTypes, DataChunkBuilder}; /// Returns its input's rows in random order. /// /// Planned at the root of a query without `ORDER BY` when the database's /// `shuffle_unordered` test option is on, so that tests find code relying on -/// a row order that is unspecified. +/// a row order that is unspecified. A stream shuffles each chunk on its own +/// (see [`per_chunk`](Self::per_chunk)). The rows keep their columns' types +/// and values. pub struct ShuffleOperator { /// Child operator. child: Box, - /// Output schema. - output_schema: Vec, - /// Materialized chunks. + /// Whether each input chunk is shuffled on its own. + per_chunk: bool, + /// The state of the random stream, seeded differently per operator. + state: u64, + /// Materialized chunks (the current one, per chunk). chunks: Vec, /// The rows as `(chunk, row)`, in the order they are returned. rows: Vec<(usize, usize)>, @@ -27,11 +31,13 @@ pub struct ShuffleOperator { } impl ShuffleOperator { - /// Creates a shuffle operator over `child`. - pub fn new(child: Box, output_schema: Vec) -> Self { + /// Creates a shuffle operator over `child` that returns all of its rows + /// in random order, after reading the whole input. + pub fn new(child: Box) -> Self { Self { child, - output_schema, + per_chunk: false, + state: random_seed(), chunks: Vec::new(), rows: Vec::new(), shuffled: false, @@ -39,25 +45,59 @@ impl ShuffleOperator { } } + /// Creates a shuffle operator over `child` that reads one input chunk at + /// a time and returns its rows in random order: memory stays at one + /// chunk, as a stream needs, and rows stay within their chunk. + pub fn per_chunk(child: Box) -> Self { + Self { + per_chunk: true, + ..Self::new(child) + } + } + /// Materializes the input and puts its rows in random order. fn shuffle(&mut self) -> Result<(), OperatorError> { while let Some(chunk) = self.child.next()? { - let chunk_index = self.chunks.len(); - for row in chunk.selected_indices() { - self.rows.push((chunk_index, row)); + self.take(chunk); + } + self.shuffle_rows(); + self.shuffled = true; + Ok(()) + } + + /// Reads input chunks until one has rows and puts them in random order; + /// `false` once the input is exhausted. + fn shuffle_next_chunk(&mut self) -> Result { + self.chunks.clear(); + self.rows.clear(); + self.position = 0; + while let Some(chunk) = self.child.next()? { + self.take(chunk); + if !self.rows.is_empty() { + self.shuffle_rows(); + return Ok(true); } - self.chunks.push(chunk); + self.chunks.clear(); } + Ok(false) + } - // Fisher-Yates, with a SplitMix64 stream seeded differently per run. - let mut state = random_seed(); + /// Keeps `chunk` and lists its rows. + fn take(&mut self, chunk: DataChunk) { + let chunk_index = self.chunks.len(); + for row in chunk.selected_indices() { + self.rows.push((chunk_index, row)); + } + self.chunks.push(chunk); + } + + /// Fisher-Yates over the listed rows, with a SplitMix64 stream. + fn shuffle_rows(&mut self) { for i in (1..self.rows.len()).rev() { let bound = u64::try_from(i + 1).unwrap_or(u64::MAX); - let j = usize::try_from(next_random(&mut state) % bound).unwrap_or(i); + let j = usize::try_from(next_random(&mut self.state) % bound).unwrap_or(i); self.rows.swap(i, j); } - self.shuffled = true; - Ok(()) } } @@ -79,14 +119,22 @@ fn next_random(state: &mut u64) -> u64 { impl Operator for ShuffleOperator { fn next(&mut self) -> OperatorResult { - if !self.shuffled { + if self.per_chunk { + if self.position >= self.rows.len() && !self.shuffle_next_chunk()? { + return Ok(None); + } + } else if !self.shuffled { self.shuffle()?; } if self.position >= self.rows.len() { return Ok(None); } - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, 2048); + let mut column_types = ColumnTypes::default(); + for chunk in &self.chunks { + column_types.add(chunk); + } + let mut builder = DataChunkBuilder::with_capacity(column_types.types(), 2048); while self.position < self.rows.len() && !builder.is_full() { let (chunk_index, row) = self.rows[self.position]; let source_chunk = &self.chunks[chunk_index]; @@ -105,6 +153,7 @@ impl Operator for ShuffleOperator { fn reset(&mut self) { self.child.reset(); + self.state = random_seed(); self.chunks.clear(); self.rows.clear(); self.shuffled = false; @@ -122,14 +171,22 @@ impl Operator for ShuffleOperator { #[cfg(test)] mod tests { + use std::sync::Arc; + use std::sync::atomic::{AtomicUsize, Ordering}; + use super::*; + use grafeo_common::types::LogicalType; - /// Returns its chunks once. - struct Chunks(Vec); + /// Returns its chunks once, counting how many it handed out. + struct Chunks(Vec, Arc); impl Operator for Chunks { fn next(&mut self) -> OperatorResult { - Ok(self.0.pop()) + let chunk = self.0.pop(); + if chunk.is_some() { + self.1.fetch_add(1, Ordering::Relaxed); + } + Ok(chunk) } fn reset(&mut self) {} @@ -143,8 +200,9 @@ mod tests { } } - /// The values 0 to 99, in three chunks, returned by a shuffle. - fn shuffled() -> Vec { + /// The values 0 to 99 as three input chunks (0..40 first), and the + /// counter of chunks handed out. + fn input() -> (Chunks, Arc) { let chunks = [0..40, 40..70, 70..100] .into_iter() .rev() @@ -157,17 +215,66 @@ mod tests { builder.finish() }) .collect(); - let mut shuffle = ShuffleOperator::new(Box::new(Chunks(chunks)), vec![LogicalType::Int64]); - let mut values = Vec::new(); + let pulled = Arc::new(AtomicUsize::new(0)); + (Chunks(chunks, Arc::clone(&pulled)), pulled) + } + + fn values(chunk: &DataChunk) -> Vec { + chunk + .selected_indices() + .map(|row| match chunk.column(0).unwrap().get_value(row) { + Some(Value::Int64(value)) => value, + other => panic!("unexpected {other:?}"), + }) + .collect() + } + + /// The values 0 to 99, in three chunks, returned by a shuffle. + fn shuffled() -> Vec { + let mut shuffle = ShuffleOperator::new(Box::new(input().0)); + let mut all = Vec::new(); while let Some(chunk) = shuffle.next().unwrap() { - for row in chunk.selected_indices() { - match chunk.column(0).unwrap().get_value(row) { - Some(Value::Int64(value)) => values.push(value), - other => panic!("unexpected {other:?}"), - } + all.extend(values(&chunk)); + } + all + } + + /// A per-chunk shuffle reads one input chunk per output chunk and keeps + /// each chunk's rows together, in an order that changes between runs. + #[test] + fn a_per_chunk_shuffle_reads_one_chunk_at_a_time() { + let mut first_chunks = std::collections::HashSet::new(); + for _ in 0..5 { + let (chunks, pulled) = input(); + let mut shuffle = ShuffleOperator::per_chunk(Box::new(chunks)); + let mut outputs = Vec::new(); + while let Some(chunk) = shuffle.next().unwrap() { + outputs.push(values(&chunk)); + assert_eq!( + pulled.load(Ordering::Relaxed), + outputs.len(), + "one input chunk per output chunk" + ); } + let sorted: Vec> = outputs + .iter() + .map(|chunk| { + let mut chunk = chunk.clone(); + chunk.sort_unstable(); + chunk + }) + .collect(); + assert_eq!( + sorted, + [ + (0..40).collect::>(), + (40..70).collect(), + (70..100).collect() + ] + ); + first_chunks.insert(outputs[0].clone()); } - values + assert!(first_chunks.len() > 1, "five runs gave the same order"); } #[test] diff --git a/crates/grafeo-core/src/execution/operators/sort.rs b/crates/grafeo-core/src/execution/operators/sort.rs index 0fef7d066..8c9055510 100644 --- a/crates/grafeo-core/src/execution/operators/sort.rs +++ b/crates/grafeo-core/src/execution/operators/sort.rs @@ -5,12 +5,12 @@ use std::cmp::Ordering; -use grafeo_common::types::{LogicalType, Value}; +use grafeo_common::types::Value; -use super::value_utils::compare_values_with_nulls; +use super::value_utils::compare_sort_values; use super::{Operator, OperatorError, OperatorResult}; use crate::execution::DataChunk; -use crate::execution::chunk::DataChunkBuilder; +use crate::execution::chunk::{ColumnTypes, DataChunkBuilder}; /// Sort direction. #[derive(Debug, Clone, Copy, PartialEq, Eq)] @@ -22,7 +22,7 @@ pub enum SortDirection { Descending, } -/// Null ordering. +/// Where nulls go, in either sort direction. #[derive(Debug, Clone, Copy, PartialEq, Eq)] #[non_exhaustive] pub enum NullOrder { @@ -44,7 +44,7 @@ pub struct SortKey { } impl SortKey { - /// Creates a new sort key with ascending order. + /// Creates a new sort key with ascending order, nulls last. pub fn ascending(column: usize) -> Self { Self { column, @@ -53,12 +53,13 @@ impl SortKey { } } - /// Creates a new sort key with descending order. + /// Creates a new sort key with descending order, nulls first: null sorts + /// as the largest value, as in openCypher. pub fn descending(column: usize) -> Self { Self { column, direction: SortDirection::Descending, - null_order: NullOrder::NullsLast, + null_order: NullOrder::NullsFirst, } } @@ -80,14 +81,18 @@ struct SortRow { /// Sort operator. /// -/// Materializes all input and sorts by the specified keys. +/// Materializes all input and sorts by the specified keys. The rows keep +/// their columns' types and values. pub struct SortOperator { /// Child operator. child: Box, /// Sort keys. sort_keys: Vec, - /// Output schema. - output_schema: Vec, + /// The column types of the materialized chunks. + column_types: ColumnTypes, + /// How many leading columns the sort returns, when the trailing ones are + /// sort keys added only to sort by (see [`with_output_width`](Self::with_output_width)). + output_width: Option, /// Materialized chunks. chunks: Vec, /// Sorted row references. @@ -100,15 +105,12 @@ pub struct SortOperator { impl SortOperator { /// Creates a new sort operator. - pub fn new( - child: Box, - sort_keys: Vec, - output_schema: Vec, - ) -> Self { + pub fn new(child: Box, sort_keys: Vec) -> Self { Self { child, sort_keys, - output_schema, + column_types: ColumnTypes::default(), + output_width: None, chunks: Vec::new(), sorted_rows: Vec::new(), sort_complete: false, @@ -116,9 +118,26 @@ impl SortOperator { } } - /// Decomposes this operator into its child and sort keys for push-based conversion. - pub fn into_parts(self) -> (Box, Vec) { - (self.child, self.sort_keys) + /// Returns only the first `width` columns of each row: the columns after + /// them hold sort keys the planner added to sort by, such as `x.age` for + /// `RETURN a AS x ORDER BY x.age`, and are not part of the result. + #[must_use] + pub fn with_output_width(mut self, width: usize) -> Self { + self.output_width = Some(width); + self + } + + /// How many leading columns each output row keeps, when the sort drops + /// trailing sort-key columns. + #[must_use] + pub fn output_width(&self) -> Option { + self.output_width + } + + /// Decomposes this operator into its child, sort keys and output width for + /// push-based conversion. + pub fn into_parts(self) -> (Box, Vec, Option) { + (self.child, self.sort_keys, self.output_width) } /// Materializes and sorts the input. @@ -132,6 +151,7 @@ impl SortOperator { row_index: row_idx, }); } + self.column_types.add(&chunk); self.chunks.push(chunk); } @@ -151,12 +171,12 @@ impl SortOperator { .column(key.column) .and_then(|c| c.get_value(b.row_index)); - let cmp = compare_values_with_nulls(&val_a, &val_b, key.null_order); - - let cmp = match key.direction { - SortDirection::Ascending => cmp, - SortDirection::Descending => cmp.reverse(), - }; + let cmp = compare_sort_values( + val_a.as_ref(), + val_b.as_ref(), + key.direction, + key.null_order, + ); if cmp != Ordering::Equal { return cmp; @@ -180,14 +200,18 @@ impl Operator for SortOperator { return Ok(None); } - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, 2048); + let types = self.column_types.types(); + let width = self + .output_width + .map_or(types.len(), |width| width.min(types.len())); + let mut builder = DataChunkBuilder::with_capacity(&types[..width], 2048); while self.output_position < self.sorted_rows.len() && !builder.is_full() { let row_ref = &self.sorted_rows[self.output_position]; let source_chunk = &self.chunks[row_ref.chunk_index]; - // Copy all columns - for col_idx in 0..source_chunk.column_count() { + // Copy the returned columns + for col_idx in 0..width.min(source_chunk.column_count()) { if let (Some(src_col), Some(dst_col)) = (source_chunk.column(col_idx), builder.column_mut(col_idx)) { @@ -212,6 +236,7 @@ impl Operator for SortOperator { fn reset(&mut self) { self.child.reset(); + self.column_types = ColumnTypes::default(); self.chunks.clear(); self.sorted_rows.clear(); self.sort_complete = false; @@ -231,6 +256,7 @@ impl Operator for SortOperator { mod tests { use super::*; use crate::execution::chunk::DataChunkBuilder; + use grafeo_common::types::LogicalType; struct MockOperator { chunks: Vec, @@ -288,11 +314,7 @@ mod tests { fn test_sort_ascending() { let mock = MockOperator::new(vec![create_unsorted_chunk()]); - let mut sort = SortOperator::new( - Box::new(mock), - vec![SortKey::ascending(0)], - vec![LogicalType::Int64, LogicalType::String], - ); + let mut sort = SortOperator::new(Box::new(mock), vec![SortKey::ascending(0)]); let mut results = Vec::new(); while let Some(chunk) = sort.next().unwrap() { @@ -319,15 +341,60 @@ mod tests { ); } + /// Values of different types sort in one order (maps and lists, strings, + /// booleans, numbers), and nulls go where the key says, in either + /// direction: last ascending and first descending unless it says otherwise. + #[test] + fn mixed_values_sort_in_one_order_with_nulls_where_the_key_says() { + let input = || { + let mut builder = DataChunkBuilder::new(&[LogicalType::Any]); + for value in [ + Value::Int64(3), + Value::String("a".into()), + Value::Float64(2.5), + Value::Null, + Value::Bool(true), + Value::List(vec![Value::Int64(1)].into()), + ] { + builder.column_mut(0).unwrap().push_value(value); + builder.advance_row(); + } + MockOperator::new(vec![builder.finish()]) + }; + let sorted = |key: SortKey| { + let mut sort = SortOperator::new(Box::new(input()), vec![key]); + let mut out = Vec::new(); + while let Some(chunk) = sort.next().unwrap() { + for row in chunk.selected_indices() { + out.push(chunk.column(0).unwrap().get_value(row).unwrap().to_string()); + } + } + out + }; + + assert_eq!( + sorted(SortKey::ascending(0)), + ["[1]", "\"a\"", "true", "2.5", "3", "NULL"] + ); + assert_eq!( + sorted(SortKey::descending(0)), + ["NULL", "3", "2.5", "true", "\"a\"", "[1]"] + ); + assert_eq!( + sorted(SortKey::descending(0).with_null_order(NullOrder::NullsLast)), + ["3", "2.5", "true", "\"a\"", "[1]", "NULL"] + ); + assert_eq!( + sorted(SortKey::ascending(0).with_null_order(NullOrder::NullsFirst)), + ["NULL", "[1]", "\"a\"", "true", "2.5", "3"] + ); + } + #[test] fn test_sort_descending() { let mock = MockOperator::new(vec![create_unsorted_chunk()]); - let mut sort = SortOperator::new( - Box::new(mock), - vec![SortKey::descending(0)], - vec![LogicalType::Int64, LogicalType::String], - ); + let mut sort = SortOperator::new(Box::new(mock), vec![SortKey::descending(0)]); let mut results = Vec::new(); while let Some(chunk) = sort.next().unwrap() { @@ -344,11 +411,7 @@ mod tests { fn test_sort_by_string() { let mock = MockOperator::new(vec![create_unsorted_chunk()]); - let mut sort = SortOperator::new( - Box::new(mock), - vec![SortKey::ascending(1)], // Sort by string column - vec![LogicalType::Int64, LogicalType::String], - ); + let mut sort = SortOperator::new(Box::new(mock), vec![SortKey::ascending(1)]); let mut results = Vec::new(); while let Some(chunk) = sort.next().unwrap() { @@ -378,11 +441,7 @@ mod tests { fn test_sort_empty_input() { let mock = MockOperator::new(vec![]); - let mut sort = SortOperator::new( - Box::new(mock), - vec![SortKey::ascending(0)], - vec![LogicalType::Int64], - ); + let mut sort = SortOperator::new(Box::new(mock), vec![SortKey::ascending(0)]); assert!(sort.next().unwrap().is_none()); } @@ -397,11 +456,7 @@ mod tests { let chunk = builder.finish(); let mock = MockOperator::new(vec![chunk]); - let mut sort = SortOperator::new( - Box::new(mock), - vec![SortKey::ascending(0)], - vec![LogicalType::Int64], - ); + let mut sort = SortOperator::new(Box::new(mock), vec![SortKey::ascending(0)]); let mut results = Vec::new(); while let Some(chunk) = sort.next().unwrap() { @@ -423,11 +478,7 @@ mod tests { let chunk = builder.finish(); let mock = MockOperator::new(vec![chunk]); - let mut sort = SortOperator::new( - Box::new(mock), - vec![SortKey::ascending(0)], - vec![LogicalType::Int64], - ); + let mut sort = SortOperator::new(Box::new(mock), vec![SortKey::ascending(0)]); let mut results = Vec::new(); while let Some(chunk) = sort.next().unwrap() { @@ -454,7 +505,6 @@ mod tests { let mut sort = SortOperator::new( Box::new(mock), vec![SortKey::ascending(0), SortKey::ascending(1)], - vec![LogicalType::String, LogicalType::Int64], ); let mut results = Vec::new(); @@ -500,11 +550,7 @@ mod tests { let chunk2 = b2.finish(); let mock = MockOperator::new(vec![chunk1, chunk2]); - let mut sort = SortOperator::new( - Box::new(mock), - vec![SortKey::ascending(0)], - vec![LogicalType::Int64], - ); + let mut sort = SortOperator::new(Box::new(mock), vec![SortKey::ascending(0)]); let mut results = Vec::new(); while let Some(chunk) = sort.next().unwrap() { @@ -526,11 +572,7 @@ mod tests { let chunk = builder.finish(); let mock = MockOperator::new(vec![chunk]); - let mut sort = SortOperator::new( - Box::new(mock), - vec![SortKey::ascending(0)], - vec![LogicalType::Int64], - ); + let mut sort = SortOperator::new(Box::new(mock), vec![SortKey::ascending(0)]); let mut results = Vec::new(); while let Some(chunk) = sort.next().unwrap() { @@ -550,11 +592,7 @@ mod tests { let chunk = builder.finish(); let mock = MockOperator::new(vec![chunk]); - let mut sort = SortOperator::new( - Box::new(mock), - vec![SortKey::ascending(0)], - vec![LogicalType::Int64], - ); + let mut sort = SortOperator::new(Box::new(mock), vec![SortKey::ascending(0)]); let mut count = 0; while let Some(chunk) = sort.next().unwrap() { @@ -569,22 +607,14 @@ mod tests { #[test] fn test_sort_name() { let mock = MockOperator::new(vec![]); - let sort = SortOperator::new( - Box::new(mock), - vec![SortKey::ascending(0)], - vec![LogicalType::Int64], - ); + let sort = SortOperator::new(Box::new(mock), vec![SortKey::ascending(0)]); assert_eq!(sort.name(), "Sort"); } #[test] fn test_sort_into_any() { let mock = MockOperator::new(vec![]); - let op = SortOperator::new( - Box::new(mock), - vec![SortKey::ascending(0)], - vec![LogicalType::Int64], - ); + let op = SortOperator::new(Box::new(mock), vec![SortKey::ascending(0)]); let any = Box::new(op).into_any(); assert!(any.downcast::().is_ok()); } @@ -595,12 +625,41 @@ mod tests { let op = SortOperator::new( Box::new(mock), vec![SortKey::ascending(0), SortKey::descending(1)], - vec![LogicalType::Int64, LogicalType::String], ); - let (mut child, sort_keys) = op.into_parts(); + let (mut child, sort_keys, output_width) = op.into_parts(); + assert_eq!(output_width, None); assert_eq!(sort_keys.len(), 2); assert_eq!(sort_keys[0].column, 0); assert_eq!(sort_keys[1].column, 1); assert!(child.next().unwrap().is_none()); } + + /// A sort with an output width returns the leading columns only, in the + /// sorted order and with their types: the trailing ones were sort keys. + #[test] + fn output_width_drops_the_sort_key_columns() { + let mut builder = DataChunkBuilder::new(&[LogicalType::Edge, LogicalType::Int64]); + for (id, key) in [(7_u64, 3_i64), (8, 1), (9, 2)] { + builder + .column_mut(0) + .unwrap() + .push_edge_id(grafeo_common::types::EdgeId::new(id)); + builder.column_mut(1).unwrap().push_int64(key); + builder.advance_row(); + } + let mock = MockOperator::new(vec![builder.finish()]); + let mut sort = + SortOperator::new(Box::new(mock), vec![SortKey::ascending(1)]).with_output_width(1); + assert_eq!(sort.output_width(), Some(1)); + + let chunk = sort.next().unwrap().unwrap(); + assert_eq!(chunk.column_count(), 1); + let edges = chunk.column(0).unwrap(); + assert_eq!(edges.data_type(), &LogicalType::Edge); + let ids: Vec = chunk + .selected_indices() + .map(|row| edges.get_edge_id(row).unwrap().as_u64()) + .collect(); + assert_eq!(ids, [8, 9, 7]); + } } diff --git a/crates/grafeo-core/src/execution/operators/top_k.rs b/crates/grafeo-core/src/execution/operators/top_k.rs index cfb89f3ae..3943e3677 100644 --- a/crates/grafeo-core/src/execution/operators/top_k.rs +++ b/crates/grafeo-core/src/execution/operators/top_k.rs @@ -22,13 +22,13 @@ use std::cmp::Ordering; use std::collections::BinaryHeap; use std::sync::Arc; -use grafeo_common::types::{LogicalType, Value}; +use grafeo_common::types::Value; use super::sort::SortKey; -use super::value_utils::compare_values_with_nulls; +use super::value_utils::compare_sort_values; use super::{Operator, OperatorResult}; use crate::execution::DataChunk; -use crate::execution::chunk::DataChunkBuilder; +use crate::execution::chunk::{ColumnTypes, DataChunkBuilder}; /// Streaming bounded top-K operator. pub struct TopKOperator { @@ -39,7 +39,8 @@ pub struct TopKOperator { /// marginal cost is negligible at k=50, N=1M. sort_keys: Arc>, limit: usize, - output_schema: Vec, + /// The column types of the input chunks: the output keeps them. + column_types: ColumnTypes, state: TopKState, #[cfg(test)] materialized_rows: std::sync::atomic::AtomicUsize, @@ -71,11 +72,8 @@ impl TopKOperator { /// regardless of `child`'s cardinality. /// /// Equivalent in output to `LimitOperator(SortOperator(child, sort_keys), limit)`, - /// including stability on ties. - /// - /// `output_schema` must have the same width as `child`'s output; the - /// operator asserts this on first pull (`debug_assert`) to catch planner - /// bugs that would silently truncate or null-pad rows. + /// including stability on ties. The rows keep their columns' types and + /// values. /// /// # Example /// @@ -103,9 +101,7 @@ impl TopKOperator { /// let mut top_k = TopKOperator::new( /// Box::new(source), /// vec![SortKey::descending(0)], - /// 3, - /// vec![LogicalType::Int64], - /// ); + /// 3); /// /// let chunk = top_k.next().unwrap().unwrap(); /// let mut out = vec![]; @@ -115,17 +111,12 @@ impl TopKOperator { /// assert_eq!(out, vec![319, 88, 33]); /// ``` #[must_use] - pub fn new( - child: Box, - sort_keys: Vec, - limit: usize, - output_schema: Vec, - ) -> Self { + pub fn new(child: Box, sort_keys: Vec, limit: usize) -> Self { Self { child, sort_keys: Arc::new(sort_keys), limit, - output_schema, + column_types: ColumnTypes::default(), state: TopKState::Building { heap: BinaryHeap::new(), next_insertion_id: 0, @@ -161,16 +152,9 @@ impl Operator for TopKOperator { unreachable!("matches! guard above") }; - let mut schema_checked = false; while let Some(chunk) = self.child.next()? { - if !schema_checked { - debug_assert_eq!( - chunk.column_count(), - self.output_schema.len(), - "TopKOperator output_schema width must match child schema width", - ); - schema_checked = true; - } + self.column_types.add(&chunk); + let width = self.column_types.types().len(); for row_idx in chunk.selected_indices() { let new_sort_values = @@ -189,7 +173,7 @@ impl Operator for TopKOperator { continue; } - let row_values = extract_row_values(&chunk, row_idx, self.output_schema.len()); + let row_values = extract_row_values(&chunk, row_idx, width); #[cfg(test)] self.materialized_rows .fetch_add(1, std::sync::atomic::Ordering::Relaxed); @@ -218,10 +202,11 @@ impl Operator for TopKOperator { if let TopKState::Draining { rows, position } = &mut self.state { if *position < rows.len() { - let mut builder = DataChunkBuilder::with_capacity(&self.output_schema, 2048); + let types = self.column_types.types(); + let mut builder = DataChunkBuilder::with_capacity(types, 2048); while *position < rows.len() && !builder.is_full() { let entry = &rows[*position]; - for col_idx in 0..self.output_schema.len() { + for col_idx in 0..types.len() { if let Some(dst_col) = builder.column_mut(col_idx) { let val = entry.row_values[col_idx].clone().unwrap_or(Value::Null); dst_col.push_value(val); @@ -242,6 +227,7 @@ impl Operator for TopKOperator { fn reset(&mut self) { self.child.reset(); + self.column_types = ColumnTypes::default(); self.state = TopKState::Building { heap: BinaryHeap::new(), next_insertion_id: 0, @@ -291,13 +277,13 @@ fn extract_row_values(chunk: &DataChunk, row_idx: usize, n_cols: usize) -> Vec], top: &HeapEntry, keys: &[SortKey]) -> bool { - use super::sort::SortDirection; for (i, key) in keys.iter().enumerate() { - let cmp = compare_values_with_nulls(&new[i], &top.sort_values[i], key.null_order); - let user_cmp = match key.direction { - SortDirection::Ascending => cmp, - SortDirection::Descending => cmp.reverse(), - }; + let user_cmp = compare_sort_values( + new[i].as_ref(), + top.sort_values[i].as_ref(), + key.direction, + key.null_order, + ); match user_cmp { Ordering::Less => return true, Ordering::Greater => return false, @@ -323,26 +309,19 @@ impl PartialOrd for HeapEntry { impl Ord for HeapEntry { fn cmp(&self, other: &Self) -> Ordering { - use super::sort::SortDirection; // Both entries share the same Arc> (one per // TopKOperator); use self's view. // // Goal: BinaryHeap is a max-heap. peek() must return the - // worst-by-user-order so we can evict it on overflow. - // User ASC: worst = largest value, peek wants largest, so - // Ord must say "larger is greater": heap_cmp = cmp. - // User DESC: worst = smallest value, peek wants smallest, so - // Ord must say "smaller is greater": heap_cmp = cmp.reverse(). + // worst-by-user-order so we can evict it on overflow, so Ord is the + // user order itself: a row that comes later is greater. for (i, key) in self.sort_keys.iter().enumerate() { - let cmp = compare_values_with_nulls( - &self.sort_values[i], - &other.sort_values[i], + let heap_cmp = compare_sort_values( + self.sort_values[i].as_ref(), + other.sort_values[i].as_ref(), + key.direction, key.null_order, ); - let heap_cmp = match key.direction { - SortDirection::Ascending => cmp, - SortDirection::Descending => cmp.reverse(), - }; if heap_cmp != Ordering::Equal { return heap_cmp; } @@ -359,6 +338,7 @@ mod tests { use super::*; use crate::execution::DataChunk; use crate::execution::chunk::DataChunkBuilder; + use grafeo_common::types::LogicalType; struct MockOperator { chunks: Vec, @@ -420,12 +400,7 @@ mod tests { #[test] fn top_k_returns_top_k_descending() { let mock = MockOperator::new(vec![chunk_int64(&[19, 88, 33, 8, 319])]); - let mut top_k = TopKOperator::new( - Box::new(mock), - vec![SortKey::descending(0)], - 3, - vec![LogicalType::Int64], - ); + let mut top_k = TopKOperator::new(Box::new(mock), vec![SortKey::descending(0)], 3); let out = collect_int64_col(&mut top_k); assert_eq!(out, vec![319, 88, 33]); } @@ -466,12 +441,7 @@ mod tests { (3, "Mia"), (88, "Butch"), ])]); - let mut top_k = TopKOperator::new( - Box::new(mock), - vec![SortKey::descending(0)], - 2, - vec![LogicalType::Int64, LogicalType::String], - ); + let mut top_k = TopKOperator::new(Box::new(mock), vec![SortKey::descending(0)], 2); let out = collect_int_str(&mut top_k); assert_eq!(out, vec![(88, "Jules".into()), (88, "Butch".into())]); } @@ -484,12 +454,7 @@ mod tests { (88, "Mia"), (3, "Butch"), ])]); - let mut top_k = TopKOperator::new( - Box::new(mock), - vec![SortKey::ascending(0)], - 2, - vec![LogicalType::Int64, LogicalType::String], - ); + let mut top_k = TopKOperator::new(Box::new(mock), vec![SortKey::ascending(0)], 2); let out = collect_int_str(&mut top_k); assert_eq!(out, vec![(3, "Jules".into()), (3, "Butch".into())]); } @@ -504,12 +469,7 @@ mod tests { #[allow(clippy::cast_possible_truncation, clippy::cast_possible_wrap)] let values: Vec = (0..1000_i64).map(|i| (i * 31 + 7) % 1000).collect(); let mock = MockOperator::new(vec![chunk_int64(&values)]); - let mut top_k = TopKOperator::new( - Box::new(mock), - vec![SortKey::ascending(0)], - 5, - vec![LogicalType::Int64], - ); + let mut top_k = TopKOperator::new(Box::new(mock), vec![SortKey::ascending(0)], 5); let out = collect_int64_col(&mut top_k); assert_eq!(out.len(), 5); @@ -541,7 +501,6 @@ mod tests { Box::new(mock), vec![SortKey::descending(0), SortKey::ascending(1)], 2, - vec![LogicalType::Int64, LogicalType::String], ); let out = collect_int_str(&mut top_k); assert_eq!(out, vec![(88, "3".into()), (88, "5".into())]); @@ -565,7 +524,6 @@ mod tests { Box::new(mock), vec![SortKey::ascending(0).with_null_order(NullOrder::NullsFirst)], 3, - vec![LogicalType::Int64], ); // ORDER BY x ASC NULLS FIRST gives [Null, Null, 3, 19, 88]; LIMIT 3 = [Null, Null, 3]. @@ -599,7 +557,6 @@ mod tests { Box::new(mock), vec![SortKey::ascending(0).with_null_order(NullOrder::NullsLast)], 3, - vec![LogicalType::Int64], ); // ORDER BY x ASC NULLS LAST gives [3, 19, 88, Null, Null]; LIMIT 3 = [3, 19, 88]. @@ -619,51 +576,68 @@ mod tests { ); } + /// `NULLS LAST` holds when descending too, and descending puts nulls + /// first by default (null sorts as the largest value). + #[test] + fn top_k_places_nulls_as_the_key_says_when_descending() { + use super::super::sort::NullOrder; + let input = || { + let mut b = DataChunkBuilder::new(&[LogicalType::Int64]); + for v in [Some(19_i64), None, Some(88), None, Some(3)] { + match v { + Some(n) => b.column_mut(0).unwrap().push_int64(n), + None => b.column_mut(0).unwrap().push_value(Value::Null), + } + b.advance_row(); + } + MockOperator::new(vec![b.finish()]) + }; + let top = |key: SortKey| { + let mut top_k = TopKOperator::new(Box::new(input()), vec![key], 3); + let mut out = Vec::new(); + while let Some(chunk) = top_k.next().unwrap() { + for row in chunk.selected_indices() { + out.push(chunk.column(0).unwrap().get_value(row).unwrap()); + } + } + out + }; + + assert_eq!( + top(SortKey::descending(0).with_null_order(NullOrder::NullsLast)), + [Value::Int64(88), Value::Int64(19), Value::Int64(3)] + ); + assert_eq!( + top(SortKey::descending(0)), + [Value::Null, Value::Null, Value::Int64(88)] + ); + } + #[test] fn top_k_empty_input() { let mock = MockOperator::new(vec![]); - let mut top_k = TopKOperator::new( - Box::new(mock), - vec![SortKey::descending(0)], - 5, - vec![LogicalType::Int64], - ); + let mut top_k = TopKOperator::new(Box::new(mock), vec![SortKey::descending(0)], 5); assert_eq!(collect_int64_col(&mut top_k), Vec::::new()); } #[test] fn top_k_k_zero_returns_no_rows() { let mock = MockOperator::new(vec![chunk_int64(&[3, 19, 88])]); - let mut top_k = TopKOperator::new( - Box::new(mock), - vec![SortKey::descending(0)], - 0, - vec![LogicalType::Int64], - ); + let mut top_k = TopKOperator::new(Box::new(mock), vec![SortKey::descending(0)], 0); assert_eq!(collect_int64_col(&mut top_k), Vec::::new()); } #[test] fn top_k_k_greater_than_n() { let mock = MockOperator::new(vec![chunk_int64(&[19, 88, 3])]); - let mut top_k = TopKOperator::new( - Box::new(mock), - vec![SortKey::descending(0)], - 10, - vec![LogicalType::Int64], - ); + let mut top_k = TopKOperator::new(Box::new(mock), vec![SortKey::descending(0)], 10); assert_eq!(collect_int64_col(&mut top_k), vec![88, 19, 3]); } #[test] fn top_k_returns_top_k_ascending() { let mock = MockOperator::new(vec![chunk_int64(&[19, 88, 33, 8, 319])]); - let mut top_k = TopKOperator::new( - Box::new(mock), - vec![SortKey::ascending(0)], - 3, - vec![LogicalType::Int64], - ); + let mut top_k = TopKOperator::new(Box::new(mock), vec![SortKey::ascending(0)], 3); assert_eq!(collect_int64_col(&mut top_k), vec![8, 19, 33]); } @@ -674,24 +648,14 @@ mod tests { chunk_int64(&[33, 8]), chunk_int64(&[40, 319]), ]); - let mut top_k = TopKOperator::new( - Box::new(mock), - vec![SortKey::descending(0)], - 3, - vec![LogicalType::Int64], - ); + let mut top_k = TopKOperator::new(Box::new(mock), vec![SortKey::descending(0)], 3); assert_eq!(collect_int64_col(&mut top_k), vec![319, 88, 40]); } #[test] fn top_k_into_parts_round_trip() { let mock = MockOperator::new(vec![chunk_int64(&[3, 19, 88])]); - let top_k = TopKOperator::new( - Box::new(mock), - vec![SortKey::descending(0)], - 5, - vec![LogicalType::Int64], - ); + let top_k = TopKOperator::new(Box::new(mock), vec![SortKey::descending(0)], 5); let (mut child, sort_keys, limit) = top_k.into_parts(); assert_eq!(sort_keys.len(), 1); assert_eq!(limit, 5); @@ -702,12 +666,7 @@ mod tests { #[test] fn top_k_name() { let mock = MockOperator::new(vec![]); - let top_k = TopKOperator::new( - Box::new(mock), - vec![SortKey::descending(0)], - 5, - vec![LogicalType::Int64], - ); + let top_k = TopKOperator::new(Box::new(mock), vec![SortKey::descending(0)], 5); assert_eq!(top_k.name(), "TopK"); } @@ -718,7 +677,6 @@ mod tests { Box::new(mock), vec![SortKey::descending(0)], 5, - vec![LogicalType::Int64], )); let any = op.into_any(); assert!(any.downcast::().is_ok()); diff --git a/crates/grafeo-core/src/execution/operators/unwind.rs b/crates/grafeo-core/src/execution/operators/unwind.rs index e65db4587..4207d4b5a 100644 --- a/crates/grafeo-core/src/execution/operators/unwind.rs +++ b/crates/grafeo-core/src/execution/operators/unwind.rs @@ -1,8 +1,8 @@ //! Unwind operator for expanding lists into individual rows. use super::{Operator, OperatorResult}; -use crate::execution::chunk::{DataChunk, DataChunkBuilder}; -use grafeo_common::types::{LogicalType, Value}; +use crate::execution::chunk::{DataChunk, DataChunkBuilder, copied_column_types}; +use grafeo_common::types::{LogicalType, PropertyKey, Value}; /// Unwind operator that expands a list column into individual rows. /// @@ -139,11 +139,35 @@ impl UnwindOperator { .expect("current_list is Some: set before emit_row call"); let element = list[self.current_list_idx].clone(); - // Build output row: copy all columns from input + add the unwound element - let mut builder = DataChunkBuilder::new(&self.output_schema); + // The unwound element comes after the declared input columns, followed + // by any ordinality/offset columns. + let extra_cols = usize::from(self.emit_ordinality) + usize::from(self.emit_offset); + let element_col_idx = self.output_schema.len() - 1 - extra_cols; + + // Build output row: copy the declared input columns + add the unwound + // element. The copied columns keep the input's types (a node or edge + // stays one, see `ColumnTypes`); the new ones have the declared types. + // The row is as wide as the declared schema. + let mut types = chunk.column_types(); + types.truncate(element_col_idx); + let copied = types.len(); + let mut types = copied_column_types(&types, &self.output_schema); + types.extend(self.output_schema.iter().skip(copied).cloned()); + // An item of a node or edge list is an ID (a returned node or edge, a + // map with `_id`, stands for its ID) or null; any other item goes into + // a column of any value, so it is never read as a node or an edge. + let element = entity_item(element, &self.output_schema[element_col_idx]); + if matches!( + self.output_schema[element_col_idx], + LogicalType::Node | LogicalType::Edge + ) && !matches!(element, Value::Int64(_) | Value::Null) + { + types[element_col_idx] = LogicalType::Any; + } + let mut builder = DataChunkBuilder::new(&types); // Copy existing columns (except the list column which we're replacing) - for col_idx in 0..chunk.column_count() { + for col_idx in 0..copied { if col_idx == self.list_col_idx { continue; // Skip the list column } @@ -156,9 +180,6 @@ impl UnwindOperator { } // Add the unwound element column. - // It's at the end of the output schema, minus any ordinality/offset columns. - let extra_cols = usize::from(self.emit_ordinality) + usize::from(self.emit_offset); - let element_col_idx = self.output_schema.len() - 1 - extra_cols; if let Some(out_col) = builder.column_mut(element_col_idx) { out_col.push_value(element); } @@ -217,6 +238,21 @@ impl Operator for UnwindOperator { } } +/// The item of a node or edge list as the entity's ID: a map with `_id` (a +/// node or edge a subquery returned) stands for that ID. Other items, and the +/// items of other lists, stay as they are. +fn entity_item(item: Value, item_type: &LogicalType) -> Value { + match (item_type, &item) { + (LogicalType::Node | LogicalType::Edge, Value::Map(map)) => { + match map.get(&PropertyKey::new("_id")) { + Some(id @ Value::Int64(_)) => id.clone(), + _ => item, + } + } + _ => item, + } +} + #[cfg(test)] mod tests { use super::*; @@ -252,6 +288,148 @@ mod tests { } } + /// The input columns keep their types: a node stays a node. + #[test] + fn unwind_keeps_the_input_column_types() { + use grafeo_common::types::NodeId; + + let mut builder = DataChunkBuilder::new(&[LogicalType::Node, LogicalType::Any]); + builder + .column_mut(0) + .unwrap() + .push_node_id(NodeId::new(101)); + builder + .column_mut(1) + .unwrap() + .push_value(Value::List(vec![Value::Int64(1), Value::Int64(2)].into())); + builder.advance_row(); + let child = MockOperator { + chunks: vec![builder.finish()], + position: 0, + }; + let mut unwind = UnwindOperator::new( + Box::new(child), + 1, + "k".to_string(), + vec![LogicalType::Any, LogicalType::Any, LogicalType::Any], + false, + false, + ); + + let mut elements = Vec::new(); + while let Some(chunk) = unwind.next().unwrap() { + assert_eq!(chunk.column_types()[0], LogicalType::Node); + assert_eq!( + chunk.column(0).unwrap().get_node_id(0).unwrap().as_u64(), + 101 + ); + assert!(chunk.column(0).unwrap().get_edge_id(0).is_none()); + elements.push(chunk.column(2).unwrap().get_value(0)); + } + assert_eq!(elements, [Some(Value::Int64(1)), Some(Value::Int64(2))]); + } + + /// The items of a node list go into a node column: IDs, a returned node (a + /// map with `_id`) as its ID, and null. Any other item goes into a column + /// of any value, so it is never read as a node. + #[test] + fn a_node_list_gives_nodes_and_other_items_stay_values() { + use grafeo_common::types::PropertyKey; + use std::collections::BTreeMap; + + let returned_node = Value::Map(Arc::new(BTreeMap::from([( + PropertyKey::new("_id"), + Value::Int64(5), + )]))); + let mut builder = DataChunkBuilder::new(&[LogicalType::Any]); + builder.column_mut(0).unwrap().push_value(Value::List( + vec![ + Value::Int64(3), + returned_node, + Value::String("x".into()), + Value::Null, + ] + .into(), + )); + builder.advance_row(); + let child = MockOperator { + chunks: vec![builder.finish()], + position: 0, + }; + let mut unwind = UnwindOperator::new( + Box::new(child), + 0, + "n".to_string(), + vec![LogicalType::Node], + false, + false, + ); + + let mut items = Vec::new(); + while let Some(chunk) = unwind.next().unwrap() { + let column = chunk.column(0).unwrap(); + items.push(( + column.data_type().clone(), + column.get_node_id(0).map(|id| id.as_u64()), + column.get_value(0), + )); + } + assert_eq!(items[0].0, LogicalType::Node); + assert_eq!(items[0].1, Some(3)); + assert_eq!(items[1].0, LogicalType::Node); + assert_eq!(items[1].1, Some(5)); + assert_eq!(items[2].0, LogicalType::Any); + assert_eq!(items[2].2, Some(Value::String("x".into()))); + assert_eq!(items[3].0, LogicalType::Node); + assert_eq!(items[3].1, None); + assert_eq!(items.len(), 4); + } + + /// The row is as wide as the declared schema: an input column past the + /// declared ones does not take the place of the unwound element. + #[test] + fn unwind_passes_on_only_the_declared_input_columns() { + let mut builder = + DataChunkBuilder::new(&[LogicalType::Int64, LogicalType::Any, LogicalType::Int64]); + builder.column_mut(0).unwrap().push_value(Value::Int64(7)); + builder + .column_mut(1) + .unwrap() + .push_value(Value::List(vec![Value::Int64(1), Value::Int64(2)].into())); + builder.column_mut(2).unwrap().push_value(Value::Int64(99)); + builder.advance_row(); + let child = MockOperator { + chunks: vec![builder.finish()], + position: 0, + }; + // Declared: a column, the list and the element; the input's third + // column is not one of them. + let mut unwind = UnwindOperator::new( + Box::new(child), + 1, + "k".to_string(), + vec![LogicalType::Int64, LogicalType::Any, LogicalType::Any], + false, + false, + ); + + let mut rows = Vec::new(); + while let Some(chunk) = unwind.next().unwrap() { + assert_eq!(chunk.column_count(), 3); + rows.push(( + chunk.column(0).unwrap().get_value(0), + chunk.column(2).unwrap().get_value(0), + )); + } + assert_eq!( + rows, + [ + (Some(Value::Int64(7)), Some(Value::Int64(1))), + (Some(Value::Int64(7)), Some(Value::Int64(2))) + ] + ); + } + #[test] fn test_unwind_basic() { // Create a chunk with a list column [1, 2, 3] diff --git a/crates/grafeo-core/src/execution/operators/value_utils.rs b/crates/grafeo-core/src/execution/operators/value_utils.rs index 149e71c8c..57e238926 100644 --- a/crates/grafeo-core/src/execution/operators/value_utils.rs +++ b/crates/grafeo-core/src/execution/operators/value_utils.rs @@ -4,10 +4,11 @@ //! to avoid duplicating comparison logic across six different files. use std::cmp::Ordering; +use std::collections::HashMap; use grafeo_common::types::Value; -use super::sort::NullOrder; +use super::sort::{NullOrder, SortDirection}; /// Converts a value to `f64` for numeric aggregations. /// @@ -53,48 +54,232 @@ pub fn compare_values(a: &Value, b: &Value) -> Option { } } -/// Compares two values with total ordering (returns `Equal` for incomparable types). +/// Compares two values for `ORDER BY`: a total order over every value, the +/// orderability of openCypher. /// -/// Used by sort operators where a total order is required. +/// Values of different types order by type, ascending: maps, lists, paths, +/// vectors, zoned datetimes, datetimes, dates, zoned times, local times, +/// durations, bytes, strings, booleans, counters, numbers, then null. A sort +/// key's own nulls do not get here: they go first or last as its +/// `NULLS FIRST` / `NULLS LAST` says (see [`compare_sort_values`]); null +/// inside a list or map is larger than any other value. +/// +/// Within a type: numbers numerically, integers and floats exactly, with NaN +/// after positive infinity; strings by code point; `false` before `true`; +/// temporal values in time order; durations by months, then days, then +/// nanoseconds; lists and paths element by element, a prefix first; maps by +/// size, then their keys, then their values; vectors and bytes element by +/// element; counters by their value. +#[must_use] pub fn compare_values_total(a: &Value, b: &Value) -> Ordering { + let by_type = type_rank(a).cmp(&type_rank(b)); + if by_type.is_ne() { + return by_type; + } match (a, b) { (Value::Bool(a), Value::Bool(b)) => a.cmp(b), (Value::Int64(a), Value::Int64(b)) => a.cmp(b), - (Value::Float64(a), Value::Float64(b)) => a.partial_cmp(b).unwrap_or(Ordering::Equal), + (Value::Float64(a), Value::Float64(b)) => compare_floats(*a, *b), + (Value::Int64(a), Value::Float64(b)) => compare_int_float(*a, *b), + (Value::Float64(a), Value::Int64(b)) => compare_int_float(*b, *a).reverse(), (Value::String(a), Value::String(b)) => a.cmp(b), - (Value::Int64(a), Value::Float64(b)) => { - (*a as f64).partial_cmp(b).unwrap_or(Ordering::Equal) - } - (Value::Float64(a), Value::Int64(b)) => { - a.partial_cmp(&(*b as f64)).unwrap_or(Ordering::Equal) - } + (Value::Bytes(a), Value::Bytes(b)) => a.cmp(b), (Value::Timestamp(a), Value::Timestamp(b)) => a.cmp(b), (Value::Date(a), Value::Date(b)) => a.cmp(b), (Value::Time(a), Value::Time(b)) => a.cmp(b), + (Value::ZonedDatetime(a), Value::ZonedDatetime(b)) => a.cmp(b), + (Value::Duration(a), Value::Duration(b)) => { + (a.months(), a.days(), a.nanos()).cmp(&(b.months(), b.days(), b.nanos())) + } + (Value::List(a), Value::List(b)) => compare_sequences(a.iter(), b.iter()), + ( + Value::Path { + nodes: a_nodes, + edges: a_edges, + }, + Value::Path { + nodes: b_nodes, + edges: b_edges, + }, + ) => compare_sequences( + path_elements(a_nodes, a_edges), + path_elements(b_nodes, b_edges), + ), + (Value::Map(a), Value::Map(b)) => a + .len() + .cmp(&b.len()) + .then_with(|| a.keys().cmp(b.keys())) + .then_with(|| compare_sequences(a.values(), b.values())), + (Value::Vector(a), Value::Vector(b)) => a + .iter() + .zip(b.iter()) + .map(|(x, y)| compare_floats(f64::from(*x), f64::from(*y))) + .find(|order| order.is_ne()) + .unwrap_or_else(|| a.len().cmp(&b.len())), + (Value::GCounter(_) | Value::OnCounter { .. }, _) => { + counter_value(a).cmp(&counter_value(b)) + } + // Values of the same rank not matched above are equal: nulls, and + // values of a type added later. _ => Ordering::Equal, } } -/// Compares two optional values with null handling. +/// The position of a value's type in the order across types. +fn type_rank(value: &Value) -> u8 { + match value { + Value::Map(_) => 0, + Value::List(_) => 1, + Value::Path { .. } => 2, + Value::Vector(_) => 3, + Value::ZonedDatetime(_) => 4, + Value::Timestamp(_) => 5, + Value::Date(_) => 6, + Value::Time(time) if time.offset_seconds().is_some() => 7, + Value::Time(_) => 8, + Value::Duration(_) => 9, + Value::Bytes(_) => 10, + Value::String(_) => 11, + Value::Bool(_) => 12, + Value::GCounter(_) | Value::OnCounter { .. } => 13, + Value::Int64(_) | Value::Float64(_) => 15, + Value::Null => 16, + // A type added later goes before the numbers: openCypher keeps types + // it does not define out from after them. + _ => 14, + } +} + +/// Compares two floats numerically, with NaN after every other number and +/// equal to itself; `-0.0` equals `0.0`. A total order, so floats can be +/// sorted with it (`partial_cmp` with NaN as equal is not: a sort may panic). +pub(crate) fn compare_floats(a: f64, b: f64) -> Ordering { + match (a.is_nan(), b.is_nan()) { + (true, true) => Ordering::Equal, + (true, false) => Ordering::Greater, + (false, true) => Ordering::Less, + // Without NaN, `partial_cmp` always has an answer. + (false, false) => a.partial_cmp(&b).unwrap_or(Ordering::Equal), + } +} + +/// Compares an integer with a float exactly. Converting the integer to a +/// float rounds above 2^53, which would make `2^53 + 1` equal to the float +/// `2^53` while the integers `2^53` and `2^53 + 1` differ. NaN is after every +/// integer. +fn compare_int_float(int: i64, float: f64) -> Ordering { + // 2^63, the smallest float above every i64. + const BEYOND_I64: f64 = 9_223_372_036_854_775_808.0; + if float.is_nan() || float >= BEYOND_I64 { + return Ordering::Less; + } + if float < -BEYOND_I64 { + return Ordering::Greater; + } + let whole = float.trunc(); + #[allow( + clippy::cast_possible_truncation, + reason = "`whole` is an integer in [-2^63, 2^63), checked above, so the cast is exact" + )] + let whole_int = whole as i64; + // On the same whole part, the float's fraction decides. + int.cmp(&whole_int) + .then_with(|| whole.partial_cmp(&float).unwrap_or(Ordering::Equal)) +} + +/// Compares two sequences element by element; a prefix comes first. +fn compare_sequences<'a>( + a: impl IntoIterator, + b: impl IntoIterator, +) -> Ordering { + let mut b = b.into_iter(); + for x in a { + let Some(y) = b.next() else { + return Ordering::Greater; + }; + let order = compare_values_total(x, y); + if order.is_ne() { + return order; + } + } + if b.next().is_some() { + Ordering::Less + } else { + Ordering::Equal + } +} + +/// The nodes and edges of a path, alternating from its first node. +fn path_elements<'a>(nodes: &'a [Value], edges: &'a [Value]) -> impl Iterator { + nodes + .iter() + .enumerate() + .flat_map(move |(index, node)| std::iter::once(node).chain(edges.get(index))) +} + +/// The value of a counter. +fn counter_value(value: &Value) -> i128 { + let total = |counts: &HashMap| { + i128::try_from( + counts + .values() + .map(|&count| u128::from(count)) + .sum::(), + ) + .unwrap_or(i128::MAX) + }; + match value { + Value::GCounter(counts) => total(counts), + Value::OnCounter { pos, neg } => total(pos).saturating_sub(total(neg)), + _ => 0, + } +} + +/// Orders the values of one sort key: nulls (and missing values) first or +/// last as `null_order` says, in either direction, and the other values by +/// [`compare_values_total`] in `direction`. /// -/// Used by sort and top-K operators to handle the `NULLS FIRST` / `NULLS LAST` -/// directive uniformly. Both `None` and `Some(Value::Null)` are treated as null. -pub fn compare_values_with_nulls( - a: &Option, - b: &Option, +/// Used by the sort and top-K operators for `ORDER BY ... [ASC | DESC] +/// [NULLS FIRST | NULLS LAST]`. +#[must_use] +pub fn compare_sort_values( + a: Option<&Value>, + b: Option<&Value>, + direction: SortDirection, null_order: NullOrder, ) -> Ordering { - match (a, b) { - (None, None) | (Some(Value::Null), Some(Value::Null)) => Ordering::Equal, - (None, _) | (Some(Value::Null), _) => match null_order { - NullOrder::NullsFirst => Ordering::Less, - NullOrder::NullsLast => Ordering::Greater, - }, - (_, None) | (_, Some(Value::Null)) => match null_order { - NullOrder::NullsFirst => Ordering::Greater, - NullOrder::NullsLast => Ordering::Less, - }, - (Some(a), Some(b)) => compare_values_total(a, b), + order_by( + a, + b, + direction == SortDirection::Descending, + null_order == NullOrder::NullsFirst, + ) +} + +/// [`compare_sort_values`] with the direction and the null order as flags, for +/// the sorts that have sort keys of their own. +pub(crate) fn order_by( + a: Option<&Value>, + b: Option<&Value>, + descending: bool, + nulls_first: bool, +) -> Ordering { + let nulls = if nulls_first { + Ordering::Less + } else { + Ordering::Greater + }; + match ( + a.filter(|value| !value.is_null()), + b.filter(|value| !value.is_null()), + ) { + (None, None) => Ordering::Equal, + (None, Some(_)) => nulls, + (Some(_), None) => nulls.reverse(), + (Some(a), Some(b)) => { + let order = compare_values_total(a, b); + if descending { order.reverse() } else { order } + } } } @@ -120,6 +305,11 @@ pub fn is_greater_than(current: &Option, new: &Value) -> bool { #[cfg(test)] mod tests { + use std::collections::BTreeMap; + use std::sync::Arc; + + use grafeo_common::types::{Date, Duration, PropertyKey, Time, Timestamp, ZonedDatetime}; + use super::*; #[test] @@ -176,14 +366,6 @@ mod tests { assert_eq!(compare_values(&Value::Bool(true), &Value::Int64(1)), None); } - #[test] - fn total_ordering_incomparable_returns_equal() { - assert_eq!( - compare_values_total(&Value::Bool(true), &Value::Int64(1)), - Ordering::Equal - ); - } - #[test] fn is_less_than_none_always_true() { assert!(is_less_than(&None, &Value::Int64(5))); @@ -213,4 +395,381 @@ mod tests { fn is_greater_than_smaller() { assert!(!is_greater_than(&Some(Value::Int64(10)), &Value::Int64(5))); } + + fn list(values: Vec) -> Value { + Value::List(values.into()) + } + + fn map(entries: &[(&str, Value)]) -> Value { + let map: BTreeMap = entries + .iter() + .map(|(key, value)| (PropertyKey::new(*key), value.clone())) + .collect(); + Value::Map(Arc::new(map)) + } + + fn time(hour: u32) -> Time { + Time::from_hms(hour, 0, 0).unwrap() + } + + /// One value of each type, in ascending order. + fn one_of_each_type() -> Vec { + vec![ + map(&[("k", Value::Int64(1))]), + list(vec![Value::Int64(1)]), + Value::Path { + nodes: vec![Value::Int64(0), Value::Int64(1)].into(), + edges: vec![Value::Int64(0)].into(), + }, + Value::Vector(vec![1.0_f32, 2.0].into()), + Value::ZonedDatetime(ZonedDatetime::from_timestamp_offset( + Timestamp::from_secs(0), + 3600, + )), + Value::Timestamp(Timestamp::from_secs(0)), + Value::Date(Date::from_ymd(2024, 1, 15).unwrap()), + Value::Time(time(12).with_offset(3600)), + Value::Time(time(12)), + Value::Duration(Duration::new(1, 0, 0)), + Value::Bytes(vec![1_u8, 2].into()), + Value::String("a".into()), + Value::Bool(false), + Value::GCounter(Arc::new(HashMap::from([("r".to_string(), 3_u64)]))), + Value::Int64(1), + Value::Null, + ] + } + + /// Sorts `values` with `compare_values_total`; also checks that the + /// result is in order, pair by pair. + fn sorted(mut values: Vec) -> Vec { + values.sort_by(compare_values_total); + for pair in values.windows(2) { + assert_ne!( + compare_values_total(&pair[0], &pair[1]), + Ordering::Greater, + "{:?} before {:?}", + pair[0], + pair[1] + ); + } + values + } + + #[test] + fn values_of_different_types_follow_the_opencypher_order() { + let ascending = one_of_each_type(); + for rotation in 0..ascending.len() { + let mut shuffled = ascending.clone(); + shuffled.rotate_left(rotation); + shuffled.reverse(); + assert_eq!(sorted(shuffled), ascending, "rotation {rotation}"); + } + } + + #[test] + fn integers_and_floats_compare_exactly() { + let two_53 = 9_007_199_254_740_992_i64; + let cases = [ + ( + Value::Int64(two_53), + Value::Float64(2f64.powi(53)), + Ordering::Equal, + ), + ( + Value::Int64(two_53 + 1), + Value::Float64(2f64.powi(53)), + Ordering::Greater, + ), + (Value::Int64(2), Value::Float64(2.5), Ordering::Less), + (Value::Int64(3), Value::Float64(2.5), Ordering::Greater), + (Value::Int64(-2), Value::Float64(-2.5), Ordering::Greater), + (Value::Int64(-3), Value::Float64(-2.5), Ordering::Less), + (Value::Int64(0), Value::Float64(-0.0), Ordering::Equal), + (Value::Float64(-0.0), Value::Float64(0.0), Ordering::Equal), + ( + Value::Int64(i64::MAX), + Value::Float64(2f64.powi(63)), + Ordering::Less, + ), + ( + Value::Int64(i64::MIN), + Value::Float64(-(2f64.powi(63))), + Ordering::Equal, + ), + ( + Value::Int64(i64::MAX), + Value::Float64(f64::INFINITY), + Ordering::Less, + ), + ( + Value::Int64(i64::MIN), + Value::Float64(f64::NEG_INFINITY), + Ordering::Greater, + ), + ]; + for (a, b, expected) in cases { + assert_eq!(compare_values_total(&a, &b), expected, "{a:?} vs {b:?}"); + assert_eq!( + compare_values_total(&b, &a), + expected.reverse(), + "{b:?} vs {a:?}" + ); + } + } + + #[test] + fn nan_is_the_largest_number_and_equals_itself() { + let nan = Value::Float64(f64::NAN); + assert_eq!( + compare_values_total(&nan, &Value::Float64(f64::INFINITY)), + Ordering::Greater + ); + assert_eq!( + compare_values_total(&nan, &Value::Int64(i64::MAX)), + Ordering::Greater + ); + assert_eq!(compare_values_total(&nan, &nan), Ordering::Equal); + assert_eq!( + sorted(vec![ + Value::Int64(1), + nan.clone(), + Value::Int64(-1), + Value::Float64(f64::INFINITY), + Value::Float64(f64::NEG_INFINITY), + ]) + .iter() + .map(|value| format!("{value:?}")) + .collect::>(), + [ + "Float64(-inf)", + "Int64(-1)", + "Int64(1)", + "Float64(inf)", + "Float64(NaN)" + ] + ); + } + + /// The openCypher examples: element by element, a missing element + /// before any value (null too), null after any value. + #[test] + fn lists_compare_element_by_element() { + let cases = [ + ( + vec![Value::Int64(1)], + vec![Value::Int64(1), Value::Int64(0)], + ), + (vec![Value::Int64(1)], vec![Value::Int64(1), Value::Null]), + ( + vec![ + Value::Int64(1), + Value::String("foo".into()), + Value::Int64(3), + ], + vec![ + Value::Int64(1), + Value::Int64(2), + Value::String("bar".into()), + ], + ), + ( + vec![Value::Int64(1), Value::Int64(2)], + vec![Value::Null, Value::Int64(1)], + ), + ( + vec![Value::Null, Value::Int64(1)], + vec![Value::Null, Value::Int64(2)], + ), + (vec![], vec![Value::Null]), + ]; + for (smaller, larger) in cases { + let (smaller, larger) = (list(smaller), list(larger)); + assert_eq!( + compare_values_total(&smaller, &larger), + Ordering::Less, + "{smaller:?} < {larger:?}" + ); + assert_eq!(compare_values_total(&larger, &smaller), Ordering::Greater); + } + } + + #[test] + fn maps_compare_by_size_then_keys_then_values() { + let cases = [ + ( + map(&[("z", Value::Int64(9))]), + map(&[("a", Value::Int64(1)), ("b", Value::Int64(1))]), + ), + ( + map(&[("a", Value::Int64(2))]), + map(&[("b", Value::Int64(1))]), + ), + ( + map(&[("a", Value::Int64(1))]), + map(&[("a", Value::Int64(2))]), + ), + (map(&[("a", Value::Int64(1))]), map(&[("a", Value::Null)])), + ]; + for (smaller, larger) in cases { + assert_eq!( + compare_values_total(&smaller, &larger), + Ordering::Less, + "{smaller:?} < {larger:?}" + ); + assert_eq!(compare_values_total(&larger, &smaller), Ordering::Greater); + } + } + + #[test] + fn paths_compare_node_edge_node() { + let path = |ids: &[i64]| Value::Path { + nodes: ids.iter().step_by(2).map(|&id| Value::Int64(id)).collect(), + edges: ids + .iter() + .skip(1) + .step_by(2) + .map(|&id| Value::Int64(id)) + .collect(), + }; + // The openCypher example: n1 r1 n3 before n1 r2 n2, decided by the edge. + assert_eq!( + compare_values_total(&path(&[1, 1, 3]), &path(&[1, 2, 2])), + Ordering::Less + ); + assert_eq!( + compare_values_total(&path(&[1]), &path(&[1, 1, 3])), + Ordering::Less + ); + } + + #[test] + fn temporal_values_order_in_time_within_their_type() { + assert_eq!( + compare_values_total( + &Value::Duration(Duration::new(1, 0, 0)), + &Value::Duration(Duration::new(0, 40, 0)) + ), + Ordering::Greater, + "months before days" + ); + assert_eq!( + compare_values_total( + &Value::Duration(Duration::new(0, 1, 5)), + &Value::Duration(Duration::new(0, 1, 7)) + ), + Ordering::Less + ); + // A zoned time before every local time, whatever the clock says. + assert_eq!( + compare_values_total(&Value::Time(time(23).with_offset(0)), &Value::Time(time(1))), + Ordering::Less + ); + assert_eq!( + compare_values_total( + &Value::Time(time(9).with_offset(3600)), + &Value::Time(time(9).with_offset(0)) + ), + Ordering::Less, + "09:00+01:00 is 08:00 UTC" + ); + } + + /// The order is total on a mix of every type and the edge cases of each: + /// antisymmetric, transitive, and `sort_by` gives a sorted result. + /// Comparing values of different types as equal broke transitivity, and + /// `slice::sort_by` may panic on such a comparator. + #[test] + fn the_order_is_total_on_mixed_values() { + let mut pool = one_of_each_type(); + pool.extend([ + Value::Int64(-3), + Value::Int64(9_007_199_254_740_993), + Value::Float64(9_007_199_254_740_992.0), + Value::Float64(2.5), + Value::Float64(-0.0), + Value::Float64(f64::NAN), + Value::Float64(f64::INFINITY), + Value::String(String::new().into()), + Value::String("b".into()), + Value::Bool(true), + list(vec![]), + list(vec![Value::Null]), + list(vec![Value::Int64(1), Value::Null]), + list(vec![Value::String("x".into())]), + map(&[]), + map(&[("a", Value::Null)]), + Value::Vector(vec![f32::NAN].into()), + Value::Bytes(vec![].into()), + Value::OnCounter { + pos: Arc::new(HashMap::from([("r".to_string(), 1_u64)])), + neg: Arc::new(HashMap::from([("r".to_string(), 5_u64)])), + }, + ]); + for a in &pool { + for b in &pool { + let order = compare_values_total(a, b); + assert_eq!( + compare_values_total(b, a), + order.reverse(), + "{a:?} vs {b:?}" + ); + for c in &pool { + if order.is_le() && compare_values_total(b, c).is_le() { + assert!( + compare_values_total(a, c).is_le(), + "{a:?} <= {b:?} <= {c:?}" + ); + } + } + } + } + // A long mix in a scrambled order sorts without panicking. + let mut long: Vec = (0..2000_u64) + .map(|i| pool[usize::try_from(i * 7919 % 41).unwrap() % pool.len()].clone()) + .collect(); + long.rotate_left(17); + sorted(long); + } + + #[test] + fn sort_values_put_nulls_where_the_key_says_in_both_directions() { + let (one, two) = (Value::Int64(1), Value::Int64(2)); + let null = Value::Null; + for direction in [SortDirection::Ascending, SortDirection::Descending] { + assert_eq!( + compare_sort_values(Some(&null), Some(&one), direction, NullOrder::NullsFirst), + Ordering::Less, + "{direction:?} NULLS FIRST" + ); + assert_eq!( + compare_sort_values(Some(&null), Some(&one), direction, NullOrder::NullsLast), + Ordering::Greater, + "{direction:?} NULLS LAST" + ); + assert_eq!( + compare_sort_values(None, Some(&null), direction, NullOrder::NullsLast), + Ordering::Equal, + "a missing value is null" + ); + } + assert_eq!( + compare_sort_values( + Some(&one), + Some(&two), + SortDirection::Descending, + NullOrder::NullsLast + ), + Ordering::Greater + ); + assert_eq!( + compare_sort_values( + Some(&one), + Some(&two), + SortDirection::Ascending, + NullOrder::NullsLast + ), + Ordering::Less + ); + } } diff --git a/crates/grafeo-core/src/execution/operators/writer.rs b/crates/grafeo-core/src/execution/operators/writer.rs index 19a855d01..b5850acb9 100644 --- a/crates/grafeo-core/src/execution/operators/writer.rs +++ b/crates/grafeo-core/src/execution/operators/writer.rs @@ -240,6 +240,33 @@ impl GraphWriter { } } + /// Fails when this writer's transaction cannot see the node: one it + /// deleted earlier, or one that does not exist. A write to it would change + /// nothing anyone sees. + fn require_node(&self, id: NodeId) -> Result<(), OperatorError> { + if self.has_node(id) { + Ok(()) + } else { + Err(OperatorError::Execution(format!( + "Node {} does not exist or has been deleted in this transaction", + id.as_u64() + ))) + } + } + + /// Fails when this writer's transaction cannot see the edge, like + /// [`require_node`](Self::require_node). + fn require_edge(&self, id: EdgeId) -> Result<(), OperatorError> { + if self.has_edge(id) { + Ok(()) + } else { + Err(OperatorError::Execution(format!( + "Relationship {} does not exist or has been deleted in this transaction", + id.as_u64() + ))) + } + } + fn record(&self, entity: Entity) -> Result<(), OperatorError> { if let (Some(tracker), Some(transaction_id)) = (&self.write_tracker, self.transaction_id) { match entity { @@ -318,13 +345,15 @@ impl GraphWriter { /// /// # Errors /// - /// Returns a write conflict or the first constraint violated. + /// Returns an error for a node the transaction deleted, a write conflict + /// or the first constraint violated. pub fn set_node_properties( &self, id: NodeId, assignments: &[(String, Value)], replace: bool, ) -> Result<(), OperatorError> { + self.require_node(id)?; self.record(Entity::Node(id))?; if let Some(validator) = &self.validator { let needs_node = replace @@ -347,9 +376,10 @@ impl GraphWriter { /// /// # Errors /// - /// Returns a write conflict or the constraint the removal would violate - /// (`NOT NULL`, `NODE KEY`). + /// Returns an error for a node the transaction deleted, a write conflict + /// or the constraint the removal would violate (`NOT NULL`, `NODE KEY`). pub fn remove_node_property(&self, id: NodeId, key: &str) -> Result { + self.require_node(id)?; self.record(Entity::Node(id))?; let Some(node) = self.node(id) else { return Ok(false); @@ -370,13 +400,14 @@ impl GraphWriter { } /// Adds labels to a node, after checking the node against the - /// constraints of the labels it gets. Returns how many were new (none for - /// a node that does not exist). + /// constraints of the labels it gets. Returns how many were new. /// /// # Errors /// - /// Returns a write conflict or the first constraint violated. + /// Returns an error for a node the transaction deleted, a write conflict + /// or the first constraint violated. pub fn add_labels(&self, id: NodeId, labels: &[String]) -> Result { + self.require_node(id)?; self.record(Entity::Node(id))?; let Some(node) = self.node(id) else { return Ok(0); @@ -411,13 +442,14 @@ impl GraphWriter { Ok(added) } - /// Removes labels from a node. Returns how many it had (none for a node - /// that does not exist). + /// Removes labels from a node. Returns how many it had. /// /// # Errors /// - /// Returns a write conflict. + /// Returns an error for a node the transaction deleted, or a write + /// conflict. pub fn remove_labels(&self, id: NodeId, labels: &[String]) -> Result { + self.require_node(id)?; self.record(Entity::Node(id))?; if self.node(id).is_none() { return Ok(0); @@ -532,13 +564,15 @@ impl GraphWriter { /// /// # Errors /// - /// Returns a write conflict or the first constraint violated. + /// Returns an error for an edge the transaction deleted, a write conflict + /// or the first constraint violated. pub fn set_edge_properties( &self, id: EdgeId, assignments: &[(String, Value)], replace: bool, ) -> Result<(), OperatorError> { + self.require_edge(id)?; self.record(Entity::Edge(id))?; if let Some(validator) = &self.validator && let Some(edge) = self.edge(id) @@ -557,8 +591,10 @@ impl GraphWriter { /// /// # Errors /// - /// Returns a write conflict or the constraint the removal would violate. + /// Returns an error for an edge the transaction deleted, a write conflict + /// or the constraint the removal would violate. pub fn remove_edge_property(&self, id: EdgeId, key: &str) -> Result { + self.require_edge(id)?; self.record(Entity::Edge(id))?; let Some(edge) = self.edge(id) else { return Ok(false); @@ -659,7 +695,9 @@ impl GraphWriter { .create_node_versioned(&label_refs, self.epoch(), self.transaction()); self.record(Entity::Node(id))?; self.count(|c| &c.nodes_created, 1); - self.count(|c| &c.labels_added, labels.len()); + // `(:A:A)` gives the node one label. + let distinct: std::collections::BTreeSet<&str> = label_refs.into_iter().collect(); + self.count(|c| &c.labels_added, distinct.len()); Ok(id) } diff --git a/crates/grafeo-core/src/execution/parallel/merge.rs b/crates/grafeo-core/src/execution/parallel/merge.rs index 8081b6179..bdbecd544 100644 --- a/crates/grafeo-core/src/execution/parallel/merge.rs +++ b/crates/grafeo-core/src/execution/parallel/merge.rs @@ -4,6 +4,7 @@ //! each worker produces partial results that must be merged into final output. use crate::execution::chunk::DataChunk; +use crate::execution::operators::value_utils::order_by; use crate::execution::vector::ValueVector; use grafeo_common::types::Value; use std::cmp::Ordering; @@ -199,7 +200,7 @@ pub struct SortKey { pub column: usize, /// Sort direction (ascending = true). pub ascending: bool, - /// Nulls first (true) or last (false). + /// Nulls first (true) or last (false), in either direction. pub nulls_first: bool, } @@ -238,17 +239,12 @@ struct MergeEntry { impl MergeEntry { fn compare_to(&self, other: &Self) -> Ordering { for key in &self.keys { - let a = self.row.get(key.column); - let b = other.row.get(key.column); - - let ordering = compare_values_for_sort(a, b, key.nulls_first); - - let ordering = if key.ascending { - ordering - } else { - ordering.reverse() - }; - + let ordering = order_by( + self.row.get(key.column), + other.row.get(key.column), + !key.ascending, + key.nulls_first, + ); if ordering != Ordering::Equal { return ordering; } @@ -278,40 +274,6 @@ impl Ord for MergeEntry { } } -fn compare_values_for_sort(a: Option<&Value>, b: Option<&Value>, nulls_first: bool) -> Ordering { - match (a, b) { - (None, None) | (Some(Value::Null), Some(Value::Null)) => Ordering::Equal, - (None, _) | (Some(Value::Null), _) => { - if nulls_first { - Ordering::Less - } else { - Ordering::Greater - } - } - (_, None) | (_, Some(Value::Null)) => { - if nulls_first { - Ordering::Greater - } else { - Ordering::Less - } - } - (Some(a), Some(b)) => compare_values(a, b), - } -} - -fn compare_values(a: &Value, b: &Value) -> Ordering { - match (a, b) { - (Value::Bool(a), Value::Bool(b)) => a.cmp(b), - (Value::Int64(a), Value::Int64(b)) => a.cmp(b), - (Value::Float64(a), Value::Float64(b)) => a.partial_cmp(b).unwrap_or(Ordering::Equal), - (Value::String(a), Value::String(b)) => a.cmp(b), - (Value::Timestamp(a), Value::Timestamp(b)) => a.cmp(b), - (Value::Date(a), Value::Date(b)) => a.cmp(b), - (Value::Time(a), Value::Time(b)) => a.cmp(b), - _ => Ordering::Equal, - } -} - /// Merges multiple sorted runs into a single sorted output. /// /// Uses a min-heap for efficient k-way merge. @@ -582,6 +544,45 @@ mod tests { assert_eq!(result[5][0], Value::Int64(1)); } + /// Runs of mixed values merge in the sort order: numbers before strings + /// when descending, and nulls last when the key says so. + #[test] + fn test_merge_sorted_runs_mixed_values_nulls_last_descending() { + let runs = vec![ + vec![ + vec![Value::Int64(3)], + vec![Value::String("b".into())], + vec![Value::Null], + ], + vec![ + vec![Value::Float64(2.5)], + vec![Value::String("a".into())], + vec![Value::Null], + ], + ]; + let keys = vec![SortKey { + column: 0, + ascending: false, + nulls_first: false, + }]; + + let result: Vec = merge_sorted_runs(runs, &keys) + .into_iter() + .map(|mut row| row.remove(0)) + .collect(); + assert_eq!( + result, + [ + Value::Int64(3), + Value::Float64(2.5), + Value::String("b".into()), + Value::String("a".into()), + Value::Null, + Value::Null, + ] + ); + } + #[test] fn test_rows_to_chunks() { let rows = (0..10).map(|i| vec![Value::Int64(i)]).collect(); diff --git a/crates/grafeo-core/src/execution/pipeline_convert.rs b/crates/grafeo-core/src/execution/pipeline_convert.rs index f7878a90a..b57d3ace9 100644 --- a/crates/grafeo-core/src/execution/pipeline_convert.rs +++ b/crates/grafeo-core/src/execution/pipeline_convert.rs @@ -127,12 +127,14 @@ fn decompose_recursive_memory( let sort = any .downcast::() .expect("name() returned 'Sort' but downcast failed"); - let (child, sort_keys) = sort.into_parts(); + let (child, sort_keys, output_width) = sort.into_parts(); let push_keys: Vec<_> = sort_keys.iter().map(convert_sort_key).collect(); - push_ops.push(Box::new(SpillableSortPushOperator::with_memory_context( - push_keys, - ctx.clone(), - ))); + let mut push_sort = + SpillableSortPushOperator::with_memory_context(push_keys, ctx.clone()); + if let Some(width) = output_width { + push_sort = push_sort.with_output_width(width); + } + push_ops.push(Box::new(push_sort)); decompose_recursive_memory(child, push_ops, ctx) } "HashAggregate" => { @@ -208,9 +210,13 @@ fn decompose_recursive( let sort = any .downcast::() .expect("name() returned 'Sort' but downcast failed"); - let (child, sort_keys) = sort.into_parts(); + let (child, sort_keys, output_width) = sort.into_parts(); let push_keys: Vec<_> = sort_keys.iter().map(convert_sort_key).collect(); - push_ops.push(Box::new(SortPushOperator::new(push_keys))); + let mut push_sort = SortPushOperator::new(push_keys); + if let Some(width) = output_width { + push_sort = push_sort.with_output_width(width); + } + push_ops.push(Box::new(push_sort)); decompose_recursive(child, push_ops) } "HashAggregate" => { @@ -305,6 +311,119 @@ mod tests { } } + /// A test operator that produces one chunk: nodes 7, 8 and 9 with the + /// sort keys 3, 1 and 2 in a second column. + struct NodesWithKeys { + emitted: bool, + } + + impl Operator for NodesWithKeys { + fn next(&mut self) -> OperatorResult { + if self.emitted { + return Ok(None); + } + self.emitted = true; + let mut builder = crate::execution::chunk::DataChunkBuilder::new(&[ + LogicalType::Node, + LogicalType::Int64, + ]); + for (id, key) in [(7_u64, 3_i64), (8, 1), (9, 2)] { + builder + .column_mut(0) + .unwrap() + .push_node_id(grafeo_common::types::NodeId::new(id)); + builder.column_mut(1).unwrap().push_int64(key); + builder.advance_row(); + } + Ok(Some(builder.finish())) + } + + fn reset(&mut self) { + self.emitted = false; + } + + fn name(&self) -> &'static str { + "NodesWithKeys" + } + + fn into_any(self: Box) -> Box { + self + } + } + + /// The nodes of [`NodesWithKeys`] sorted by their key, which the sort + /// drops: like `RETURN n ORDER BY n.key`. + fn nodes_sorted_by_a_dropped_key() -> Box { + let scan = Box::new(NodesWithKeys { emitted: false }); + Box::new(SortOperator::new(scan, vec![SortKey::ascending(1)]).with_output_width(1)) + } + + /// Runs a converted pipeline and returns the node IDs it produces, after + /// checking that each chunk holds one node column. + fn run_for_node_ids( + source: Box, + push_ops: Vec>, + ) -> Vec { + use crate::execution::pipeline::Pipeline; + use crate::execution::sink::CollectorSink; + use crate::execution::source::OperatorSource; + + let source = Box::new(OperatorSource::new(source)); + let mut pipeline = Pipeline::new(source, push_ops, Box::new(CollectorSink::new())); + pipeline.execute().unwrap(); + let collector = pipeline + .into_sink() + .into_any() + .downcast::() + .unwrap(); + let mut ids = Vec::new(); + for chunk in collector.into_chunks() { + assert_eq!(chunk.column_count(), 1); + let nodes = chunk.column(0).unwrap(); + assert_eq!(nodes.data_type(), &LogicalType::Node); + ids.extend( + chunk + .selected_indices() + .map(|row| nodes.get_node_id(row).unwrap().as_u64()), + ); + } + ids + } + + /// A sort that drops its trailing sort-key columns becomes a push sort + /// like any other (so it can spill), and returns only the leading column, + /// still a node column, in the order of the dropped key. + #[test] + fn convert_sort_that_drops_columns_produces_one_push_op() { + let (source, push_ops) = convert_to_pipeline(nodes_sorted_by_a_dropped_key()); + assert_eq!(source.name(), "NodesWithKeys"); + assert_eq!(push_ops.len(), 1); + assert!(push_ops[0].name().contains("Sort")); + assert_eq!(run_for_node_ids(source, push_ops), [8, 9, 7]); + } + + /// With a memory context the same sort is the spillable push sort. + #[test] + #[cfg(feature = "spill")] + fn convert_sort_that_drops_columns_with_memory_produces_a_spillable_sort() { + use crate::execution::memory::OperatorMemoryContext; + use crate::execution::spill::SpillManager; + use grafeo_common::memory::buffer::BufferManager; + use std::sync::Arc; + + let temp_dir = tempfile::TempDir::new().unwrap(); + let context = OperatorMemoryContext::new( + BufferManager::with_budget(1024 * 1024), + Arc::new(SpillManager::new(temp_dir.path()).unwrap()), + ); + let (source, push_ops) = + convert_to_pipeline_with_memory(nodes_sorted_by_a_dropped_key(), Some(context)); + assert_eq!(source.name(), "NodesWithKeys"); + assert_eq!(push_ops.len(), 1); + assert_eq!(push_ops[0].name(), "SpillableSortPush"); + assert_eq!(run_for_node_ids(source, push_ops), [8, 9, 7]); + } + #[test] fn convert_bare_scan_produces_empty_pipeline() { let scan: Box = Box::new(TestScanOperator::new()); @@ -336,8 +455,7 @@ mod tests { let scan: Box = Box::new(TestScanOperator::new()); let predicate: Box = Box::new(AlwaysTruePredicate); let filter: Box = Box::new(FilterOperator::new(scan, predicate)); - let limit: Box = - Box::new(LimitOperator::new(filter, 10, vec![LogicalType::Int64])); + let limit: Box = Box::new(LimitOperator::new(filter, 10)); let (source, push_ops) = convert_to_pipeline(limit); assert_eq!(source.name(), "TestScan"); @@ -351,8 +469,7 @@ mod tests { fn convert_sort_scan_produces_one_push_op() { let scan: Box = Box::new(TestScanOperator::new()); let keys = vec![SortKey::ascending(0)]; - let sort: Box = - Box::new(SortOperator::new(scan, keys, vec![LogicalType::Int64])); + let sort: Box = Box::new(SortOperator::new(scan, keys)); let (source, push_ops) = convert_to_pipeline(sort); assert_eq!(source.name(), "TestScan"); @@ -390,8 +507,7 @@ mod tests { #[test] fn convert_distinct_scan_produces_one_push_op() { let scan: Box = Box::new(TestScanOperator::new()); - let distinct: Box = - Box::new(DistinctOperator::new(scan, vec![LogicalType::Int64])); + let distinct: Box = Box::new(DistinctOperator::new(scan)); let (source, push_ops) = convert_to_pipeline(distinct); assert_eq!(source.name(), "TestScan"); @@ -402,11 +518,7 @@ mod tests { #[test] fn convert_distinct_on_columns_scan() { let scan: Box = Box::new(TestScanOperator::new()); - let distinct: Box = Box::new(DistinctOperator::on_columns( - scan, - vec![0], - vec![LogicalType::Int64], - )); + let distinct: Box = Box::new(DistinctOperator::on_columns(scan, vec![0])); let (source, push_ops) = convert_to_pipeline(distinct); assert_eq!(source.name(), "TestScan"); @@ -420,10 +532,8 @@ mod tests { let predicate: Box = Box::new(AlwaysTruePredicate); let filter: Box = Box::new(FilterOperator::new(scan, predicate)); let keys = vec![SortKey::ascending(0)]; - let sort: Box = - Box::new(SortOperator::new(filter, keys, vec![LogicalType::Int64])); - let limit: Box = - Box::new(LimitOperator::new(sort, 5, vec![LogicalType::Int64])); + let sort: Box = Box::new(SortOperator::new(filter, keys)); + let limit: Box = Box::new(LimitOperator::new(sort, 5)); let (source, push_ops) = convert_to_pipeline(limit); assert_eq!(source.name(), "TestScan"); @@ -445,8 +555,7 @@ mod tests { let predicate: Box = Box::new(AlwaysTruePredicate); let filter: Box = Box::new(FilterOperator::new(scan, predicate)); let keys = vec![SortKey::ascending(0)]; - let sort: Box = - Box::new(SortOperator::new(filter, keys, vec![LogicalType::Int64])); + let sort: Box = Box::new(SortOperator::new(filter, keys)); // Convert to pipeline let (source, push_ops) = convert_to_pipeline(sort); @@ -509,11 +618,7 @@ mod tests { // Build: Scan -> Distinct(on column 0) let scan: Box = Box::new(TestScanOperator::new()); - let distinct: Box = Box::new(DistinctOperator::on_columns( - scan, - vec![0], - vec![LogicalType::Int64], - )); + let distinct: Box = Box::new(DistinctOperator::on_columns(scan, vec![0])); let (source, push_ops) = convert_to_pipeline(distinct); assert_eq!(push_ops.len(), 1); diff --git a/crates/grafeo-core/src/execution/spill/external_sort.rs b/crates/grafeo-core/src/execution/spill/external_sort.rs index eeafb57cc..6e2e7deb4 100644 --- a/crates/grafeo-core/src/execution/spill/external_sort.rs +++ b/crates/grafeo-core/src/execution/spill/external_sort.rs @@ -7,6 +7,7 @@ use super::file::{SpillFile, SpillFileReader}; use super::manager::SpillManager; use super::serializer::{deserialize_row, serialize_row}; +use crate::execution::operators::value_utils::order_by; use grafeo_common::types::Value; use std::cmp::Ordering; use std::collections::BinaryHeap; @@ -22,7 +23,7 @@ pub enum SortDirection { Descending, } -/// Null handling in sort. +/// Where nulls go, in either sort direction. #[derive(Debug, Clone, Copy, PartialEq, Eq)] #[non_exhaustive] pub enum NullOrder { @@ -359,28 +360,12 @@ impl PartialOrd for HeapEntry { /// Compares two rows by sort keys. fn compare_rows(a: &[Value], b: &[Value], keys: &[SortKey]) -> Ordering { for key in keys { - let a_val = a.get(key.column); - let b_val = b.get(key.column); - - let ordering = match (a_val, b_val) { - (Some(Value::Null), Some(Value::Null)) => Ordering::Equal, - (Some(Value::Null), _) => match key.null_order { - NullOrder::First => Ordering::Less, - NullOrder::Last => Ordering::Greater, - }, - (_, Some(Value::Null)) => match key.null_order { - NullOrder::First => Ordering::Greater, - NullOrder::Last => Ordering::Less, - }, - (Some(a), Some(b)) => compare_values(a, b), - _ => Ordering::Equal, - }; - - let ordering = match key.direction { - SortDirection::Ascending => ordering, - SortDirection::Descending => ordering.reverse(), - }; - + let ordering = order_by( + a.get(key.column), + b.get(key.column), + key.direction == SortDirection::Descending, + key.null_order == NullOrder::First, + ); if ordering != Ordering::Equal { return ordering; } @@ -389,20 +374,6 @@ fn compare_rows(a: &[Value], b: &[Value], keys: &[SortKey]) -> Ordering { Ordering::Equal } -/// Compares two values. -fn compare_values(a: &Value, b: &Value) -> Ordering { - match (a, b) { - (Value::Bool(a), Value::Bool(b)) => a.cmp(b), - (Value::Int64(a), Value::Int64(b)) => a.cmp(b), - (Value::Float64(a), Value::Float64(b)) => a.partial_cmp(b).unwrap_or(Ordering::Equal), - (Value::String(a), Value::String(b)) => a.cmp(b), - (Value::Timestamp(a), Value::Timestamp(b)) => a.cmp(b), - (Value::Date(a), Value::Date(b)) => a.cmp(b), - (Value::Time(a), Value::Time(b)) => a.cmp(b), - _ => Ordering::Equal, - } -} - /// Adapter to write to SpillFile through std::io::Write. struct SpillFileWriter<'a>(&'a mut SpillFile); diff --git a/crates/grafeo-core/src/graph/compact/graph_store_impl.rs b/crates/grafeo-core/src/graph/compact/graph_store_impl.rs index b86602624..9d9791a39 100644 --- a/crates/grafeo-core/src/graph/compact/graph_store_impl.rs +++ b/crates/grafeo-core/src/graph/compact/graph_store_impl.rs @@ -269,6 +269,11 @@ impl GraphStore for CompactStore { .cloned() } + /// The store is immutable: no transaction has edges of its own in it. + fn edge_type_versioned(&self, id: EdgeId, _: EpochId, _: TransactionId) -> Option { + self.edge_type(id) + } + fn find_nodes_by_property(&self, property: &str, value: &Value) -> Vec { let key = PropertyKey::new(property); let mut results = Vec::new(); diff --git a/crates/grafeo-core/src/graph/compact/layered.rs b/crates/grafeo-core/src/graph/compact/layered.rs index 5a441ca73..712b69bf3 100644 --- a/crates/grafeo-core/src/graph/compact/layered.rs +++ b/crates/grafeo-core/src/graph/compact/layered.rs @@ -744,6 +744,28 @@ impl GraphStore for LayeredStore { .or_else(|| self.overlay.load().edge_type(id)) } + fn edge_type_versioned( + &self, + id: EdgeId, + epoch: EpochId, + transaction_id: TransactionId, + ) -> Option { + if self.is_edge_deleted_from_base(id) { + return None; + } + if self.is_edge_dirty(id) { + return self + .overlay + .load() + .edge_type_versioned(id, epoch, transaction_id); + } + self.base.load().edge_type(id).or_else(|| { + self.overlay + .load() + .edge_type_versioned(id, epoch, transaction_id) + }) + } + fn has_property_index(&self, property: &str) -> bool { // Property indexes only live on the overlay LpgStore (the columnar // base has no index store). Without this delegate the trait default diff --git a/crates/grafeo-core/src/graph/lpg/mod.rs b/crates/grafeo-core/src/graph/lpg/mod.rs index 5f34b70b4..349ac61d6 100644 --- a/crates/grafeo-core/src/graph/lpg/mod.rs +++ b/crates/grafeo-core/src/graph/lpg/mod.rs @@ -12,6 +12,7 @@ //! //! Start with [`LpgStore`] - that's where everything lives. +#[cfg(feature = "lpg")] pub(crate) mod block; mod edge; mod node; diff --git a/crates/grafeo-core/src/graph/lpg/store/versioning.rs b/crates/grafeo-core/src/graph/lpg/store/versioning.rs index f2cca9593..3122bc2a5 100644 --- a/crates/grafeo-core/src/graph/lpg/store/versioning.rs +++ b/crates/grafeo-core/src/graph/lpg/store/versioning.rs @@ -371,7 +371,10 @@ impl LpgStore { /// Freezes an epoch from hot (arena) storage to cold (compressed) storage. /// /// This is called by the transaction manager when an epoch becomes eligible - /// for freezing (no active transactions can see it). The freeze process: + /// for freezing (no active transactions can see it). The epoch must be + /// finished: a write still in progress at it, such as a batch that has + /// allocated its records but not yet indexed them, is not frozen and its + /// records stay hot. The freeze process: /// /// 1. Collects all hot version refs for the epoch /// 2. Reads the corresponding records from arena diff --git a/crates/grafeo-core/src/graph/rdf/graph_store_adapter.rs b/crates/grafeo-core/src/graph/rdf/graph_store_adapter.rs index f4598190c..aabdb610c 100644 --- a/crates/grafeo-core/src/graph/rdf/graph_store_adapter.rs +++ b/crates/grafeo-core/src/graph/rdf/graph_store_adapter.rs @@ -494,6 +494,11 @@ impl GraphStore for RdfGraphStoreAdapter { .map(|(_, _, t)| t.clone()) } + /// A read-only view without versions: every edge is the committed one. + fn edge_type_versioned(&self, id: EdgeId, _: EpochId, _: TransactionId) -> Option { + self.edge_type(id) + } + // --- Filtered search --- fn find_nodes_by_property(&self, property: &str, value: &Value) -> Vec { diff --git a/crates/grafeo-core/src/graph/traits.rs b/crates/grafeo-core/src/graph/traits.rs index 0102c5df4..1dcbd0616 100644 --- a/crates/grafeo-core/src/graph/traits.rs +++ b/crates/grafeo-core/src/graph/traits.rs @@ -162,17 +162,22 @@ pub trait GraphStore: Send + Sync { /// Returns the type string of an edge. fn edge_type(&self, id: EdgeId) -> Option; - /// Returns the type string of an edge visible to a specific transaction. + /// Returns the type string of an edge visible to a specific transaction, + /// including the edges it created and has not committed yet. /// - /// Falls back to epoch-based `edge_type` if not overridden. + /// The default reads the whole edge with + /// [`get_edge_versioned`](Self::get_edge_versioned); stores override it + /// with a cheaper lookup. It must not fall back to + /// [`edge_type`](Self::edge_type), which reads the committed state and + /// misses the transaction's own edges. fn edge_type_versioned( &self, id: EdgeId, epoch: EpochId, transaction_id: TransactionId, ) -> Option { - let _ = (epoch, transaction_id); - self.edge_type(id) + self.get_edge_versioned(id, epoch, transaction_id) + .map(|edge| edge.edge_type) } // --- Index introspection --- diff --git a/crates/grafeo-engine/src/cdc.rs b/crates/grafeo-engine/src/cdc.rs index 1bf0b762e..b6e2edd02 100644 --- a/crates/grafeo-engine/src/cdc.rs +++ b/crates/grafeo-engine/src/cdc.rs @@ -24,7 +24,8 @@ //! # Thread safety and ordering //! //! [`CdcLog`] is the authoritative in-memory store, a -//! `RwLock>>`. Readers (history +//! `RwLock>>` (entity ids +//! repeat across graphs, so the graph is part of the key). Readers (history //! queries, retention probes) take the read lock; writers (commit-path //! recording) take the write lock. We use [`hashbrown`]'s `HashMap` for //! the Fx-hashed inner map, and [`parking_lot`]'s `RwLock` for cheap @@ -195,6 +196,11 @@ impl EntityId { pub struct ChangeEvent { /// The entity that was changed. pub entity_id: EntityId, + /// The graph the entity is in: its storage key (`name`, or `schema/name` + /// inside a schema), `None` for the default graph. Entity ids repeat + /// across graphs, so an id names an entity only together with this. + #[serde(default, skip_serializing_if = "Option::is_none")] + pub graph: Option, /// The kind of change. pub kind: ChangeKind, /// MVCC epoch when the change occurred. @@ -276,7 +282,7 @@ impl Default for CdcRetentionConfig { /// [#250]: https://github.com/GrafeoDB/grafeo/issues/250 #[derive(Debug)] pub struct CdcLog { - events: RwLock>>, + events: RwLock, EntityId), Vec>>, clock: Arc, retention: CdcRetentionConfig, } @@ -318,7 +324,7 @@ impl CdcLog { pub fn record(&self, event: ChangeEvent) { self.events .write() - .entry(event.entity_id) + .entry((event.graph.clone(), event.entity_id)) .or_default() .push(event); } @@ -327,7 +333,10 @@ impl CdcLog { pub fn record_batch(&self, events: impl IntoIterator) { let mut guard = self.events.write(); for event in events { - guard.entry(event.entity_id).or_default().push(event); + guard + .entry((event.graph.clone(), event.entity_id)) + .or_default() + .push(event); } } @@ -341,6 +350,7 @@ impl CdcLog { ) { self.record(ChangeEvent { entity_id: EntityId::Node(id), + graph: None, kind: ChangeKind::Create, epoch, timestamp: self.clock.now(), @@ -370,6 +380,7 @@ impl CdcLog { ) { self.record(ChangeEvent { entity_id: EntityId::Edge(id), + graph: None, kind: ChangeKind::Create, epoch, timestamp: self.clock.now(), @@ -402,6 +413,7 @@ impl CdcLog { let id = triple_hash(subject, predicate, object, graph); self.record(ChangeEvent { entity_id: EntityId::Triple(id), + graph: None, kind: ChangeKind::Create, epoch, timestamp: self.clock.now(), @@ -433,6 +445,7 @@ impl CdcLog { let id = triple_hash(subject, predicate, object, graph); self.record(ChangeEvent { entity_id: EntityId::Triple(id), + graph: None, kind: ChangeKind::Delete, epoch, timestamp: self.clock.now(), @@ -469,6 +482,7 @@ impl CdcLog { self.record(ChangeEvent { entity_id, + graph: None, kind: ChangeKind::Update, epoch, timestamp: self.clock.now(), @@ -495,6 +509,7 @@ impl CdcLog { ) { self.record(ChangeEvent { entity_id, + graph: None, kind: ChangeKind::Delete, epoch, timestamp: self.clock.now(), @@ -512,22 +527,43 @@ impl CdcLog { }); } - /// Returns all change events for an entity, ordered by epoch. + /// Returns all change events for an entity of the default graph, ordered + /// by epoch. #[must_use] pub fn history(&self, entity_id: EntityId) -> Vec { + self.history_in(None, entity_id) + } + + /// Returns all change events for an entity of `graph` (its storage key, + /// `None` for the default graph), ordered by epoch. + #[must_use] + pub fn history_in(&self, graph: Option<&str>, entity_id: EntityId) -> Vec { self.events .read() - .get(&entity_id) + .get(&(graph.map(str::to_string), entity_id)) .cloned() .unwrap_or_default() } - /// Returns change events for an entity since the given epoch. + /// Returns change events for an entity of the default graph since the + /// given epoch. #[must_use] pub fn history_since(&self, entity_id: EntityId, since_epoch: EpochId) -> Vec { + self.history_since_in(None, entity_id, since_epoch) + } + + /// Returns change events for an entity of `graph` (`None` for the default + /// graph) since the given epoch. + #[must_use] + pub fn history_since_in( + &self, + graph: Option<&str>, + entity_id: EntityId, + since_epoch: EpochId, + ) -> Vec { self.events .read() - .get(&entity_id) + .get(&(graph.map(str::to_string), entity_id)) .map(|events| { events .iter() @@ -538,7 +574,8 @@ impl CdcLog { .unwrap_or_default() } - /// Returns all change events across all entities in an epoch range. + /// Returns all change events across all entities and graphs in an epoch + /// range; each event names its graph. #[must_use] pub fn changes_between(&self, start_epoch: EpochId, end_epoch: EpochId) -> Vec { let guard = self.events.read(); @@ -744,13 +781,15 @@ fn triple_hash(subject: &str, predicate: &str, object: &str, graph: Option<&str> /// properties and labels one by one; without this, a consumer saw a create /// event without properties followed by one update per property. /// Changes to entities that existed before the transaction stay as they are. +/// Entity ids repeat across graphs, so the folding stays within one graph. pub(crate) fn fold_into_creates(events: Vec) -> Vec { let mut folded: Vec = Vec::with_capacity(events.len()); - let mut created: HashMap = HashMap::new(); + let mut created: HashMap<(Option, EntityId), usize> = HashMap::new(); for event in events { - let Some(&at) = created.get(&event.entity_id) else { + let key = (event.graph.clone(), event.entity_id); + let Some(&at) = created.get(&key) else { if event.kind == ChangeKind::Create { - created.insert(event.entity_id, folded.len()); + created.insert(key, folded.len()); } folded.push(event); continue; diff --git a/crates/grafeo-engine/src/config.rs b/crates/grafeo-engine/src/config.rs index 1f10c0697..4b87c20a6 100644 --- a/crates/grafeo-engine/src/config.rs +++ b/crates/grafeo-engine/src/config.rs @@ -233,7 +233,9 @@ pub struct Config { /// /// Without `ORDER BY` the row order is unspecified: it can change between /// runs, builds and versions. This test option makes that visible, so - /// tests find code that relies on an order anyway. Default: `false`. + /// tests find code that relies on an order anyway. A streamed result is + /// shuffled within each chunk, so the stream keeps its bounded memory. + /// Default: `false`. pub shuffle_unordered: bool, /// WAL durability mode. Only used when `wal_enabled` is true. @@ -527,8 +529,8 @@ impl Config { self } - /// Returns the rows of every query without `ORDER BY` in random order, - /// a test option that finds code relying on a row order that is + /// Sets whether queries without `ORDER BY` return their rows in random + /// order, a test option that finds code relying on a row order that is /// unspecified (see [`Config::shuffle_unordered`]). #[must_use] pub fn with_shuffle_unordered(mut self, shuffle: bool) -> Self { diff --git a/crates/grafeo-engine/src/database/admin.rs b/crates/grafeo-engine/src/database/admin.rs index 0872b0718..9162cafe8 100644 --- a/crates/grafeo-engine/src/database/admin.rs +++ b/crates/grafeo-engine/src/database/admin.rs @@ -2,7 +2,10 @@ use std::path::Path; +use grafeo_common::types::ArcStr; use grafeo_common::utils::error::Result; +use grafeo_common::utils::hash::FxHashMap; +use grafeo_core::graph::Direction; impl super::GrafeoDB { // ========================================================================= @@ -24,19 +27,20 @@ impl super::GrafeoDB { /// Returns the number of distinct labels in the database. #[must_use] pub fn label_count(&self) -> usize { - self.lpg_store().label_count() + self.graph_store().all_labels().len() } - /// Returns the number of distinct property keys in the database. + /// Returns the number of distinct property keys in the database, of + /// nodes and edges together. #[must_use] pub fn property_key_count(&self) -> usize { - self.lpg_store().property_key_count() + self.graph_store().all_property_keys().len() } /// Returns the number of distinct edge types in the database. #[must_use] pub fn edge_type_count(&self) -> usize { - self.lpg_store().edge_type_count() + self.graph_store().all_edge_types().len() } // ========================================================================= @@ -209,9 +213,9 @@ impl super::GrafeoDB { crate::admin::DatabaseStats { node_count: self.graph_store().node_count(), edge_count: self.graph_store().edge_count(), - label_count: self.lpg_store().label_count(), - edge_type_count: self.lpg_store().edge_type_count(), - property_key_count: self.lpg_store().property_key_count(), + label_count: self.label_count(), + edge_type_count: self.edge_type_count(), + property_key_count: self.property_key_count(), index_count: self.catalog.index_count(), memory_bytes: self.memory_usage().total_bytes, disk_bytes, @@ -245,27 +249,47 @@ impl super::GrafeoDB { /// For RDF mode, returns predicate and named graph information. #[must_use] pub fn schema(&self) -> crate::admin::SchemaInfo { - let labels = self - .lpg_store() + let store = self.graph_store(); + // The label index holds every node with the label, also those of a + // transaction that has not committed: count the nodes that have the + // label at the current epoch. + let epoch = store.current_epoch(); + let labels = store .all_labels() .into_iter() .map(|name| crate::admin::LabelInfo { - name: name.clone(), - count: self.lpg_store().nodes_with_label(&name).count(), + count: store + .nodes_by_label(&name) + .into_iter() + .filter(|&id| { + store + .get_node_at_epoch(id, epoch) + .is_some_and(|node| node.has_label(&name)) + }) + .count(), + name, }) .collect(); - let edge_types = self - .lpg_store() + // One pass over the edges counts every type. + let mut edges_per_type: FxHashMap = FxHashMap::default(); + for node in store.node_ids() { + for (_, edge) in store.edges_from(node, Direction::Outgoing) { + if let Some(edge_type) = store.edge_type(edge) { + *edges_per_type.entry(edge_type).or_default() += 1; + } + } + } + let edge_types = store .all_edge_types() .into_iter() .map(|name| crate::admin::EdgeTypeInfo { - name: name.clone(), - count: self.lpg_store().edges_with_type(&name).count(), + count: edges_per_type.get(name.as_str()).copied().unwrap_or(0), + name, }) .collect(); - let property_keys = self.lpg_store().all_property_keys(); + let property_keys = store.all_property_keys(); crate::admin::SchemaInfo::Lpg(crate::admin::LpgSchemaInfo { labels, @@ -442,7 +466,8 @@ impl super::GrafeoDB { .store(enabled, std::sync::atomic::Ordering::Relaxed); } - /// Returns the full change history for an entity (node or edge). + /// Returns the full change history for an entity (node or edge) of the + /// default graph (a session's `history` reads its current graph). /// /// Events are ordered chronologically by epoch. /// @@ -457,7 +482,8 @@ impl super::GrafeoDB { Ok(self.cdc_log.history(entity_id.into())) } - /// Returns change events for an entity since the given epoch. + /// Returns change events for an entity of the default graph since the + /// given epoch. /// /// # Errors /// @@ -471,7 +497,8 @@ impl super::GrafeoDB { Ok(self.cdc_log.history_since(entity_id.into(), since_epoch)) } - /// Returns all change events across all entities in an epoch range. + /// Returns all change events across all entities and graphs in an epoch + /// range; each event names its graph. /// /// # Errors /// diff --git a/crates/grafeo-engine/src/database/cdc_store.rs b/crates/grafeo-engine/src/database/cdc_store.rs index 0289fcc3c..4314ea00d 100644 --- a/crates/grafeo-engine/src/database/cdc_store.rs +++ b/crates/grafeo-engine/src/database/cdc_store.rs @@ -41,6 +41,9 @@ pub(crate) struct CdcGraphStore { /// Whether the events of non-versioned writes are buffered too, instead /// of being recorded as they happen. buffer_all: bool, + /// The named graph this store writes to, `None` for the default graph: + /// the events it buffers carry it. + graph: Option, } impl CdcGraphStore { @@ -51,6 +54,7 @@ impl CdcGraphStore { cdc_log, pending_events: Arc::new(Mutex::new(Vec::new())), buffer_all: false, + graph: None, } } @@ -69,9 +73,18 @@ impl CdcGraphStore { cdc_log, pending_events, buffer_all: false, + graph: None, } } + /// The same store, for the named graph `graph`: the events it buffers + /// say so, so the commit folds them per graph. + #[must_use] + pub fn for_graph(mut self, graph: String) -> Self { + self.graph = Some(graph); + self + } + /// Wraps a store sharing an existing event buffer, and buffers the events /// of non-versioned writes there too: for a direct write outside any /// transaction, which records all of its events at once, merged and @@ -86,6 +99,7 @@ impl CdcGraphStore { cdc_log, pending_events, buffer_all: true, + graph: None, } } @@ -101,15 +115,17 @@ impl CdcGraphStore { /// each transaction's events get the unique epoch from `fetch_add(1, SeqCst)`. fn buffer_event(&self, mut event: ChangeEvent) { event.epoch = EpochId::PENDING; + event.graph.clone_from(&self.graph); self.pending_events.lock().push(event); } /// Records a CDC event directly (for non-versioned/auto-commit mutations), /// or buffers it when the store buffers every event. - fn record_directly(&self, event: ChangeEvent) { + fn record_directly(&self, mut event: ChangeEvent) { if self.buffer_all { self.buffer_event(event); } else { + event.graph.clone_from(&self.graph); self.cdc_log.record(event); } } @@ -159,6 +175,7 @@ fn make_event( ) -> ChangeEvent { ChangeEvent { entity_id, + graph: None, kind, epoch, timestamp, @@ -347,6 +364,15 @@ impl GraphStore for CdcGraphStore { self.inner.edge_type(id) } + fn edge_type_versioned( + &self, + id: EdgeId, + epoch: EpochId, + transaction_id: TransactionId, + ) -> Option { + self.inner.edge_type_versioned(id, epoch, transaction_id) + } + fn has_property_index(&self, property: &str) -> bool { self.inner.has_property_index(property) } diff --git a/crates/grafeo-engine/src/database/direct.rs b/crates/grafeo-engine/src/database/direct.rs index 766e991bb..0efa597ed 100644 --- a/crates/grafeo-engine/src/database/direct.rs +++ b/crates/grafeo-engine/src/database/direct.rs @@ -7,9 +7,11 @@ //! with, and it writes as the system, stamped at its new epoch, which it //! publishes when done. Every check of a single call runs before it writes, //! so the call cannot half-apply; a batch writes as a private transaction -//! instead, which is undone when a later row fails. While a transaction is -//! open, the call runs as an implicit transaction of a session and is checked -//! for conflicts with the open one. +//! instead, which is undone when a later row fails. A call that fails still +//! uses up its epoch: the stores take it before the write, to stamp what they +//! record themselves, and it is not handed out twice. The gap it leaves holds +//! no data. While a transaction is open, the call runs as an implicit +//! transaction of a session and is checked for conflicts with the open one. //! //! This keeps a direct call close to the cost of the store write itself. Once //! transactions own their change set (#448), every direct call becomes an @@ -19,7 +21,7 @@ use std::collections::HashMap; use std::sync::Arc; use std::sync::atomic::Ordering; -use grafeo_common::types::{EdgeId, EpochId, NodeId, PropertyKey, TransactionId, Value}; +use grafeo_common::types::{EdgeId, EpochId, NodeId, PropertyKey, Value}; use grafeo_common::utils::error::{Error, QueryError, QueryErrorKind, Result}; use grafeo_core::execution::operators::{GraphWriter, OperatorError}; use grafeo_core::graph::lpg::{Edge, LpgStore, Node}; @@ -45,6 +47,7 @@ pub(crate) enum DirectTarget<'a> { /// Buffers the direct calls outside a transaction share. Only the call /// holding the transaction manager's idle gate uses them. +#[cfg(any(feature = "wal", feature = "cdc"))] #[derive(Default)] pub(crate) struct ImplicitWrites { /// The WAL records of the running call, written as one group. @@ -53,6 +56,10 @@ pub(crate) struct ImplicitWrites { /// The CDC events of the running call, recorded at its epoch. #[cfg(feature = "cdc")] cdc_events: Arc>>, + /// Held while a direct call on a compacted database builds its WAL + /// records from the state and writes them (see `log_compacted_write`). + #[cfg(all(feature = "wal", feature = "compact-store"))] + compacted_log: parking_lot::Mutex<()>, } /// What a direct call changed. A compacted database's sessions write the @@ -305,6 +312,11 @@ impl GrafeoDB { /// Writes the WAL records of a direct call that a compacted database's /// session made, from the state it left, as one group. + /// + /// Reading the state and writing the records happen under one lock, so + /// every call reads the state after all calls that logged before it: the + /// last group in the WAL holds the newest state, never an older one that + /// a call read before another call's commit and wrote after it. #[cfg(all(feature = "wal", feature = "compact-store"))] fn log_compacted_write(&self, target: DirectTarget<'_>, touched: Vec) { use grafeo_storage::wal::WalRecord; @@ -312,6 +324,7 @@ impl GrafeoDB { let (Some(_), Some(wal)) = (&self.layered_store, &self.wal) else { return; }; + let _logging = self.implicit_writes.compacted_log.lock(); let Ok(Some((graph_store, graph))) = self.direct_store(target) else { return; }; @@ -330,7 +343,7 @@ impl GrafeoDB { } if let Err(e) = buffer.flush(&[ WalRecord::TransactionCommit { - transaction_id: TransactionId::SYSTEM, + transaction_id: grafeo_common::types::TransactionId::SYSTEM, }, WalRecord::EpochAdvance { epoch: self.transaction_manager.current_epoch(), @@ -364,7 +377,7 @@ impl GrafeoDB { epoch }; - let mut target: Arc = Arc::clone(store) as Arc; + let target: Arc = Arc::clone(store) as Arc; #[cfg(feature = "wal")] let wal = self.wal.as_ref().map(|wal| { Arc::clone(self.implicit_writes.wal.get_or_init(|| { @@ -382,27 +395,32 @@ impl GrafeoDB { #[cfg(feature = "cdc")] self.implicit_writes.cdc_events.lock().clear(); #[cfg(feature = "wal")] - if let Some(buffer) = &wal { - use super::wal_store::WalGraphStore; - target = Arc::new(match graph { - None => WalGraphStore::new(Arc::clone(store), Arc::clone(buffer)), - Some(name) => WalGraphStore::new_for_graph( - Arc::clone(store), - Arc::clone(buffer), - name.to_string(), - ), - }); - } + let target: Arc = match &wal { + Some(buffer) => { + use super::wal_store::WalGraphStore; + Arc::new(match graph { + None => WalGraphStore::new(Arc::clone(store), Arc::clone(buffer)), + Some(name) => WalGraphStore::new_for_graph( + Arc::clone(store), + Arc::clone(buffer), + name.to_string(), + ), + }) + } + None => target, + }; #[cfg(not(feature = "wal"))] let _ = graph; #[cfg(feature = "cdc")] - if self.cdc_active() { - target = Arc::new(super::cdc_store::CdcGraphStore::wrap_buffered( + let target: Arc = if self.cdc_active() { + Arc::new(super::cdc_store::CdcGraphStore::wrap_buffered( target, Arc::clone(&self.cdc_log), Arc::clone(&self.implicit_writes.cdc_events), - )); - } + )) + } else { + target + }; let validator = CatalogConstraintValidator::new(Arc::clone(&self.catalog)) .with_store(Arc::clone(store) as Arc) @@ -459,7 +477,8 @@ impl GrafeoDB { use grafeo_storage::wal::WalRecord; if let Err(e) = buffer.flush(&[ WalRecord::TransactionCommit { - transaction_id: transaction.unwrap_or(TransactionId::SYSTEM), + transaction_id: transaction + .unwrap_or(grafeo_common::types::TransactionId::SYSTEM), }, WalRecord::EpochAdvance { epoch }, ]) { @@ -565,14 +584,28 @@ impl DirectCalls<'_> { pub(crate) fn remove_node_property(&self, id: NodeId, key: &str) -> Result { self.write( - |writer| writer.remove_node_property(id, key), + // The direct API reports a missing node as `false`; a query that + // writes to one fails (see `GraphWriter`). + |writer| { + if writer.has_node(id) { + writer.remove_node_property(id, key) + } else { + Ok(false) + } + }, |_| vec![Touched::NodeProperty(id, key.to_string())], ) } pub(crate) fn remove_edge_property(&self, id: EdgeId, key: &str) -> Result { self.write( - |writer| writer.remove_edge_property(id, key), + |writer| { + if writer.has_edge(id) { + writer.remove_edge_property(id, key) + } else { + Ok(false) + } + }, |_| vec![Touched::EdgeProperty(id, key.to_string())], ) } @@ -702,6 +735,9 @@ pub(crate) fn add_node_label( id: NodeId, label: &str, ) -> std::result::Result { + if !writer.has_node(id) { + return Ok(false); + } Ok(writer.add_labels(id, &[label.to_string()])? == 1) } @@ -711,6 +747,9 @@ pub(crate) fn remove_node_label( id: NodeId, label: &str, ) -> std::result::Result { + if !writer.has_node(id) { + return Ok(false); + } Ok(writer.remove_labels(id, &[label.to_string()])? == 1) } diff --git a/crates/grafeo-engine/src/database/mod.rs b/crates/grafeo-engine/src/database/mod.rs index 3786a1595..52ccd3e8a 100644 --- a/crates/grafeo-engine/src/database/mod.rs +++ b/crates/grafeo-engine/src/database/mod.rs @@ -65,7 +65,7 @@ mod upsert; #[cfg(all(feature = "wal", feature = "lpg"))] pub(crate) mod wal_store; -use grafeo_common::{grafeo_error, grafeo_warn}; +use grafeo_common::grafeo_error; #[cfg(feature = "wal")] use std::path::Path; use std::sync::Arc; @@ -197,6 +197,7 @@ pub struct GrafeoDB { read_only: bool, /// Buffers of the direct calls made outside a transaction. #[cfg(feature = "lpg")] + #[cfg(any(feature = "wal", feature = "cdc"))] implicit_writes: direct::ImplicitWrites, /// Named graph projections (virtual subgraphs), shared with sessions. projections: @@ -239,12 +240,7 @@ impl GrafeoDB { /// Unlike [`graph_store()`](Self::graph_store) (which clones an `Arc`), /// this borrows from `self` — suitable for constructing accessors that /// need `&'a dyn GraphStore` tied to the database lifetime. - #[cfg(any( - feature = "vector-index", - feature = "text-index", - feature = "hybrid-search", - feature = "embed", - ))] + #[cfg(feature = "vector-index")] fn graph_store_ref(&self) -> &dyn grafeo_core::graph::GraphStore { if let Some(ref ext_read) = self.external_read_store { ext_read.as_ref() @@ -700,6 +696,7 @@ impl GrafeoDB { current_schema: RwLock::new(None), read_only: is_read_only, #[cfg(feature = "lpg")] + #[cfg(any(feature = "wal", feature = "cdc"))] implicit_writes: direct::ImplicitWrites::default(), projections: Arc::new(RwLock::new(std::collections::HashMap::new())), #[cfg(all(feature = "compact-store", feature = "lpg"))] @@ -835,6 +832,7 @@ impl GrafeoDB { current_schema: RwLock::new(None), read_only: false, #[cfg(feature = "lpg")] + #[cfg(any(feature = "wal", feature = "cdc"))] implicit_writes: direct::ImplicitWrites::default(), projections: Arc::new(RwLock::new(std::collections::HashMap::new())), #[cfg(all(feature = "compact-store", feature = "lpg"))] @@ -932,6 +930,7 @@ impl GrafeoDB { current_schema: RwLock::new(None), read_only: true, #[cfg(feature = "lpg")] + #[cfg(any(feature = "wal", feature = "cdc"))] implicit_writes: direct::ImplicitWrites::default(), projections: Arc::new(RwLock::new(std::collections::HashMap::new())), #[cfg(all(feature = "compact-store", feature = "lpg"))] @@ -1058,7 +1057,9 @@ impl GrafeoDB { self.buffer_manager.register_consumer(overlay_consumer); self.layered_store = Some(layered); - self.read_only = false; + // A database opened read-only stays read-only; an external read-only + // store becomes an owned copy that takes writes. + self.read_only = self.config.access_mode == crate::config::AccessMode::ReadOnly; self.query_cache = Arc::new(QueryCache::default()); self.projections.write().clear(); @@ -1586,6 +1587,7 @@ impl GrafeoDB { graph_model: self.config.graph_model, query_timeout: self.config.query_timeout, max_property_size: self.config.max_property_size, + #[cfg(feature = "spill")] buffer_manager: Some(Arc::clone(&self.buffer_manager)), commit_counter: Arc::clone(&self.commit_counter), gc_interval: self.config.gc_interval, @@ -2099,7 +2101,7 @@ impl GrafeoDB { if let Some(mut flusher) = self.wal_flusher.lock().take() && let Err(e) = flusher.shutdown() { - grafeo_warn!("failed to stop the WAL flusher: {e}"); + grafeo_common::grafeo_warn!("failed to stop the WAL flusher: {e}"); } // Read-only databases: just release the shared lock, no checkpointing @@ -2115,9 +2117,9 @@ impl GrafeoDB { // For single-file format: checkpoint to .grafeo file, then clean up sidecar WAL. // We must do this BEFORE the WAL close path because checkpoint_to_file // removes the sidecar WAL directory. - #[cfg(feature = "grafeo-file")] + #[cfg(all(feature = "wal", feature = "grafeo-file"))] let is_single_file = self.file_manager.is_some(); - #[cfg(not(feature = "grafeo-file"))] + #[cfg(all(feature = "wal", not(feature = "grafeo-file")))] let is_single_file = false; #[cfg(feature = "grafeo-file")] @@ -2151,7 +2153,7 @@ impl GrafeoDB { } fm.remove_sidecar_wal()?; } else { - grafeo_warn!( + grafeo_common::grafeo_warn!( "keeping sidecar WAL for recovery: checkpoint wrote 0 sections but WAL has records" ); } @@ -2206,7 +2208,7 @@ impl GrafeoDB { }, ]) { - grafeo_warn!("Failed to log a graph change to the WAL: {}", e); + grafeo_common::grafeo_warn!("Failed to log a graph change to the WAL: {}", e); } } diff --git a/crates/grafeo-engine/src/database/persistence.rs b/crates/grafeo-engine/src/database/persistence.rs index 5abb8ad0c..fd9bf5100 100644 --- a/crates/grafeo-engine/src/database/persistence.rs +++ b/crates/grafeo-engine/src/database/persistence.rs @@ -1,6 +1,6 @@ //! Persistence, snapshots, and data export for GrafeoDB. -#[cfg(any(feature = "wal", feature = "grafeo-file"))] +#[cfg(feature = "wal")] use std::path::Path; #[cfg(any(feature = "vector-index", feature = "text-index"))] @@ -751,8 +751,8 @@ impl super::GrafeoDB { Ok(()) } - /// Saves the database to a single `.grafeo` file. - #[cfg(feature = "grafeo-file")] + /// Saves the database to a single `.grafeo` file (see [`save`](Self::save)). + #[cfg(all(feature = "wal", feature = "grafeo-file"))] fn save_as_grafeo_file(&self, path: &Path) -> Result<()> { use grafeo_storage::file::GrafeoFileManager; diff --git a/crates/grafeo-engine/src/database/section_consumer.rs b/crates/grafeo-engine/src/database/section_consumer.rs index 48e996480..94fbcfed1 100644 --- a/crates/grafeo-engine/src/database/section_consumer.rs +++ b/crates/grafeo-engine/src/database/section_consumer.rs @@ -21,7 +21,6 @@ use std::sync::Arc; all(feature = "compact-store", feature = "lpg") ))] use std::sync::Weak; -use std::sync::atomic::{AtomicUsize, Ordering}; use grafeo_common::memory::buffer::{MemoryConsumer, MemoryRegion, SpillError, priorities}; use grafeo_common::storage::Section; @@ -73,7 +72,8 @@ pub struct SectionConsumer { /// Directory where this consumer writes spill files. `None` disables spilling. spill_path: Option, /// Counter for unique spill file names within `spill_path`. - file_counter: AtomicUsize, + #[cfg(feature = "wal")] + file_counter: std::sync::atomic::AtomicUsize, /// `true` after a successful `spill_to_dir`, cleared on reload. Drives /// `current_tier()` so introspection reports the actual state of /// sections that opted into the `swap_to_mmap` path. @@ -130,7 +130,8 @@ impl SectionConsumer { }, mmap_able: flags.mmap_able, spill_path, - file_counter: AtomicUsize::new(0), + #[cfg(feature = "wal")] + file_counter: std::sync::atomic::AtomicUsize::new(0), is_spilled: std::sync::atomic::AtomicBool::new(false), } } @@ -152,7 +153,9 @@ impl SectionConsumer { .serialize() .map_err(|e| SpillError::IoError(e.to_string()))?; - let id = self.file_counter.fetch_add(1, Ordering::Relaxed); + let id = self + .file_counter + .fetch_add(1, std::sync::atomic::Ordering::Relaxed); let filename = format!("{:?}_{id}.spill", self.section.section_type()); let path = spill_dir.join(filename); @@ -858,6 +861,8 @@ mod tests { } } + // Spilling writes a file, which needs the `wal` feature's I/O. + #[cfg(feature = "wal")] #[test] fn alix_spill_writes_serialized_bytes_through_swap_to_mmap() { let dir = tempfile::tempdir().expect("tempdir"); @@ -913,6 +918,8 @@ mod tests { } } + // Spilling writes a file, which needs the `wal` feature's I/O. + #[cfg(feature = "wal")] #[test] fn jules_reload_calls_section_reload_to_ram() { let dir = tempfile::tempdir().expect("tempdir"); diff --git a/crates/grafeo-engine/src/database/upsert.rs b/crates/grafeo-engine/src/database/upsert.rs index 2ad81ee71..dc11c0af5 100644 --- a/crates/grafeo-engine/src/database/upsert.rs +++ b/crates/grafeo-engine/src/database/upsert.rs @@ -1,7 +1,8 @@ //! Upserts: create or update nodes and edges by a key property. //! -//! Each call runs one `UNWIND $rows ... MERGE ... SET ...` statement, so the -//! rows are checked, logged, reported to CDC and counted like any query. Rows +//! Each call writes with one `UNWIND $rows ... MERGE ... SET ...` statement +//! (an edge upsert looks up its endpoints first), so the rows are checked, +//! logged, reported to CDC and counted like any query. Rows //! apply in order and see the writes of the rows before them: a key repeated //! within one call behaves like repeated calls (the first creates, the next //! update). @@ -25,8 +26,9 @@ pub struct UpsertSummary { pub created: usize, /// Rows that updated an existing node or edge. pub updated: usize, - /// Rows that were not written: a row without its key, or an edge row - /// whose endpoint does not exist. + /// Rows that were not written: a row without its key, an edge row + /// without a source or target field, and an edge row whose endpoint key + /// matches no node or more than one node. pub skipped: usize, /// The indices of the skipped rows, in order (at most 1,000). pub skipped_rows: Vec, @@ -103,9 +105,11 @@ fn upsert_nodes( /// Creates or updates one edge per row between the nodes whose /// `options.endpoint_key` is the row's source and target field, matched by /// its type and `options.key`. Every other field of a row is an edge -/// property. A row whose endpoint does not exist is skipped, never created. +/// property. A row is skipped when it lacks the key, the source field or the +/// target field, or when no node or more than one node has its endpoint +/// key; endpoints are never created. fn upsert_edges( - run: impl FnOnce(&str, HashMap) -> Result, + session: &Session, edge_type: &str, rows: Vec>, options: &EdgeUpsertOptions, @@ -119,6 +123,16 @@ fn upsert_edges( ] { check_name(what, name)?; } + let (key_field, src_field, dst_field) = (&options.key, &options.src_field, &options.dst_field); + if key_field == src_field || key_field == dst_field || src_field == dst_field { + return Err(Error::Query(QueryError::new( + QueryErrorKind::Semantic, + format!( + "upsert: the key ({key_field}), source field ({src_field}) and target field \ + ({dst_field}) must be different fields" + ), + ))); + } let endpoint_labels = options .endpoint_labels .iter() @@ -130,7 +144,7 @@ fn upsert_edges( PropertyKey::new(options.dst_field.as_str()), ); let key = PropertyKey::new(options.key.as_str()); - let items = rows + let mut items: Vec<(usize, Value)> = rows .into_iter() .enumerate() .filter_map(|(index, mut row)| { @@ -139,9 +153,12 @@ fn upsert_edges( if row.get(&key).is_none_or(Value::is_null) { return None; } - Some(item( + Some(( index, - [("src", source), ("dst", target), ("props", map_value(row))], + item( + index, + [("src", source), ("dst", target), ("props", map_value(row))], + ), )) }) .collect(); @@ -152,16 +169,78 @@ fn upsert_edges( (d{endpoint_labels} {{{endpoint_key}: item.dst}}) \ MERGE (s)-[r:{edge_type} {{{key}: item.props.{key}}}]->(d) \ SET r {set} item.props \ - RETURN item.i", + RETURN item.i, id(s), id(d)", endpoint_key = quote(&options.endpoint_key), edge_type = quote(edge_type), key = quote(&options.key), ); - summarize(run, &query, items, total, |counters| counters.edges_created) + + // A row whose endpoint key more than one node has matches one pair of + // endpoints per node and comes back once per pair (a row can also come + // back once per edge of an existing pair, which is not ambiguous). Such + // an attempt is undone and the call runs again without those rows, so + // they write nothing; each attempt drops at least one row. + let result = loop { + if items.is_empty() { + return Ok(summary(total, &BTreeSet::new(), 0)); + } + let mut ambiguous = BTreeSet::new(); + let attempt = session.as_one_write(|| { + let rows = items + .iter() + .map(|(_, item)| item.clone()) + .collect::>(); + let result = session.execute_with_params( + &query, + HashMap::from([("rows".to_string(), Value::List(rows.into()))]), + )?; + ambiguous = rows_with_several_endpoint_pairs(&result); + if ambiguous.is_empty() { + Ok(result) + } else { + Err(Error::Internal("upsert: ambiguous endpoint keys".into())) + } + }); + match attempt { + Ok(result) => break result, + Err(_) if !ambiguous.is_empty() => { + items.retain(|(index, _)| !ambiguous.contains(index)); + } + Err(error) => return Err(error), + } + }; + Ok(count(&result, total, result.counters.edges_created)) +} + +/// The row indices the upsert statement returned, from its first column. +fn returned_rows(result: &QueryResult) -> impl Iterator + '_ { + result.rows().iter().filter_map(|row| match row.first() { + Some(Value::Int64(index)) => usize::try_from(*index).ok(), + _ => None, + }) +} + +/// The row indices the edge upsert statement returned with more than one +/// pair of endpoints (its second and third columns). +fn rows_with_several_endpoint_pairs(result: &QueryResult) -> BTreeSet { + let mut pairs: HashMap> = HashMap::new(); + for row in result.rows() { + if let (Some(Value::Int64(index)), Some(Value::Int64(source)), Some(Value::Int64(target))) = + (row.first(), row.get(1), row.get(2)) + && let Ok(index) = usize::try_from(*index) + { + pairs.entry(index).or_default().insert((*source, *target)); + } + } + pairs + .into_iter() + .filter(|(_, endpoints)| endpoints.len() > 1) + .map(|(index, _)| index) + .collect() } /// Runs the upsert statement over `items` and counts what it did with the -/// `total` rows: the rows it returns were written, `created` of them new. +/// `total` rows (see [`count`]). fn summarize( run: impl FnOnce(&str, HashMap) -> Result, query: &str, @@ -176,16 +255,18 @@ fn summarize( query, HashMap::from([("rows".to_string(), Value::List(items.into()))]), )?; - let written: BTreeSet = result - .rows() - .iter() - .filter_map(|row| match row.first() { - Some(Value::Int64(index)) => usize::try_from(*index).ok(), - _ => None, - }) - .collect(); - let created = usize::try_from(created(&result.counters)).unwrap_or(usize::MAX); - Ok(summary(total, &written, created)) + Ok(count(&result, total, created(&result.counters))) +} + +/// What the upsert statement did with the `total` rows: the rows it returns +/// were written, `created` of them new. +fn count(result: &QueryResult, total: usize, created: u64) -> UpsertSummary { + let written: BTreeSet = returned_rows(result).collect(); + summary( + total, + &written, + usize::try_from(created).unwrap_or(usize::MAX), + ) } fn summary(total: usize, written: &BTreeSet, created: usize) -> UpsertSummary { @@ -267,26 +348,24 @@ impl GrafeoDB { /// graph, between the nodes the row's source and target fields name, in /// one statement (see [`EdgeUpsertOptions`]). /// - /// A row whose endpoint does not exist, or without the edge key, is - /// skipped, never created. Rows apply in order: a key repeated within one - /// call creates one edge, which the later rows update. + /// A row is skipped when it lacks the edge key, the source field or the + /// target field, or when no node or more than one node has its endpoint + /// key; endpoints are never created. Rows + /// apply in order: a key repeated within one call creates one edge, which + /// the later rows update. /// /// # Errors /// - /// Returns an error if a row breaks the schema; nothing of the call is - /// written then. + /// Returns an error if a row breaks the schema, nothing of the call is + /// written then, or if the edge key, source field and target field are + /// not three different fields. pub fn upsert_edges( &self, edge_type: &str, rows: Vec>, options: &EdgeUpsertOptions, ) -> Result { - upsert_edges( - |query, params| self.execute_with_params(query, params), - edge_type, - rows, - options, - ) + upsert_edges(&self.session(), edge_type, rows, options) } } @@ -325,12 +404,7 @@ impl GraphHandle<'_> { rows: Vec>, options: &EdgeUpsertOptions, ) -> Result { - upsert_edges( - |query, params| self.execute_with_params(query, params), - edge_type, - rows, - options, - ) + upsert_edges(&self.session()?, edge_type, rows, options) } } @@ -369,11 +443,6 @@ impl Session { rows: Vec>, options: &EdgeUpsertOptions, ) -> Result { - upsert_edges( - |query, params| self.execute_with_params(query, params), - edge_type, - rows, - options, - ) + upsert_edges(self, edge_type, rows, options) } } diff --git a/crates/grafeo-engine/src/database/wal_store.rs b/crates/grafeo-engine/src/database/wal_store.rs index 2b8e3f26e..5e6715371 100644 --- a/crates/grafeo-engine/src/database/wal_store.rs +++ b/crates/grafeo-engine/src/database/wal_store.rs @@ -176,6 +176,15 @@ impl GraphStore for WalGraphStore { self.inner.edge_type(id) } + fn edge_type_versioned( + &self, + id: EdgeId, + epoch: EpochId, + transaction_id: TransactionId, + ) -> Option { + self.inner.edge_type_versioned(id, epoch, transaction_id) + } + fn has_property_index(&self, property: &str) -> bool { self.inner.has_property_index(property) } diff --git a/crates/grafeo-engine/src/query/binder.rs b/crates/grafeo-engine/src/query/binder.rs index 5e3cc7a72..a8938ad30 100644 --- a/crates/grafeo-engine/src/query/binder.rs +++ b/crates/grafeo-engine/src/query/binder.rs @@ -27,6 +27,29 @@ fn binding_error_with_hint(message: impl Into, hint: impl Into) Error::Query(QueryError::new(QueryErrorKind::Semantic, message).with_hint(hint)) } +/// What a pattern binds a variable to. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Element { + Node, + Edge, +} + +/// Whether a variable holds a value of a known type that is not a node or an +/// edge (`WITH 1 AS r`). A variable of unknown type may hold either. +fn holds_a_value(info: &VariableInfo) -> bool { + !info.is_node + && !info.is_edge + && !matches!( + info.data_type, + LogicalType::Any + | LogicalType::Null + | LogicalType::Node + | LogicalType::Edge + | LogicalType::List(_) + | LogicalType::Path + ) +} + /// Creates an "undefined variable" error with a suggestion if a similar variable exists. fn undefined_variable_error(variable: &str, context: &BindingContext, suffix: &str) -> Error { let candidates: Vec = context.variable_names(); @@ -42,6 +65,48 @@ fn undefined_variable_error(variable: &str, context: &BindingContext, suffix: &s } } +/// The parts of a subquery body joined by `UNION` (under the `DISTINCT` of +/// a plain `UNION`), or `None` for a body of one part. +fn union_parts(plan: &LogicalOperator) -> Option<&[LogicalOperator]> { + match plan { + LogicalOperator::Union(union) => Some(&union.inputs), + LogicalOperator::Distinct(distinct) => match distinct.input.as_ref() { + LogicalOperator::Union(union) => Some(&union.inputs), + _ => None, + }, + _ => None, + } +} + +/// The outer variables one part of a subquery imports: the columns of the +/// parameter scan its plan starts from, none without one. +fn leading_imports(plan: &LogicalOperator) -> Vec { + match plan { + LogicalOperator::ParameterScan(scan) => scan.columns.clone(), + LogicalOperator::NodeScan(scan) => { + scan.input.as_deref().map_or_else(Vec::new, leading_imports) + } + LogicalOperator::EdgeScan(scan) => { + scan.input.as_deref().map_or_else(Vec::new, leading_imports) + } + LogicalOperator::Expand(op) => leading_imports(&op.input), + LogicalOperator::Filter(op) => leading_imports(&op.input), + LogicalOperator::Project(op) => leading_imports(&op.input), + LogicalOperator::Return(op) => leading_imports(&op.input), + LogicalOperator::Aggregate(op) => leading_imports(&op.input), + LogicalOperator::Limit(op) => leading_imports(&op.input), + LogicalOperator::Skip(op) => leading_imports(&op.input), + LogicalOperator::Sort(op) => leading_imports(&op.input), + LogicalOperator::Distinct(op) => leading_imports(&op.input), + LogicalOperator::Unwind(op) => leading_imports(&op.input), + LogicalOperator::Bind(op) => leading_imports(&op.input), + LogicalOperator::Join(join) => leading_imports(&join.left), + LogicalOperator::LeftJoin(join) => leading_imports(&join.left), + LogicalOperator::Apply(apply) => leading_imports(&apply.input), + _ => Vec::new(), + } +} + /// Information about a bound variable. #[derive(Debug, Clone)] pub struct VariableInfo { @@ -159,6 +224,10 @@ impl Binder { LogicalOperator::Return(ret) => self.bind_return(ret), LogicalOperator::Project(project) => { self.bind_operator(&project.input)?; + // A projection that does not pass its input through (a WITH) + // ends the scope of what it leaves out: its rows hold only + // the projected columns. + let mut projected = (!project.pass_through_input).then(BindingContext::new); for projection in &project.projections { self.validate_expression(&projection.expression)?; // Add the projection alias to the context (for WITH clause support) @@ -169,17 +238,35 @@ impl Binder { // or a Case that selects between node variables (used // by optional() and union() translations). let (is_node, is_edge) = self.infer_entity_status(&projection.expression); - self.context.add_variable( - alias.clone(), - VariableInfo { - name: alias.clone(), - data_type, - is_node, - is_edge, - }, + let info = VariableInfo { + name: alias.clone(), + data_type, + is_node, + is_edge, + }; + if let Some(projected) = &mut projected { + projected.add_variable(alias.clone(), info.clone()); + } + self.context.add_variable(alias.clone(), info); + } else if let Some(projected) = &mut projected { + // An unaliased variable passes on as itself; any + // other unaliased item is a column named after its + // expression (`a.name`), which does not keep `a`. + let name = crate::query::planner::common::expression_to_string( + &projection.expression, ); + let info = self.context.get(&name).cloned().unwrap_or(VariableInfo { + name: name.clone(), + data_type: LogicalType::Any, + is_node: false, + is_edge: false, + }); + projected.add_variable(name, info); } } + if let Some(projected) = projected { + self.context = projected; + } Ok(()) } LogicalOperator::Limit(limit) => self.bind_operator(&limit.input), @@ -215,16 +302,7 @@ impl Binder { if let Some(ref input) = scan.input { self.bind_operator(input)?; } - self.context.add_variable( - scan.variable.clone(), - VariableInfo { - name: scan.variable.clone(), - data_type: LogicalType::Edge, - is_node: false, - is_edge: true, - }, - ); - Ok(()) + self.bind_element(&scan.variable, Element::Edge) } LogicalOperator::Distinct(distinct) => self.bind_operator(&distinct.input), LogicalOperator::Join(join) => self.bind_join(join), @@ -350,25 +428,15 @@ impl Binder { // RDF/SPARQL operators LogicalOperator::TripleScan(scan) => self.bind_triple_scan(scan), - LogicalOperator::Union(union) => { - for input in &union.inputs { - self.bind_operator(input)?; - } - Ok(()) - } + LogicalOperator::Union(union) => self.bind_branches(&union.inputs), LogicalOperator::LeftJoin(lj) => { - self.bind_operator(&lj.left)?; - self.bind_operator(&lj.right)?; + self.bind_join_inputs(&lj.left, &lj.right)?; if let Some(ref cond) = lj.condition { self.validate_expression(cond)?; } Ok(()) } - LogicalOperator::AntiJoin(aj) => { - self.bind_operator(&aj.left)?; - self.bind_operator(&aj.right)?; - Ok(()) - } + LogicalOperator::AntiJoin(aj) => self.bind_join_inputs(&aj.left, &aj.right), LogicalOperator::Bind(bind) => { self.bind_operator(&bind.input)?; self.validate_expression(&bind.expression)?; @@ -601,19 +669,13 @@ impl Binder { Ok(()) } LogicalOperator::Except(except) => { - self.bind_operator(&except.left)?; - self.bind_operator(&except.right)?; - Ok(()) + self.bind_branches([except.left.as_ref(), except.right.as_ref()]) } LogicalOperator::Intersect(intersect) => { - self.bind_operator(&intersect.left)?; - self.bind_operator(&intersect.right)?; - Ok(()) + self.bind_branches([intersect.left.as_ref(), intersect.right.as_ref()]) } LogicalOperator::Otherwise(otherwise) => { - self.bind_operator(&otherwise.left)?; - self.bind_operator(&otherwise.right)?; - Ok(()) + self.bind_branches([otherwise.left.as_ref(), otherwise.right.as_ref()]) } LogicalOperator::Apply(apply) => { // Snapshot context BEFORE binding the input, so we can detect @@ -653,28 +715,66 @@ impl Binder { let outer_names: HashSet = self.context.variable_names().iter().cloned().collect(); - self.bind_operator(&apply.subplan)?; + // A name the subquery imports must be a variable of the outer + // query (`*` imports all of them). + if let Some(missing) = apply + .shared_variables + .iter() + .find(|name| *name != "*" && !outer_names.contains(*name)) + { + return Err(undefined_variable_error( + missing, + &self.context, + " imported into CALL", + )); + } + + // A subquery sees the outer variables it imports: the ones its + // scope clause or importing `WITH` names (`CALL (a, b)`, + // `WITH a`), all of them for `*`, and none when it names none + // (`CALL () { ... }`, a Cypher `CALL { ... }` without an + // importing `WITH`): the planner then runs it without the + // outer row. It binds in a context of its own, so neither its + // internal variables nor a `WITH` in it that drops outer ones + // change the outer scope. + let imports_all = apply.shared_variables.iter().any(|name| name == "*"); + let bound = match union_parts(&apply.subplan) { + // Each part of a UNION imports its own outer variables (the + // Apply imports all of them): the ones its scan starts from. + Some(parts) if !imports_all => parts.iter().try_for_each(|part| { + let part_context = self.imported(&leading_imports(part)); + let outer_context = std::mem::replace(&mut self.context, part_context); + let bound = self.bind_operator(part); + self.context = outer_context; + bound + }), + _ => { + let subplan_context = if imports_all { + self.context.clone() + } else { + self.imported(&apply.shared_variables) + }; + let outer_context = std::mem::replace(&mut self.context, subplan_context); + let bound = self.bind_operator(&apply.subplan); + self.context = outer_context; + bound + } + }; + bound?; - // Remove internal-only variables added by the subplan (those that - // are not output columns). Prevents subplan internals from leaking - // into the outer query or sibling CALL blocks. + // A subquery returns new variables only, as in openCypher: an + // outer one, imported or not, would be bound twice. let mut subplan_output_ctx = BindingContext::new(); Self::register_subplan_columns(&apply.subplan, &mut subplan_output_ctx); - let subplan_output_names: HashSet = subplan_output_ctx + if let Some(clash) = subplan_output_ctx .variable_names() - .iter() - .cloned() - .collect(); - - let to_remove: Vec = self - .context - .variable_names() - .iter() - .filter(|n| !outer_names.contains(*n) && !subplan_output_names.contains(*n)) - .cloned() - .collect(); - for name in to_remove { - self.context.remove_variable(&name); + .into_iter() + .find(|name| outer_names.contains(name)) + { + return Err(binding_error_with_hint( + format!("Variable '{clash}' is already declared outside the CALL subquery"), + "return it under a new name", + )); } // Register output columns so downstream operators can reference them. @@ -683,7 +783,9 @@ impl Binder { } LogicalOperator::MultiWayJoin(mwj) => { for input in &mwj.inputs { + let before = self.context.clone(); self.bind_operator(input)?; + self.keep_join_scope(before); } for cond in &mwj.conditions { self.validate_expression(&cond.left)?; @@ -692,8 +794,13 @@ impl Binder { Ok(()) } LogicalOperator::ParameterScan(param_scan) => { - // Register parameter columns as variables (injected by outer Apply) + // Register parameter columns as variables (injected by outer + // Apply). A variable of the outer query keeps what it is, so a + // CALL subquery can match an imported edge as an edge. for col in ¶m_scan.columns { + if self.context.contains(col) { + continue; + } self.context.add_variable( col.clone(), VariableInfo { @@ -844,16 +951,77 @@ impl Binder { } // Add the scanned variable to scope + self.bind_element(&scan.variable, Element::Node) + } + + /// Binds `name` to a node or an edge of a pattern. A name already bound + /// to the other kind, or to a value, is an error: matching it would + /// compare a node with an edge (or a number) and match by a coincidence + /// of IDs. A name bound to the same kind, or to something of unknown + /// kind (an UNWIND variable, a procedure result), is bound again. + fn bind_element(&mut self, name: &str, element: Element) -> Result<()> { + if let Some(info) = self.context.get(name) { + let conflict = match element { + Element::Node if info.is_edge => Some("is an edge, so it cannot also be a node"), + Element::Edge if info.is_node => Some("is a node, so it cannot also be an edge"), + Element::Node if holds_a_value(info) => { + Some("holds a value, so it cannot be a node") + } + Element::Edge if holds_a_value(info) => { + Some("holds a value, so it cannot be an edge") + } + _ => None, + }; + if let Some(conflict) = conflict { + return Err(binding_error(format!("Variable '{name}' {conflict}"))); + } + } + let is_edge = element == Element::Edge; self.context.add_variable( - scan.variable.clone(), + name.to_string(), VariableInfo { - name: scan.variable.clone(), - data_type: LogicalType::Node, - is_node: true, - is_edge: false, + name: name.to_string(), + data_type: if is_edge { + LogicalType::Edge + } else { + LogicalType::Node + }, + is_node: !is_edge, + is_edge, }, ); + Ok(()) + } + /// Binds the branches of a set operation, each in the scope the first one + /// started from: a branch is a query of its own, so a name may be a node + /// in one branch and an edge in the next. After the last branch the scope + /// has every name a branch bound; a name the branches bind differently is + /// of unknown kind. + fn bind_branches<'a>( + &mut self, + branches: impl IntoIterator, + ) -> Result<()> { + let before = self.context.clone(); + let mut after: IndexMap = IndexMap::new(); + for branch in branches { + self.context = before.clone(); + self.bind_operator(branch)?; + for (name, info) in &self.context.variables { + match after.get_mut(name) { + Some(seen) if seen.is_node != info.is_node || seen.is_edge != info.is_edge => { + seen.is_node = false; + seen.is_edge = false; + seen.data_type = LogicalType::Any; + } + Some(_) => {} + None => { + after.insert(name.clone(), info.clone()); + } + } + } + } + self.context = BindingContext { variables: after }; Ok(()) } @@ -883,27 +1051,11 @@ impl Binder { // Add edge variable if present if let Some(ref edge_var) = expand.edge_variable { - self.context.add_variable( - edge_var.clone(), - VariableInfo { - name: edge_var.clone(), - data_type: LogicalType::Edge, - is_node: false, - is_edge: true, - }, - ); + self.bind_element(edge_var, Element::Edge)?; } // Add target variable - self.context.add_variable( - expand.to_variable.clone(), - VariableInfo { - name: expand.to_variable.clone(), - data_type: LogicalType::Node, - is_node: true, - is_edge: false, - }, - ); + self.bind_element(&expand.to_variable, Element::Node)?; // Add path variables for variable-length paths if let Some(ref path_alias) = expand.path_alias { @@ -996,6 +1148,13 @@ impl Binder { } LogicalOperator::Sort(s) => Self::register_subplan_columns(&s.input, ctx), LogicalOperator::Limit(l) => Self::register_subplan_columns(&l.input, ctx), + LogicalOperator::Skip(s) => Self::register_subplan_columns(&s.input, ctx), + // The branches of a UNION return the same columns. + LogicalOperator::Union(u) => { + if let Some(first) = u.inputs.first() { + Self::register_subplan_columns(first, ctx); + } + } LogicalOperator::Distinct(d) => Self::register_subplan_columns(&d.input, ctx), LogicalOperator::Aggregate(agg) => { // Aggregate produces named output columns @@ -1319,8 +1478,7 @@ impl Binder { /// Binds a join operator. fn bind_join(&mut self, join: &crate::query::plan::JoinOp) -> Result<()> { // Bind both sides of the join - self.bind_operator(&join.left)?; - self.bind_operator(&join.right)?; + self.bind_join_inputs(&join.left, &join.right)?; // Validate join conditions for condition in &join.conditions { @@ -1331,6 +1489,39 @@ impl Binder { Ok(()) } + /// The outer variables named in `names`, with what they are in the + /// current context. + fn imported(&self, names: &[String]) -> BindingContext { + let mut imported = BindingContext::new(); + for name in names { + if let Some(info) = self.context.get(name) { + imported.add_variable(name.clone(), info.clone()); + } + } + imported + } + + /// Binds the two inputs of a join. The right one sees the left one's + /// variables, and the join's rows hold the columns of both, so a + /// projection inside the right input does not end the scope of the left + /// one's variables. + fn bind_join_inputs(&mut self, left: &LogicalOperator, right: &LogicalOperator) -> Result<()> { + self.bind_operator(left)?; + let after_left = self.context.clone(); + self.bind_operator(right)?; + self.keep_join_scope(after_left); + Ok(()) + } + + /// Makes the context the variables of `before` followed by those bound + /// since, for an input that joins its rows to the ones `before` holds. + fn keep_join_scope(&mut self, before: BindingContext) { + let since = std::mem::replace(&mut self.context, before); + for (name, info) in since.variables { + self.context.add_variable(name, info); + } + } + /// Binds an aggregate operator. fn bind_aggregate(&mut self, agg: &crate::query::plan::AggregateOp) -> Result<()> { // Bind the input first @@ -1516,6 +1707,7 @@ mod tests { ], distinct: false, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: Some("e".to_string()), @@ -1558,6 +1750,7 @@ mod tests { }], distinct: false, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "undefined".to_string(), // not defined! to_variable: "b".to_string(), edge_variable: None, @@ -1970,6 +2163,103 @@ mod tests { assert!(result.is_err(), "WITH on undefined variable should fail"); } + /// `MATCH (n) WITH n.name AS name`, then a projection of `n` or `name`. + fn project_after_with(read: &str, pass_through_input: bool) -> LogicalPlan { + use crate::query::plan::{ProjectOp, Projection}; + + let with = LogicalOperator::Project(ProjectOp { + projections: vec![Projection { + expression: LogicalExpression::Property { + variable: "n".to_string(), + property: "name".to_string(), + }, + alias: Some("name".to_string()), + }], + input: Box::new(LogicalOperator::NodeScan(NodeScanOp { + variable: "n".to_string(), + label: None, + input: None, + })), + pass_through_input, + }); + LogicalPlan::new(LogicalOperator::Return(ReturnOp { + items: vec![ReturnItem { + expression: LogicalExpression::Variable(read.to_string()), + alias: None, + }], + distinct: false, + input: Box::new(with), + })) + } + + #[test] + fn test_project_ends_the_scope_of_what_it_leaves_out() { + let error = Binder::new() + .bind(&project_after_with("n", false)) + .unwrap_err(); + assert!( + error.to_string().contains("Undefined variable 'n'"), + "{error}" + ); + // A pass-through projection (GQL LET) keeps its input's variables. + let ctx = Binder::new().bind(&project_after_with("n", true)).unwrap(); + assert!(ctx.contains("n") && ctx.contains("name")); + } + + #[test] + fn test_an_unaliased_projection_passes_on_its_column_only() { + use crate::query::plan::{ProjectOp, Projection}; + + // The column of an unaliased `n.name` is named after it; `n` is gone. + let plan = LogicalPlan::new(LogicalOperator::Project(ProjectOp { + projections: vec![Projection { + expression: LogicalExpression::Property { + variable: "n".to_string(), + property: "name".to_string(), + }, + alias: None, + }], + input: Box::new(LogicalOperator::NodeScan(NodeScanOp { + variable: "n".to_string(), + label: None, + input: None, + })), + pass_through_input: false, + })); + let ctx = Binder::new().bind(&plan).unwrap(); + assert_eq!(ctx.variable_names(), ["n.name"]); + } + + #[test] + fn test_a_projection_in_a_join_input_keeps_the_other_side() { + use crate::query::plan::{JoinOp, JoinType, ProjectOp, Projection}; + + // The right input projects `b` away; the join's rows still hold `a`. + let scan = |variable: &str| { + LogicalOperator::NodeScan(NodeScanOp { + variable: variable.to_string(), + label: None, + input: None, + }) + }; + let right = LogicalOperator::Project(ProjectOp { + projections: vec![Projection { + expression: LogicalExpression::Variable("b".to_string()), + alias: Some("c".to_string()), + }], + input: Box::new(scan("b")), + pass_through_input: false, + }); + let plan = LogicalPlan::new(LogicalOperator::Join(JoinOp { + left: Box::new(scan("a")), + right: Box::new(right), + join_type: JoinType::Cross, + conditions: vec![], + })); + let ctx = Binder::new().bind(&plan).unwrap(); + assert_eq!(ctx.variable_names(), ["a", "c"]); + } + // --- UNWIND --- #[test] @@ -2438,6 +2728,7 @@ mod tests { }], distinct: false, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "x".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -4415,6 +4706,7 @@ mod tests { use crate::query::plan::{ExpandDirection, ExpandOp, PathMode}; let plan = LogicalPlan::new(LogicalOperator::Expand(ExpandOp { + quantified: true, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, diff --git a/crates/grafeo-engine/src/query/executor/mod.rs b/crates/grafeo-engine/src/query/executor/mod.rs index 9fb532e9b..a40ad65f8 100644 --- a/crates/grafeo-engine/src/query/executor/mod.rs +++ b/crates/grafeo-engine/src/query/executor/mod.rs @@ -89,7 +89,6 @@ impl Executor { self } - /// Checks whether the deadline has been exceeded. /// Reports the writes `counter` counts in the result's counters. #[must_use] pub fn with_write_counter(mut self, counter: Arc) -> Self { @@ -105,6 +104,7 @@ impl Executor { result } + /// Checks whether the deadline has been exceeded. fn check_deadline(&self) -> Result<()> { #[cfg(not(target_arch = "wasm32"))] if let Some(deadline) = self.deadline diff --git a/crates/grafeo-engine/src/query/executor/user_procedure.rs b/crates/grafeo-engine/src/query/executor/user_procedure.rs index 8e862b405..7c4e1f633 100644 --- a/crates/grafeo-engine/src/query/executor/user_procedure.rs +++ b/crates/grafeo-engine/src/query/executor/user_procedure.rs @@ -8,7 +8,7 @@ use std::sync::Arc; use grafeo_common::types::{EpochId, TransactionId, Value}; use grafeo_core::execution::DataChunk; -use grafeo_core::execution::operators::{Operator, OperatorError, OperatorResult}; +use grafeo_core::execution::operators::{Operator, OperatorError, OperatorResult, WriteCounter}; use grafeo_core::graph::{GraphStoreMut, GraphStoreSearch}; use crate::catalog::Catalog; @@ -30,6 +30,8 @@ pub struct ProcedureContext { pub viewing_epoch: EpochId, /// Catalog for sub-planner resolution. pub catalog: Option>, + /// The calling statement's write counter: the body's writes count there. + pub write_counter: Arc, } /// An operator that executes a user-defined stored procedure. @@ -61,6 +63,8 @@ pub struct UserProcedureOperator { viewing_epoch: EpochId, /// Catalog for sub-planner. catalog: Option>, + /// The calling statement's write counter. + write_counter: Arc, /// Buffered result rows from execution. result_rows: Option>>, /// Current row index into buffered results. @@ -94,6 +98,7 @@ impl UserProcedureOperator { transaction_id: ctx.transaction_id, viewing_epoch: ctx.viewing_epoch, catalog: ctx.catalog, + write_counter: ctx.write_counter, result_rows: None, row_index: 0, output_columns, @@ -111,9 +116,13 @@ impl UserProcedureOperator { } // Use the module-level translate function - let logical_plan = crate::query::translators::gql::translate(&body).map_err(|e| { + let mut logical_plan = crate::query::translators::gql::translate(&body).map_err(|e| { OperatorError::Execution(format!("Failed to translate procedure body: {e}")) })?; + // A pattern through a node or edge bound before is checked, as in a + // session's plan (the optimizer's first pass; the body skips the + // optimizer). + logical_plan.root = crate::query::optimizer::close_cycles(logical_plan.root); // Plan physical operators let planner = if let Some(ref tx_mgr) = self.transaction_manager { @@ -134,7 +143,8 @@ impl UserProcedureOperator { p = p.with_catalog(Arc::clone(cat)); } p - }; + } + .with_write_counter(Arc::clone(&self.write_counter)); let physical = planner .plan(&logical_plan) diff --git a/crates/grafeo-engine/src/query/optimizer/cardinality.rs b/crates/grafeo-engine/src/query/optimizer/cardinality.rs index 46e43a1ad..b04224694 100644 --- a/crates/grafeo-engine/src/query/optimizer/cardinality.rs +++ b/crates/grafeo-engine/src/query/optimizer/cardinality.rs @@ -1650,6 +1650,7 @@ mod tests { estimator.add_table_stats("Person", TableStats::new(100)); let expand = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -1677,6 +1678,7 @@ mod tests { estimator.add_table_stats("Person", TableStats::new(100)); let expand = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -1704,6 +1706,7 @@ mod tests { estimator.add_table_stats("Person", TableStats::new(100)); let expand = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -1977,6 +1980,7 @@ mod tests { estimator.set_avg_fanout(5.0); let expand = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, diff --git a/crates/grafeo-engine/src/query/optimizer/cost.rs b/crates/grafeo-engine/src/query/optimizer/cost.rs index 4949a194f..411c01bef 100644 --- a/crates/grafeo-engine/src/query/optimizer/cost.rs +++ b/crates/grafeo-engine/src/query/optimizer/cost.rs @@ -885,6 +885,7 @@ mod tests { fn test_cost_model_expand() { let model = CostModel::new(); let expand = ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -912,6 +913,7 @@ mod tests { // Outgoing KNOWS: fanout = 5 let knows_out = ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -927,6 +929,7 @@ mod tests { // Outgoing WORKS_AT: fanout = 1 (each person works at one company) let works_out = ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -948,6 +951,7 @@ mod tests { // Incoming WORKS_AT: fanout = 50 (company has many employees) let works_in = ExpandOp { + quantified: false, from_variable: "c".to_string(), to_variable: "p".to_string(), edge_variable: None, @@ -972,6 +976,7 @@ mod tests { fn test_cost_model_expand_unknown_edge_type_uses_global_fanout() { let model = CostModel::new().with_avg_fanout(7.0); let expand = ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -987,6 +992,7 @@ mod tests { // Without edge type (uses global fanout too) let expand_no_type = ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -1425,6 +1431,7 @@ mod tests { // Multi-type outgoing: KNOWS(5) + FOLLOWS(20) = 25 let multi_expand = ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -1440,6 +1447,7 @@ mod tests { // Single type: KNOWS(5) only let single_expand = ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, diff --git a/crates/grafeo-engine/src/query/optimizer/cycles.rs b/crates/grafeo-engine/src/query/optimizer/cycles.rs new file mode 100644 index 000000000..2935c38e7 --- /dev/null +++ b/crates/grafeo-engine/src/query/optimizer/cycles.rs @@ -0,0 +1,168 @@ +//! Patterns that come back to a node or an edge their input already binds. + +use std::collections::HashSet; + +use crate::query::plan::{BinaryOp, FilterOp, LogicalExpression, LogicalOperator}; + +/// Rewrites every expand that binds a variable its input already binds: a +/// target node, such as the last hop of `(a)-->(b)-->(a)` or of +/// `MATCH (a)-->(b) MATCH (b)-->(a)`, or an edge, such as `r` in +/// `MATCH ()-[r]->() MATCH (x)-[r]->(y)`. The expand binds a fresh variable +/// instead, under a filter that it is the same node or edge. Binding the +/// variable itself again would return every path of that shape instead of +/// the ones through the bound node or edge. +pub(crate) fn close_cycles(op: LogicalOperator) -> LogicalOperator { + let mut taken = HashSet::new(); + plan_names(&op, &mut taken); + close(op, &mut FreshNames { taken, next: 0 }, None) +} + +/// Names for the fresh variables, `_cycle_end_N` for a node and +/// `_bound_edge_N` for an edge, skipping the ones the plan already uses: a +/// user may name a variable `_cycle_end_0` too. +struct FreshNames { + taken: HashSet, + next: usize, +} + +impl FreshNames { + fn next(&mut self, prefix: &str) -> String { + loop { + let name = format!("{prefix}_{}", self.next); + self.next += 1; + if !self.taken.contains(&name) { + return name; + } + } + } +} + +/// Rewrites `op` and every operator below it. `imports` holds the variables +/// of the row a subquery runs for, the ones `CALL { WITH * ... }` imports. +fn close( + op: LogicalOperator, + fresh: &mut FreshNames, + imports: Option<&HashSet>, +) -> LogicalOperator { + if let LogicalOperator::Apply(mut apply) = op { + apply.input = Box::new(close(*apply.input, fresh, imports)); + let outer = apply.input.bound_variables(imports); + apply.subplan = Box::new(close(*apply.subplan, fresh, outer.as_ref())); + return LogicalOperator::Apply(apply); + } + let op = op.map_children(|child| close(child, fresh, imports)); + let LogicalOperator::Expand(mut expand) = op else { + return op; + }; + let Some(bound) = expand.input.bound_variables(imports) else { + return LogicalOperator::Expand(expand); + }; + let mut checks = Vec::new(); + if bound.contains(&expand.to_variable) { + let target = std::mem::replace(&mut expand.to_variable, fresh.next("_cycle_end")); + checks.push(same(&expand.to_variable, target)); + } + if let Some(edge) = &mut expand.edge_variable + && bound.contains(edge.as_str()) + { + let bound_edge = std::mem::replace(edge, fresh.next("_bound_edge")); + checks.push(same(edge, bound_edge)); + } + let Some(predicate) = checks + .into_iter() + .reduce(|left, right| LogicalExpression::Binary { + left: Box::new(left), + op: BinaryOp::And, + right: Box::new(right), + }) + else { + return LogicalOperator::Expand(expand); + }; + LogicalOperator::Filter(FilterOp { + predicate, + input: Box::new(LogicalOperator::Expand(expand)), + pushdown_hint: None, + }) +} + +/// `fresh = bound`: the fresh variable is the node or edge bound before. +fn same(fresh: &str, bound: String) -> LogicalExpression { + LogicalExpression::Binary { + left: Box::new(LogicalExpression::Variable(fresh.to_string())), + op: BinaryOp::Eq, + right: Box::new(LogicalExpression::Variable(bound)), + } +} + +/// Adds every name that `op` or an operator below it binds. +fn plan_names(op: &LogicalOperator, names: &mut HashSet) { + let bound: Vec<&String> = match op { + LogicalOperator::NodeScan(scan) => vec![&scan.variable], + LogicalOperator::EdgeScan(scan) => vec![&scan.variable], + LogicalOperator::Expand(expand) => [&expand.from_variable, &expand.to_variable] + .into_iter() + .chain(&expand.edge_variable) + .chain(&expand.path_alias) + .collect(), + LogicalOperator::Project(project) => project + .projections + .iter() + .filter_map(|projection| projection.alias.as_ref()) + .collect(), + LogicalOperator::Aggregate(aggregate) => aggregate + .aggregates + .iter() + .filter_map(|aggregate| aggregate.alias.as_ref()) + .collect(), + LogicalOperator::HorizontalAggregate(aggregate) => { + vec![&aggregate.list_column, &aggregate.alias] + } + LogicalOperator::Return(ret) => ret + .items + .iter() + .filter_map(|item| item.alias.as_ref()) + .collect(), + LogicalOperator::CreateNode(create) => vec![&create.variable], + LogicalOperator::CreateEdge(create) => [&create.from_variable, &create.to_variable] + .into_iter() + .chain(&create.variable) + .collect(), + LogicalOperator::Merge(merge) => vec![&merge.variable], + LogicalOperator::MergeRelationship(merge) => vec![ + &merge.variable, + &merge.source_variable, + &merge.target_variable, + ], + LogicalOperator::Bind(bind) => vec![&bind.variable], + LogicalOperator::Unwind(unwind) => std::iter::once(&unwind.variable) + .chain(&unwind.ordinality_var) + .chain(&unwind.offset_var) + .collect(), + LogicalOperator::MapCollect(collect) => { + vec![&collect.key_var, &collect.value_var, &collect.alias] + } + LogicalOperator::ShortestPath(path) => { + vec![&path.source_var, &path.target_var, &path.path_alias] + } + LogicalOperator::VectorScan(scan) => vec![&scan.variable], + LogicalOperator::VectorJoin(join) => std::iter::once(&join.right_variable) + .chain(&join.score_variable) + .collect(), + LogicalOperator::TextScan(scan) => std::iter::once(&scan.variable) + .chain(&scan.score_column) + .collect(), + LogicalOperator::ParameterScan(scan) => scan.columns.iter().collect(), + LogicalOperator::CallProcedure(call) => call + .yield_items + .iter() + .flatten() + .map(|item| item.alias.as_ref().unwrap_or(&item.field_name)) + .collect(), + LogicalOperator::LoadData(load) => vec![&load.variable], + _ => Vec::new(), + }; + names.extend(bound.into_iter().cloned()); + for child in op.children() { + plan_names(child, names); + } +} diff --git a/crates/grafeo-engine/src/query/optimizer/mod.rs b/crates/grafeo-engine/src/query/optimizer/mod.rs index 2f98d3742..64e63d30f 100644 --- a/crates/grafeo-engine/src/query/optimizer/mod.rs +++ b/crates/grafeo-engine/src/query/optimizer/mod.rs @@ -13,6 +13,8 @@ pub mod cardinality; pub mod cost; +mod cycles; +pub(crate) use cycles::close_cycles; pub mod join_order; pub use cardinality::{ @@ -242,7 +244,9 @@ impl Optimizer { /// Returns an error if optimization fails. pub fn optimize(&self, plan: LogicalPlan) -> Result { let _span = grafeo_debug_span!("grafeo::query::optimize"); - let mut root = plan.root; + // Correctness first: a pattern through a node or edge that was bound + // before must be checked. + let mut root = cycles::close_cycles(plan.root); // Apply optimization rules if self.enable_filter_pushdown { @@ -1656,6 +1660,7 @@ mod tests { }, pushdown_hint: None, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -1714,6 +1719,7 @@ mod tests { }, pushdown_hint: None, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, diff --git a/crates/grafeo-engine/src/query/plan.rs b/crates/grafeo-engine/src/query/plan.rs index 1fd04998a..ab7bde130 100644 --- a/crates/grafeo-engine/src/query/plan.rs +++ b/crates/grafeo-engine/src/query/plan.rs @@ -4,7 +4,7 @@ //! and physical execution. Both GQL and Cypher queries are translated to this //! common representation. -use std::collections::HashMap; +use std::collections::{HashMap, HashSet}; use std::fmt; use grafeo_common::types::Value; @@ -803,6 +803,241 @@ impl LogicalOperator { } } +impl LogicalOperator { + /// The variables the rows of this operator hold, or `None` when they are + /// not known here (a procedure without `YIELD`, a `RETURN *` of a + /// subquery, RDF and graph management plans): callers then leave its plan + /// as it is. A `WITH` (`Project`) holds only what it projects. `imports` + /// is what a subquery's `CALL { WITH * ... }` imports: the variables of + /// the row it runs for. Every operator has an arm, so a new one says what + /// it binds before it compiles. + #[must_use] + pub(crate) fn bound_variables( + &self, + imports: Option<&HashSet>, + ) -> Option> { + let mut bound = HashSet::new(); + match self { + Self::Empty => {} + Self::NodeScan(scan) => { + if let Some(input) = &scan.input { + bound = input.bound_variables(imports)?; + } + bound.insert(scan.variable.clone()); + } + Self::EdgeScan(scan) => { + if let Some(input) = &scan.input { + bound = input.bound_variables(imports)?; + } + bound.insert(scan.variable.clone()); + } + Self::Expand(expand) => { + bound = expand.input.bound_variables(imports)?; + bound.insert(expand.to_variable.clone()); + bound.extend(expand.edge_variable.iter().cloned()); + bound.extend(expand.path_alias.iter().cloned()); + } + Self::Filter(filter) => return filter.input.bound_variables(imports), + Self::Limit(limit) => return limit.input.bound_variables(imports), + Self::Skip(skip) => return skip.input.bound_variables(imports), + Self::Sort(sort) => return sort.input.bound_variables(imports), + Self::Distinct(distinct) => return distinct.input.bound_variables(imports), + Self::Project(project) => { + if project.pass_through_input { + bound = project.input.bound_variables(imports)?; + } + for projection in &project.projections { + match (&projection.alias, &projection.expression) { + (Some(alias), _) => { + bound.insert(alias.clone()); + } + (None, LogicalExpression::Variable(name)) => { + bound.insert(name.clone()); + } + _ => {} + } + } + } + Self::Aggregate(aggregate) => { + for key in &aggregate.group_by { + if let LogicalExpression::Variable(name) = key { + bound.insert(name.clone()); + } + } + bound.extend(aggregate.aggregates.iter().filter_map(|a| a.alias.clone())); + } + Self::Unwind(unwind) => { + bound = unwind.input.bound_variables(imports)?; + bound.insert(unwind.variable.clone()); + bound.extend(unwind.ordinality_var.iter().cloned()); + bound.extend(unwind.offset_var.iter().cloned()); + } + Self::Bind(bind) => { + bound = bind.input.bound_variables(imports)?; + bound.insert(bind.variable.clone()); + } + Self::Join(join) => { + bound = join.left.bound_variables(imports)?; + bound.extend(join.right.bound_variables(imports)?); + } + Self::LeftJoin(join) => { + bound = join.left.bound_variables(imports)?; + bound.extend(join.right.bound_variables(imports)?); + } + // A subquery starts from the variables it imports from the row it + // runs for: the ones its `WITH` names, or all of them for `WITH *`. + Self::ParameterScan(scan) => { + if scan.columns.iter().any(|column| column == "*") { + return imports.cloned(); + } + bound.extend(scan.columns.iter().cloned()); + } + // `CALL { ... }` adds the columns its subquery returns to each row. + Self::Apply(apply) => { + bound = apply.input.bound_variables(imports)?; + apply.subplan.add_returned_variables(&mut bound)?; + } + Self::Return(ret) => { + for item in &ret.items { + match (&item.alias, &item.expression) { + (Some(alias), _) => { + bound.insert(alias.clone()); + } + (None, LogicalExpression::Variable(name)) if name == "*" => { + return ret.input.bound_variables(imports); + } + (None, LogicalExpression::Variable(name)) => { + bound.insert(name.clone()); + } + _ => {} + } + } + } + // A write passes its input's rows on, with what it creates. + Self::CreateNode(create) => { + if let Some(input) = &create.input { + bound = input.bound_variables(imports)?; + } + bound.insert(create.variable.clone()); + } + Self::CreateEdge(create) => { + bound = create.input.bound_variables(imports)?; + bound.extend(create.variable.iter().cloned()); + } + Self::Merge(merge) => { + bound = merge.input.bound_variables(imports)?; + bound.insert(merge.variable.clone()); + } + Self::MergeRelationship(merge) => { + bound = merge.input.bound_variables(imports)?; + bound.insert(merge.variable.clone()); + } + Self::DeleteNode(op) => return op.input.bound_variables(imports), + Self::DeleteEdge(op) => return op.input.bound_variables(imports), + Self::SetProperty(op) => return op.input.bound_variables(imports), + Self::AddLabel(op) => return op.input.bound_variables(imports), + Self::RemoveLabel(op) => return op.input.bound_variables(imports), + Self::ShortestPath(path) => { + bound = path.input.bound_variables(imports)?; + bound.insert(path.path_alias.clone()); + } + Self::MapCollect(collect) => { + bound.insert(collect.alias.clone()); + } + Self::HorizontalAggregate(aggregate) => { + bound = aggregate.input.bound_variables(imports)?; + bound.insert(aggregate.alias.clone()); + } + Self::VectorScan(scan) => { + if let Some(input) = &scan.input { + bound = input.bound_variables(imports)?; + } + bound.insert(scan.variable.clone()); + } + Self::VectorJoin(join) => { + bound = join.input.bound_variables(imports)?; + bound.insert(join.right_variable.clone()); + bound.extend(join.score_variable.iter().cloned()); + } + Self::TextScan(scan) => { + bound.insert(scan.variable.clone()); + bound.extend(scan.score_column.iter().cloned()); + } + Self::LoadData(load) => { + bound.insert(load.variable.clone()); + } + Self::CallProcedure(call) => { + let yields = call.yield_items.as_ref()?; + bound.extend(yields.iter().map(|item| { + item.alias + .clone() + .unwrap_or_else(|| item.field_name.clone()) + })); + } + // The rows of a filtering join are the left side's; those of a + // set operation have the columns of every branch. + Self::AntiJoin(join) => return join.left.bound_variables(imports), + Self::Except(op) => return op.left.bound_variables(imports), + Self::Intersect(op) => return op.left.bound_variables(imports), + Self::Otherwise(op) => return op.left.bound_variables(imports), + Self::Union(union) => return union.inputs.first()?.bound_variables(imports), + Self::MultiWayJoin(join) => { + for input in &join.inputs { + bound.extend(input.bound_variables(imports)?); + } + } + // RDF and graph management plans have no node patterns to close. + Self::TripleScan(_) + | Self::Construct(_) + | Self::InsertTriple(_) + | Self::DeleteTriple(_) + | Self::Modify(_) + | Self::ClearGraph(_) + | Self::CreateGraph(_) + | Self::DropGraph(_) + | Self::LoadGraph(_) + | Self::CopyGraph(_) + | Self::MoveGraph(_) + | Self::AddGraph(_) + | Self::CreatePropertyGraph(_) => return None, + } + Some(bound) + } + + /// Adds the variables a subquery's `RETURN` names, or returns `None` for a + /// `RETURN *` (the translators expand the one that ends a `CALL` + /// subquery, so one left here returns variables not known here). + fn add_returned_variables(&self, bound: &mut HashSet) -> Option<()> { + match self { + Self::Return(ret) => { + for item in &ret.items { + match (&item.alias, &item.expression) { + (Some(alias), _) => { + bound.insert(alias.clone()); + } + (None, LogicalExpression::Variable(name)) if name == "*" => return None, + (None, LogicalExpression::Variable(name)) => { + bound.insert(name.clone()); + } + _ => {} + } + } + Some(()) + } + Self::Sort(sort) => sort.input.add_returned_variables(bound), + Self::Limit(limit) => limit.input.add_returned_variables(bound), + Self::Skip(skip) => skip.input.add_returned_variables(bound), + Self::Distinct(distinct) => distinct.input.add_returned_variables(bound), + // The branches of a UNION return the same names. + Self::Union(union) => union + .inputs + .first() + .map_or(Some(()), |first| first.add_returned_variables(bound)), + _ => Some(()), + } + } +} + impl LogicalOperator { /// Formats this operator tree as a human-readable plan for EXPLAIN output. pub fn explain_tree(&self) -> String { @@ -843,7 +1078,7 @@ impl LogicalOperator { ExpandDirection::Both => "--", }; let hops = match (op.min_hops, op.max_hops) { - (1, Some(1)) => String::new(), + (1, Some(1)) if !op.quantified => String::new(), (min, Some(max)) if min == max => format!("*{min}"), (min, Some(max)) => format!("*{min}..{max}"), (min, None) => format!("*{min}.."), @@ -1199,6 +1434,7 @@ fn fmt_expr(expr: &LogicalExpression) -> String { match expr { LogicalExpression::Variable(name) => name.clone(), LogicalExpression::Property { variable, property } => format!("{variable}.{property}"), + LogicalExpression::MapAccess { base, key } => format!("{}.{key}", fmt_expr(base)), LogicalExpression::Literal(val) => format!("{val}"), LogicalExpression::Binary { left, op, right } => { format!("{} {op:?} {}", fmt_expr(left), fmt_expr(right)) @@ -1286,6 +1522,9 @@ pub struct ExpandOp { pub path_alias: Option, /// Path traversal mode (WALK, TRAIL, SIMPLE, ACYCLIC). pub path_mode: PathMode, + /// Whether the pattern has a quantifier (`*1..1`, `{1,1}`): its edge + /// variable then binds the list of the path's edges, also for one hop. + pub quantified: bool, } /// Direction for edge expansion. @@ -2900,6 +3139,7 @@ mod tests { right: Box::new(LogicalExpression::Literal(Value::Int64(30))), }, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".into(), to_variable: "b".into(), edge_variable: None, @@ -3602,6 +3842,7 @@ mod tests { assert_eq!(edge_scan_any.display_label(), "e:*"); let expand = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".into(), to_variable: "b".into(), edge_variable: None, @@ -3616,6 +3857,7 @@ mod tests { assert_eq!(expand.display_label(), "(a)->[:KNOWS]->(b)"); let expand_in = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".into(), to_variable: "b".into(), edge_variable: None, @@ -3630,6 +3872,7 @@ mod tests { assert_eq!(expand_in.display_label(), "(a)<-[:*]<-(b)"); let expand_both = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".into(), to_variable: "b".into(), edge_variable: None, @@ -4007,8 +4250,9 @@ mod tests { #[test] fn explain_tree_expand_variants() { - let mk = |min, max, dir| { + let mk_quantified = |min, max, dir, quantified| { LogicalOperator::Expand(ExpandOp { + quantified, from_variable: "a".into(), to_variable: "b".into(), edge_variable: None, @@ -4022,9 +4266,13 @@ mod tests { }) .explain_tree() }; + let mk = |min, max, dir| mk_quantified(min, max, dir, min != 1 || max != Some(1)); let s = mk(1, Some(1), ExpandDirection::Outgoing); assert!(s.contains("(a)->[:KNOWS]->(b)")); + // A quantified one hop binds a list of edges, so it shows its quantifier. + let s = mk_quantified(1, Some(1), ExpandDirection::Outgoing, true); + assert!(s.contains("(a)->[:KNOWS*1]->(b)"), "{s}"); let s = mk(2, Some(2), ExpandDirection::Incoming); assert!(s.contains("*2")); assert!(s.contains("<-")); @@ -4503,6 +4751,15 @@ mod tests { }; assert_eq!(fmt_expr(&p), "n.age"); + let route = LogicalExpression::MapAccess { + base: Box::new(LogicalExpression::Property { + variable: "n".into(), + property: "meta".into(), + }), + key: "route".into(), + }; + assert_eq!(fmt_expr(&route), "n.meta.route"); + let lit = LogicalExpression::Literal(Value::Int64(42)); assert_eq!(fmt_expr(&lit), "42"); @@ -4587,4 +4844,40 @@ mod tests { None ); } + + /// A `CALL` adds the variables its subquery returns to the row. A + /// `RETURN *` left unexpanded returns variables not known here, so the + /// row's are not known either (the translators expand the one that ends a + /// `CALL` subquery). + #[test] + fn bound_variables_through_a_call_subquery() { + let scan = |variable: &str| { + LogicalOperator::NodeScan(NodeScanOp { + variable: variable.into(), + label: None, + input: None, + }) + }; + let item = |name: &str| ReturnItem { + expression: LogicalExpression::Variable(name.into()), + alias: None, + }; + let call = |items: Vec| { + LogicalOperator::Apply(ApplyOp { + input: Box::new(scan("a")), + subplan: Box::new(LogicalOperator::Return(ReturnOp { + items, + distinct: false, + input: Box::new(scan("b")), + })), + shared_variables: vec![], + optional: false, + }) + }; + assert_eq!( + call(vec![item("b")]).bound_variables(None), + Some(HashSet::from(["a".to_string(), "b".to_string()])) + ); + assert_eq!(call(vec![item("*")]).bound_variables(None), None); + } } diff --git a/crates/grafeo-engine/src/query/planner/common.rs b/crates/grafeo-engine/src/query/planner/common.rs index 25312461c..df80be979 100644 --- a/crates/grafeo-engine/src/query/planner/common.rs +++ b/crates/grafeo-engine/src/query/planner/common.rs @@ -11,9 +11,9 @@ use crate::query::plan::{ use grafeo_common::types::{LogicalType, Value}; use grafeo_common::utils::error::{Error, Result}; use grafeo_core::execution::operators::{ - DistinctOperator, ExceptOperator, HashJoinOperator, IntersectOperator, - JoinType as PhysicalJoinType, LimitOperator, Operator, OtherwiseOperator, ProjectExpr, - ProjectOperator, SkipOperator, UnionOperator, + DistinctOperator, ExceptOperator, HashJoinOperator, IntersectOperator, JoinCondition, + JoinType as PhysicalJoinType, LimitOperator, NullOrder, Operator, OtherwiseOperator, + ProjectExpr, ProjectOperator, SkipOperator, UnionOperator, }; /// Builds a LIMIT physical operator. @@ -21,9 +21,8 @@ pub(crate) fn build_limit( input: Box, columns: Vec, count: usize, - schema: Vec, ) -> (Box, Vec) { - let operator = Box::new(LimitOperator::new(input, count, schema)); + let operator = Box::new(LimitOperator::new(input, count)); (operator, columns) } @@ -32,9 +31,8 @@ pub(crate) fn build_skip( input: Box, columns: Vec, count: usize, - schema: Vec, ) -> (Box, Vec) { - let operator = Box::new(SkipOperator::new(input, count, schema)); + let operator = Box::new(SkipOperator::new(input, count)); (operator, columns) } @@ -45,7 +43,6 @@ pub(crate) fn build_distinct( input: Box, columns: Vec, distinct_columns: Option<&[String]>, - schema: Vec, ) -> (Box, Vec) { let operator: Box = if let Some(dist_cols) = distinct_columns { let col_indices: Vec = dist_cols @@ -53,12 +50,12 @@ pub(crate) fn build_distinct( .filter_map(|name| columns.iter().position(|c| c == name)) .collect(); if col_indices.is_empty() { - Box::new(DistinctOperator::new(input, schema)) + Box::new(DistinctOperator::new(input)) } else { - Box::new(DistinctOperator::on_columns(input, col_indices, schema)) + Box::new(DistinctOperator::on_columns(input, col_indices)) } } else { - Box::new(DistinctOperator::new(input, schema)) + Box::new(DistinctOperator::new(input)) }; (operator, columns) } @@ -84,9 +81,8 @@ pub(crate) fn build_except( right: Box, columns: Vec, all: bool, - schema: Vec, ) -> (Box, Vec) { - let operator = Box::new(ExceptOperator::new(left, right, all, schema)); + let operator = Box::new(ExceptOperator::new(left, right, all)); (operator, columns) } @@ -96,9 +92,8 @@ pub(crate) fn build_intersect( right: Box, columns: Vec, all: bool, - schema: Vec, ) -> (Box, Vec) { - let operator = Box::new(IntersectOperator::new(left, right, all, schema)); + let operator = Box::new(IntersectOperator::new(left, right, all)); (operator, columns) } @@ -306,6 +301,7 @@ pub(crate) fn build_left_join( right_columns: &[String], left_types: &[LogicalType], right_types: &[LogicalType], + residual: Option>, ) -> (Box, Vec, Vec) { let (probe_keys, build_keys) = find_shared_join_keys(left_columns, right_columns); @@ -315,14 +311,18 @@ pub(crate) fn build_left_join( let mut join_schema: Vec = left_types.to_vec(); join_schema.extend(right_types.iter().cloned()); - let join_op: Box = Box::new(HashJoinOperator::new( + let mut hash_join = HashJoinOperator::new( left, right, probe_keys, build_keys, PhysicalJoinType::Left, join_schema.clone(), - )); + ); + if let Some(residual) = residual { + hash_join = hash_join.with_residual(residual); + } + let join_op: Box = Box::new(hash_join); // Deduplicate: keep left columns, then only right columns not already on the left let left_set: std::collections::HashSet<&str> = @@ -425,6 +425,18 @@ pub(crate) fn resolve_expression_to_column( }) } +/// Where a sort key puts nulls: as its `NULLS FIRST` or `NULLS LAST` says, +/// in either direction, and otherwise as the largest value (last ascending, +/// first descending), as in openCypher. +pub(crate) fn physical_null_order(key: &crate::query::plan::SortKey) -> NullOrder { + use crate::query::plan::{NullsOrdering, SortOrder}; + + match (key.nulls, key.order) { + (Some(NullsOrdering::First), _) | (None, SortOrder::Descending) => NullOrder::NullsFirst, + (Some(NullsOrdering::Last), _) | (None, SortOrder::Ascending) => NullOrder::NullsLast, + } +} + /// Whether a plan's rows come out in an order it defines: an `ORDER BY` at /// the top, possibly under a projection, `RETURN`, `DISTINCT`, `SKIP` or /// `LIMIT`. @@ -906,6 +918,7 @@ mod tests { &right_cols, &left_types, &right_types, + None, ); assert_eq!(output_columns, vec!["s", "name", "age"]); diff --git a/crates/grafeo-engine/src/query/planner/lpg/aggregate.rs b/crates/grafeo-engine/src/query/planner/lpg/aggregate.rs index ac20edc12..22df4c76c 100644 --- a/crates/grafeo-engine/src/query/planner/lpg/aggregate.rs +++ b/crates/grafeo-engine/src/query/planner/lpg/aggregate.rs @@ -1,9 +1,9 @@ //! Aggregate and factorized aggregate planning. use super::{ - AggregateOp, Arc, Direction, Error, ExpandDirection, ExpandStep, ExpressionPredicate, - FactorizedAggregate, FactorizedAggregateOperator, FilterExpression, FilterOperator, - GraphStoreSearch, HashAggregateOperator, HashMap, LazyFactorizedChainOperator, + AggregateOp, Arc, Direction, EntityValue, Error, ExpandDirection, ExpandStep, + ExpressionPredicate, FactorizedAggregate, FactorizedAggregateOperator, FilterExpression, + FilterOperator, GraphStoreSearch, HashAggregateOperator, HashMap, LazyFactorizedChainOperator, LogicalAggregateFunction, LogicalExpression, LogicalType, Operator, PhysicalAggregateExpr, ProjectExpr, ProjectOperator, Result, SimpleAggregateOperator, convert_aggregate_function, expression_to_string, resolved_column_name, @@ -208,18 +208,44 @@ impl super::Planner { }) .collect::>>()?; - // Build output schema and column names + // Build output schema and column names, and what each column holds: a + // group key that is a node or an edge (or a list of them) stays one, + // and so does the list `collect` makes of nodes or edges. Every other + // column holds values. let mut output_schema = Vec::new(); let mut output_columns = Vec::new(); + let mut output_entities = Vec::new(); // Add group-by columns for expr in &agg.group_by { - output_schema.push(LogicalType::Any); // Group-by values can be any type + let entity = match expr { + LogicalExpression::Variable(name) => self.column_entity(name), + _ => None, + }; + output_schema.push(entity_type(entity)); output_columns.push(expression_to_string(expr)); + output_entities.push(entity); } // Add aggregate result columns for agg_expr in &agg.aggregates { + let collected = match (agg_expr.function, &agg_expr.expression) { + // The list of what `collect` gathers keeps its kind: a node or + // edge column, or an expression that yields one (`head(rs)`, + // `last(relationships(p))`). + (LogicalAggregateFunction::Collect, Some(expression)) => { + let item = match expression { + LogicalExpression::Variable(name) => self.column_entity(name), + other => self.entity_value(other), + }; + match item { + Some(EntityValue::Node) => Some(EntityValue::Nodes), + Some(EntityValue::Edge) => Some(EntityValue::Edges), + _ => None, + } + } + _ => None, + }; let result_type = match agg_expr.function { LogicalAggregateFunction::Count | LogicalAggregateFunction::CountNonNull => { LogicalType::Int64 @@ -232,7 +258,8 @@ impl super::Planner { // to avoid type mismatch when pushing the finalized value. LogicalType::Any } - LogicalAggregateFunction::Collect => LogicalType::Any, // List type (using Any since List is a complex type) + // A list of nodes or edges, or of any values + LogicalAggregateFunction::Collect => entity_type(collected), LogicalAggregateFunction::GroupConcat => LogicalType::String, LogicalAggregateFunction::Sample => LogicalType::Any, // Statistical functions return Float64 @@ -262,12 +289,11 @@ impl super::Planner { crate::query::planner::common::aggregate_column_name(agg_expr) }), ); + output_entities.push(collected); } - // Register all aggregate output columns as scalar (group-by values and - // aggregate results are materialized scalar values, not entity references) - for col in &output_columns { - self.scalar_columns.borrow_mut().insert(col.clone()); + for (column, entity) in output_columns.iter().zip(&output_entities) { + self.set_column_entity(column, *entity); } // Choose operator based on whether there are group-by columns @@ -474,3 +500,15 @@ impl super::Planner { crate::query::planner::common::resolve_expression_to_column(expr, variable_columns, "") } } + +/// The declared type of a column that holds `entity`: a node, an edge, a list +/// of them, or any value. +fn entity_type(entity: Option) -> LogicalType { + match entity { + Some(EntityValue::Node) => LogicalType::Node, + Some(EntityValue::Edge) => LogicalType::Edge, + Some(EntityValue::Nodes) => LogicalType::List(Box::new(LogicalType::Node)), + Some(EntityValue::Edges) => LogicalType::List(Box::new(LogicalType::Edge)), + _ => LogicalType::Any, + } +} diff --git a/crates/grafeo-engine/src/query/planner/lpg/expand.rs b/crates/grafeo-engine/src/query/planner/lpg/expand.rs index 6fe6407fe..d43ab7ea5 100644 --- a/crates/grafeo-engine/src/query/planner/lpg/expand.rs +++ b/crates/grafeo-engine/src/query/planner/lpg/expand.rs @@ -33,19 +33,20 @@ impl super::Planner { ExpandDirection::Both => Direction::Both, }; - // Check if this is a variable-length path + // Check if this is a variable-length path. The GQL, Cypher and SQL/PGQ + // translators set `quantified` for one; other plans (Gremlin, built + // in code) only set hop bounds, so those count too. let is_variable_length = - expand.min_hops != 1 || expand.max_hops.is_none() || expand.max_hops != Some(1); + expand.quantified || expand.min_hops != 1 || expand.max_hops != Some(1); // Use VariableLengthExpandOperator when multi-hop OR when a named path // needs path detail columns (length, nodes, edges) let needs_path_details = expand.path_alias.is_some(); - // Translators name the edges they need internally with a leading `_`. - let binds_edge_list = is_variable_length - && expand - .edge_variable - .as_deref() - .is_some_and(|name| !name.starts_with('_')); + // A named edge variable of a variable-length pattern binds the list of + // the path's edges. The variable a translator adds for an anonymous + // edge with a property map gets one too: that map is checked over + // the path's edges, never on this column. + let binds_edge_list = is_variable_length && expand.edge_variable.is_some(); let operator: Box = if is_variable_length || needs_path_details { // Use VariableLengthExpandOperator for multi-hop paths or named paths @@ -85,8 +86,6 @@ impl super::Planner { .with_path_length_output() .with_path_detail_output(); } - // A named edge variable of a variable-length pattern binds the - // list of the path's edges. if binds_edge_list { expand_op = expand_op.with_edge_list_output(); } diff --git a/crates/grafeo-engine/src/query/planner/lpg/expression.rs b/crates/grafeo-engine/src/query/planner/lpg/expression.rs index 5122e6527..bc0ce31b8 100644 --- a/crates/grafeo-engine/src/query/planner/lpg/expression.rs +++ b/crates/grafeo-engine/src/query/planner/lpg/expression.rs @@ -17,8 +17,8 @@ //! inline. use super::{ - Direction, Error, ExpandDirection, ExpandOp, FilterExpression, LogicalExpression, - LogicalOperator, Result, Value, convert_binary_op, convert_unary_op, + Direction, Error, ExpandDirection, FilterExpression, LogicalExpression, LogicalOperator, + PathMode, Result, Value, convert_binary_op, convert_unary_op, }; impl super::Planner { @@ -189,30 +189,34 @@ impl super::Planner { }) } LogicalExpression::ExistsSubquery(subplan) => { - // Extract the pattern from the subplan - // For EXISTS { MATCH (n)-[:TYPE]->() }, we extract start_var, direction, edge_type - let (start_var, direction, edge_types, end_labels) = - self.extract_exists_pattern(subplan)?; - + // For EXISTS { MATCH (n)-[:TYPE]->() }: a check on n's own edges. + let check = self.extract_exists_pattern(subplan)?; Ok(FilterExpression::ExistsSubquery { - start_var, - direction, - edge_types, - end_labels, - min_hops: None, - max_hops: None, + start_var: check.start_var, + end_var: check.end_var, + edge_var: check.edge_var, + direction: check.direction, + edge_types: check.edge_types, + end_labels: check.end_labels, + min_hops: (!check.one_edge).then_some(1), + max_hops: check.max_hops, }) } LogicalExpression::CountSubquery(subplan) => { - // Reuse the same pattern extraction as EXISTS (fast path for simple edges) - let (start_var, direction, edge_types, end_labels) = - self.extract_exists_pattern(subplan)?; - + // The same edge check, counted; only a single edge counts the same. + let check = self.extract_exists_pattern(subplan)?; + if !check.one_edge { + return Err(Error::Internal( + "Unsupported COUNT subquery pattern".to_string(), + )); + } Ok(FilterExpression::CountSubquery { - start_var, - direction, - edge_types, - end_labels, + start_var: check.start_var, + end_var: check.end_var, + edge_var: check.edge_var, + direction: check.direction, + edge_types: check.edge_types, + end_labels: check.end_labels, }) } LogicalExpression::ValueSubquery(_) => { @@ -281,30 +285,42 @@ impl super::Planner { } } - /// Extracts the pattern from an EXISTS subplan for the simple single-hop fast path. + /// Extracts the edge check that an `EXISTS` subplan reduces to, for the + /// fast path that looks only at the start node's own edges. /// - /// Returns `(start_variable, direction, edge_type, end_labels)` only for bare - /// single-hop patterns like `(n)-[:TYPE]->()` or `()-[:TYPE]->(n)`. Rejects - /// multi-hop patterns, inner WHERE filters, label constraints on target nodes, - /// and non-correlated patterns (both endpoints anonymous), all of which are - /// handled correctly by the semi-join rewrite in `plan_filter`. + /// Accepts a single edge, like `(n)-[:TYPE]->()`, `(n)-[:TYPE]->(:Label)` + /// or `()-[:TYPE]->(n)`, and a path of at least one edge with no + /// condition on its end, like `(n)-[:TYPE*]->()`, in the WALK path mode. + /// Everything else goes to the semi-join rewrite in `plan_filter`: a path + /// with a minimum other than one hop, a labeled end or another path mode, + /// a label on the start, an inner `WHERE` other than a label on the end, + /// a node pattern apart from the edge, and patterns with no named node. /// - /// When the correlated variable appears on the target side of the pattern - /// (e.g., `()-[:CALLS]->(m)` where `m` is from the outer scope), the direction - /// is flipped so the runtime can evaluate from the correlated node. - pub(super) fn extract_exists_pattern( - &self, - subplan: &LogicalOperator, - ) -> Result<(String, Direction, Vec, Option>)> { + /// The start is the pattern's named source, or its target when the source + /// is anonymous (e.g. `()-[:CALLS]->(m)`, with the direction flipped). + /// Which of the pattern's variables the outer row binds is known only per + /// row: the evaluation matches its end and edge to the row's when it binds + /// them, and `plan_filter` takes the fast path only for a start the + /// outer row binds. + pub(super) fn extract_exists_pattern(&self, subplan: &LogicalOperator) -> Result { + let unsupported = || Error::Internal("Unsupported EXISTS subquery pattern".to_string()); match subplan { LogicalOperator::Expand(expand) => { - // Only accept single-hop: the Expand's input (source plan) must be - // a plain NodeScan. Another Expand means multi-hop; a Filter means - // inner WHERE or label constraint. Both require the semi-join path. - if !matches!(expand.input.as_ref(), LogicalOperator::NodeScan(_)) { - return Err(Error::Internal( - "Unsupported EXISTS subquery pattern".to_string(), - )); + // The Expand's input must be the plain scan of its source: another + // Expand means more edges, a Filter an inner WHERE or a second label, + // and a scan with an input a pattern before this one. + let LogicalOperator::NodeScan(source) = expand.input.as_ref() else { + return Err(unsupported()); + }; + if expand.min_hops != 1 || source.input.is_some() { + return Err(unsupported()); + } + let one_edge = expand.max_hops == Some(1) && !expand.quantified; + // A path mode other than WALK (TRAIL, SIMPLE, ACYCLIC) limits the + // paths of a longer pattern, which the check does not; a single + // edge is expanded the same way in every mode. + if !one_edge && expand.path_mode != PathMode::Walk { + return Err(unsupported()); } let from_is_anon = expand.from_variable.starts_with("_anon_"); @@ -319,105 +335,132 @@ impl super::Planner { } if from_is_anon { - // Correlated variable is on the target side, e.g. ()-[:CALLS]->(m). - // Flip direction: "does m have an incoming CALLS edge?" + // Outer variable on the target side, e.g. ()-[:CALLS]->(m). + // Flip direction: "does m have an incoming CALLS edge?" The + // source's label becomes a label of the far end. let direction = match expand.direction { ExpandDirection::Outgoing => Direction::Incoming, ExpandDirection::Incoming => Direction::Outgoing, ExpandDirection::Both => Direction::Both, }; - let end_labels = self.extract_source_labels_from_expand(expand); - Ok(( - expand.to_variable.clone(), + let end_labels = source.label.clone().map(|label| vec![label]); + if end_labels.is_some() && !one_edge { + return Err(unsupported()); + } + Ok(EdgeCheck { + start_var: expand.to_variable.clone(), + end_var: expand.from_variable.clone(), + edge_var: expand.edge_variable.clone(), direction, - expand.edge_types.clone(), + edge_types: expand.edge_types.clone(), end_labels, - )) + one_edge, + max_hops: expand.max_hops, + }) } else { - // Normal case: correlated variable on the source side, e.g. (m)-[:CALLS]->() - // - // No end_labels: the Expand's input is the source NodeScan, whose labels - // belong to the correlated variable (already filtered by the outer scope). - // Target labels would create a Filter wrapping the Expand, which is - // rejected above and correctly routed to the semi-join path. + // Outer variable on the source side, e.g. (m)-[:CALLS]->(). A + // label on it here (`(m:Label)-[:CALLS]->()`) is a condition + // the edge check cannot make. + if source.label.is_some() { + return Err(unsupported()); + } let direction = match expand.direction { ExpandDirection::Outgoing => Direction::Outgoing, ExpandDirection::Incoming => Direction::Incoming, ExpandDirection::Both => Direction::Both, }; - Ok(( - expand.from_variable.clone(), + Ok(EdgeCheck { + start_var: expand.from_variable.clone(), + end_var: expand.to_variable.clone(), + edge_var: expand.edge_variable.clone(), direction, - expand.edge_types.clone(), - None, - )) + edge_types: expand.edge_types.clone(), + end_labels: None, + one_edge, + max_hops: expand.max_hops, + }) } } + // A node pattern after the edge, like the second MATCH of + // `MATCH (n)-[:R]->(m) MATCH (m:Label)`, reuses the node of the edge it + // names: a label on the end joins the end labels. Any other node is a + // pattern of its own, which the edge check cannot make. LogicalOperator::NodeScan(scan) => { - if let Some(input) = &scan.input { - self.extract_exists_pattern(input) - } else { - Err(Error::Internal( + let Some(input) = &scan.input else { + return Err(Error::Internal( "EXISTS subquery must contain an edge pattern".to_string(), - )) + )); + }; + let mut check = self.extract_exists_pattern(input)?; + if scan.variable == check.end_var && (scan.label.is_none() || check.one_edge) { + if let Some(label) = &scan.label { + check + .end_labels + .get_or_insert_with(Vec::new) + .push(label.clone()); + } + Ok(check) + } else if scan.variable == check.start_var && scan.label.is_none() { + Ok(check) + } else { + Err(unsupported()) } } - // A Filter wrapping an Expand typically arises from a label constraint - // on the anonymous endpoint, e.g. EXISTS { (u)<-[:AUTH]-(:Identity) }. - // Extract the inner Expand pattern and fold the label filter into end_labels. - // Only pure hasLabel filters are supported; property filters (WHERE m.age > 30) - // must go through the semi-join path for correct evaluation. + // A label on the far end of a single edge, e.g. + // EXISTS { (u)<-[:AUTH]-(:Identity) }, joins the end labels. Any other + // filter (a property, a label on another variable, a label on the end + // of a longer path) goes to the semi-join path. LogicalOperator::Filter(filter_op) => { - let end_labels = self.extract_labels_from_filter_predicate(&filter_op.predicate); - match end_labels { - Some(labels) => { - let (start_var, direction, edge_types, _) = - self.extract_exists_pattern(&filter_op.input)?; - Ok((start_var, direction, edge_types, Some(labels))) - } - None => { - // Non-label filter (property checks, etc.): reject so the - // planner uses the semi-join rewrite instead. - Err(Error::Internal( - "Unsupported EXISTS subquery pattern".to_string(), - )) - } + let (variable, label) = self + .label_condition(&filter_op.predicate) + .ok_or_else(unsupported)?; + let mut check = self.extract_exists_pattern(&filter_op.input)?; + if variable != check.end_var || !check.one_edge { + return Err(unsupported()); } + check.end_labels.get_or_insert_with(Vec::new).push(label); + Ok(check) } - _ => Err(Error::Internal( - "Unsupported EXISTS subquery pattern".to_string(), - )), + _ => Err(unsupported()), } } - /// Extracts label names from a hasLabel filter predicate. - /// - /// Given `FunctionCall { name: "hasLabel", args: [Variable(x), Literal("Label")] }`, - /// returns `Some(vec!["Label"])`. - fn extract_labels_from_filter_predicate( - &self, - predicate: &LogicalExpression, - ) -> Option> { + /// The variable and label of a `hasLabel(variable, 'Label')` predicate. + fn label_condition<'a>(&self, predicate: &'a LogicalExpression) -> Option<(&'a str, String)> { match predicate { LogicalExpression::FunctionCall { name, args, .. } if name == "hasLabel" => { - if let Some(LogicalExpression::Literal(Value::String(label))) = args.get(1) { - Some(vec![label.to_string()]) - } else { - None + match (args.first(), args.get(1)) { + ( + Some(LogicalExpression::Variable(variable)), + Some(LogicalExpression::Literal(Value::String(label))), + ) => Some((variable, label.to_string())), + _ => None, } } _ => None, } } +} - /// Extracts source (input) node labels from an Expand for the flipped EXISTS case. - /// - /// When the pattern is `(:Label)-[:TYPE]->(m)` and we flip to start from `m`, - /// the source node's labels become the "end" labels for the reversed traversal. - fn extract_source_labels_from_expand(&self, expand: &ExpandOp) -> Option> { - match expand.input.as_ref() { - LogicalOperator::NodeScan(scan) => scan.label.clone().map(|l| vec![l]), - _ => None, - } - } +/// The check on a node's own edges that an `EXISTS` or `COUNT` subquery +/// reduces to (see `extract_exists_pattern`). +pub(super) struct EdgeCheck { + /// The variable the check starts from. + pub start_var: String, + /// The pattern's other end. + pub end_var: String, + /// The pattern's edge variable, if it has one. + pub edge_var: Option, + /// The direction of the edge, seen from the start. + pub direction: Direction, + /// The edge types to match (empty matches any). + pub edge_types: Vec, + /// The labels the other end must all have. + pub end_labels: Option>, + /// Whether the pattern is a single edge. A longer path is accepted only for + /// `EXISTS`, where it exists when its first edge does (unless the row + /// binds its other end or edge), but counts differently. + pub one_edge: bool, + /// The maximum number of hops of a longer path (`None`: unbounded). + pub max_hops: Option, } diff --git a/crates/grafeo-engine/src/query/planner/lpg/filter.rs b/crates/grafeo-engine/src/query/planner/lpg/filter.rs index c5b3b97c0..0575b4a11 100644 --- a/crates/grafeo-engine/src/query/planner/lpg/filter.rs +++ b/crates/grafeo-engine/src/query/planner/lpg/filter.rs @@ -3,10 +3,11 @@ //! The order of the rewrite/pushdown attempts in [`Planner::plan_filter`][pf] //! is load-bearing, not a performance tweak: //! -//! 1. Subquery rewrites (`extract_complex_exists`, `extract_exists_from_or`, -//! `extract_count_comparison`) must run first. The later steps assume -//! a scalar predicate with no subquery shape; once a rewrite fires, -//! the rest of the method is bypassed. +//! 1. Subquery rewrites (`extract_complex_exists` for semi-joins, then the +//! subqueries planned per row, `plan_filter_with_subqueries`) must run +//! first. The later steps assume a predicate whose subqueries +//! the edge check answers; once a rewrite fires, the rest of the method +//! is bypassed. //! 2. Zone-map short-circuit fires before any index lookup so we can //! skip opening an index file at all when summary statistics prove //! emptiness. @@ -23,14 +24,18 @@ //! //! [pf]: super::Planner::plan_filter +use std::collections::HashSet; + use grafeo_common::collections::GrafeoSet; +use super::subquery::reads_outer_values; +use grafeo_common::types::NodeId; + use super::{ - ApplyOperator, Arc, BinaryOp, DistinctOperator, EmptyOperator, ExpressionPredicate, - FilterExpression, FilterOp, FilterOperator, GraphStoreSearch, HashAggregateOperator, - HashJoinOperator, HashMap, LogicalExpression, LogicalOperator, NodeListOperator, Operator, - PhysicalAggregateExpr, PhysicalJoinType, RangeBounds, RangeScanOperator, Result, TransactionId, - UnaryOp, UnionOperator, Value, convert_binary_op, convert_filter_expression, + Arc, BinaryOp, EmptyOperator, Error, ExpressionPredicate, FilterOp, FilterOperator, + GraphStoreSearch, HashJoinOperator, HashMap, LogicalExpression, LogicalOperator, + NodeListOperator, Operator, PhysicalJoinType, RangeBounds, RangeScanOperator, Result, + TransactionId, UnaryOp, Value, }; /// Cross-type equality comparison with Int64/Float64 coercion. @@ -43,6 +48,17 @@ fn values_equal_coerced(a: &Value, b: &Value) -> bool { } impl super::Planner { + /// Whether node `id` is visible to this query and, when `label` is given, + /// carries it. An index lookup checks its few results this way, instead of + /// collecting every node with the label. + fn visible_with_label(&self, id: NodeId, label: Option<&str>) -> bool { + let node = match self.transaction_id { + Some(tx) => self.store.get_node_versioned(id, self.viewing_epoch, tx), + None => self.store.get_node_at_epoch(id, self.viewing_epoch), + }; + node.is_some_and(|node| label.is_none_or(|label| node.has_label(label))) + } + /// Plans a filter operator. /// /// Uses zone map pre-filtering to potentially skip scans when predicates @@ -54,29 +70,25 @@ impl super::Planner { filter: &FilterOp, ) -> Result<(Box, Vec)> { // Check for complex EXISTS/NOT EXISTS patterns and rewrite as semi/anti join. - // Simple single-hop EXISTS patterns are handled by the fast path in - // convert_expression() -> extract_exists_pattern(). + // Simple single-hop EXISTS patterns from a node of the row are handled + // by the fast path in convert_expression() -> extract_exists_pattern(). + let outer = filter.input.bound_variables(None); if let Some((subquery, is_negated, remaining)) = - self.extract_complex_exists(&filter.predicate) + self.extract_complex_exists(&filter.predicate, outer.as_ref()) { return self.plan_exists_as_semi_join(&filter.input, subquery, is_negated, remaining); } - // Complex EXISTS inside OR predicates can't use semi-join (which filters). - // Instead, split the OR into two branches (EXISTS via semi-join, scalar - // via filter), union the results, and deduplicate. - if let Some((subquery, is_negated, other_pred)) = - self.extract_exists_from_or(&filter.predicate) - { - return self.plan_exists_or_as_union(&filter.input, subquery, is_negated, other_pred); - } - - // Check for COUNT subquery comparisons and rewrite as Apply + Aggregate + Filter. - // Handles patterns like: COUNT { MATCH ... } > 5, COUNT { ... } = 0, etc. - if let Some((subquery, op, threshold, remaining)) = - Self::extract_count_comparison(&filter.predicate) - { - return self.plan_count_as_apply(&filter.input, subquery, op, threshold, remaining); + // EXISTS and COUNT subqueries the edge check cannot answer run per row + // of the input (see `subquery.rs`). + // With the variables the input's rows hold, a subquery that shares + // none of them is lifted too (counted once instead of per row). + let input_columns: Option> = filter + .input + .bound_variables(None) + .map(|names| names.into_iter().collect()); + if self.has_subquery_to_lift(&filter.predicate, input_columns.as_deref()) { + return self.plan_filter_with_subqueries(filter); } // Check zone maps for simple property predicates before scanning @@ -163,6 +175,65 @@ impl super::Planner { Ok((operator, columns)) } + /// Plans a filter whose predicate has `EXISTS` or `COUNT` subqueries that + /// run per row: the parts of an `AND` without one are a filter of their + /// own first (which may use an index), then each subquery adds its count + /// to the row and the rest of the predicate reads it. + fn plan_filter_with_subqueries( + &self, + filter: &FilterOp, + ) -> Result<(Box, Vec)> { + let mut conjuncts = Vec::new(); + split_conjuncts(&filter.predicate, &mut conjuncts); + let input_columns: Option> = filter + .input + .bound_variables(None) + .map(|names| names.into_iter().collect()); + let (with_subqueries, plain): (Vec<_>, Vec<_>) = conjuncts + .into_iter() + .partition(|conjunct| self.has_subquery_to_lift(conjunct, input_columns.as_deref())); + let (input_op, columns) = match join_conjuncts(plain) { + Some(predicate) => self.plan_filter(&FilterOp { + predicate, + input: filter.input.clone(), + pushdown_hint: filter.pushdown_hint.clone(), + })?, + None => self.plan_operator(&filter.input)?, + }; + let predicate = join_conjuncts(with_subqueries) + .ok_or_else(|| Error::Internal("filter without a subquery to lift".to_string()))?; + self.filter_rest(input_op, columns, &predicate, filter.input.has_mutations()) + } + + /// Filters `input` by `predicate`, whose `EXISTS` and `COUNT` subqueries + /// the edge check cannot answer run per row first (see `subquery.rs`). + fn filter_rest( + &self, + input: Box, + columns: Vec, + predicate: &LogicalExpression, + input_writes: bool, + ) -> Result<(Box, Vec)> { + let (predicate, input, columns) = + self.lift_subqueries(predicate, input, columns, input_writes)?; + let variable_columns: HashMap = columns + .iter() + .enumerate() + .map(|(i, name)| (name.clone(), i)) + .collect(); + let predicate = ExpressionPredicate::new( + self.convert_expression(&predicate)?, + variable_columns, + Arc::clone(&self.store) as Arc, + ) + .with_transaction_context(self.viewing_epoch, self.transaction_id) + .with_session_context(self.session_context.clone()); + Ok(( + Box::new(FilterOperator::new(input, Box::new(predicate))), + columns, + )) + } + /// Extracts an EXISTS or NOT EXISTS subquery from a filter predicate for /// semi-join rewriting. /// @@ -180,18 +251,22 @@ impl super::Planner { /// depths. The recursive semi-join handler (`plan_exists_as_semi_join`) calls /// this function again on each remaining predicate, peeling off one EXISTS /// per level until only scalar predicates remain. + /// + /// `outer` holds the variables of the rows the predicate filters (`None`: + /// not known), see [`Self::exists_fast_path_fits`]. fn extract_complex_exists<'a>( &self, predicate: &'a LogicalExpression, + outer: Option<&HashSet>, ) -> Option<(&'a LogicalOperator, bool, Option)> { match predicate { LogicalExpression::ExistsSubquery(subplan) => { // Top-level EXISTS: only use semi-join for complex patterns. // Simple single-hop patterns use the fast path in convert_expression(). - if self.extract_exists_pattern(subplan).is_err() { - Some((subplan.as_ref(), false, None)) - } else { + if self.exists_fast_path_fits(subplan, outer) || reads_outer_values(subplan) { None + } else { + Some((subplan.as_ref(), false, None)) } } LogicalExpression::Unary { @@ -199,10 +274,10 @@ impl super::Planner { operand, } => { if let LogicalExpression::ExistsSubquery(subplan) = operand.as_ref() { - if self.extract_exists_pattern(subplan).is_err() { - Some((subplan.as_ref(), true, None)) - } else { + if self.exists_fast_path_fits(subplan, outer) || reads_outer_values(subplan) { None + } else { + Some((subplan.as_ref(), true, None)) } } else { None @@ -217,15 +292,20 @@ impl super::Planner { // When multiple EXISTS appear in the same WHERE, extracting the // first one (even if simple) lets the recursive semi-join handler // find and extract the remaining complex ones from the rest. - if let Some((subplan, negated)) = Self::extract_exists_from_expr(left) { + if let Some((subplan, negated)) = Self::extract_exists_from_expr(left) + && !reads_outer_values(subplan) + { return Some((subplan, negated, Some(right.as_ref().clone()))); } - if let Some((subplan, negated)) = Self::extract_exists_from_expr(right) { + if let Some((subplan, negated)) = Self::extract_exists_from_expr(right) + && !reads_outer_values(subplan) + { return Some((subplan, negated, Some(left.as_ref().clone()))); } // Recurse into left subtree (handles left-leaning AND trees where // EXISTS nodes are buried deeper than immediate children) - if let Some((subplan, negated, inner_remaining)) = self.extract_complex_exists(left) + if let Some((subplan, negated, inner_remaining)) = + self.extract_complex_exists(left, outer) { let remaining = match inner_remaining { Some(inner) => LogicalExpression::Binary { @@ -239,7 +319,7 @@ impl super::Planner { } // Recurse into right subtree if let Some((subplan, negated, inner_remaining)) = - self.extract_complex_exists(right) + self.extract_complex_exists(right, outer) { let remaining = match inner_remaining { Some(inner) => LogicalExpression::Binary { @@ -257,6 +337,29 @@ impl super::Planner { } } + /// Whether an `EXISTS` subplan takes the fast path for rows that bind + /// `outer` (`None`: not known): its pattern must start from a node of the + /// row, and its other end and edge must be new to it. Anything else is a + /// semi-join, which matches every variable the subquery shares with the + /// row by name. + fn exists_fast_path_fits( + &self, + subplan: &LogicalOperator, + outer: Option<&HashSet>, + ) -> bool { + let Ok(check) = self.extract_exists_pattern(subplan) else { + return false; + }; + outer.is_some_and(|outer| { + outer.contains(&check.start_var) + && !outer.contains(&check.end_var) + && check + .edge_var + .as_ref() + .is_none_or(|edge| !outer.contains(edge)) + }) + } + /// Helper: extracts EXISTS or NOT EXISTS from a single expression node. /// Returns `(subplan, is_negated)`. fn extract_exists_from_expr(expr: &LogicalExpression) -> Option<(&LogicalOperator, bool)> { @@ -276,73 +379,6 @@ impl super::Planner { } } - /// Checks if a top-level OR predicate contains a complex EXISTS subquery that - /// cannot use the inline fast path. Returns the EXISTS subplan, its negation - /// flag, and the other side of the OR. - fn extract_exists_from_or<'a>( - &self, - predicate: &'a LogicalExpression, - ) -> Option<(&'a LogicalOperator, bool, &'a LogicalExpression)> { - let LogicalExpression::Binary { - op: BinaryOp::Or, - left, - right, - } = predicate - else { - return None; - }; - - // Check left side for complex EXISTS - if let Some((subplan, negated)) = Self::extract_exists_from_expr(left) - && self.extract_exists_pattern(subplan).is_err() - { - return Some((subplan, negated, right)); - } - // Check right side for complex EXISTS - if let Some((subplan, negated)) = Self::extract_exists_from_expr(right) - && self.extract_exists_pattern(subplan).is_err() - { - return Some((subplan, negated, left)); - } - None - } - - /// Plans a filter with `complex_EXISTS OR other_pred` using Union + Distinct. - /// - /// The OR is split into two branches: - /// 1. Semi-join (or anti-join for NOT EXISTS) for the EXISTS branch - /// 2. Normal filter for the scalar predicate branch - /// - /// The results are combined with UNION ALL and deduplicated. - fn plan_exists_or_as_union( - &self, - input: &LogicalOperator, - subquery: &LogicalOperator, - is_negated: bool, - other_predicate: &LogicalExpression, - ) -> Result<(Box, Vec)> { - // Branch 1: rows matching EXISTS (via semi-join / anti-join) - let (exists_op, exists_cols) = - self.plan_exists_as_semi_join(input, subquery, is_negated, None)?; - - // Branch 2: rows matching the scalar predicate (via normal filter) - let other_filter = super::FilterOp { - predicate: other_predicate.clone(), - input: Box::new(input.clone()), - pushdown_hint: None, - }; - let (other_op, _other_cols) = self.plan_filter(&other_filter)?; - - // Union both branches - let schema = self.derive_schema_from_columns(&exists_cols); - let union_op = UnionOperator::new(vec![exists_op, other_op], schema.clone()); - - // Deduplicate (OR semantics: each row appears at most once) - let distinct_op = DistinctOperator::new(Box::new(union_op), schema); - - Ok((Box::new(distinct_op), exists_cols)) - } - /// Plans a complex EXISTS/NOT EXISTS as a hash-based semi-join or anti-join. /// /// The inner subquery is planned as a full operator tree via `plan_operator()`. @@ -359,17 +395,7 @@ impl super::Planner { is_negated: bool, remaining_predicate: Option, ) -> Result<(Box, Vec)> { - // Detect correlated subquery (contains ParameterScan from translator) - if let Some(param_vars) = Self::extract_parameter_scan_vars(subquery) { - return self.plan_correlated_exists( - outer_input, - subquery, - ¶m_vars, - is_negated, - remaining_predicate, - ); - } - + let input_writes = outer_input.has_mutations(); let (left_op, left_columns) = self.plan_operator(outer_input)?; let (right_op, right_columns) = self.plan_operator(subquery)?; @@ -407,8 +433,9 @@ impl super::Planner { // contains more EXISTS subqueries that need semi-join rewriting. if let Some(ref remaining) = remaining_predicate { // Recursively handle nested EXISTS in the remaining predicate + let outer: HashSet = output_columns.iter().cloned().collect(); if let Some((nested_sub, nested_neg, nested_rest)) = - self.extract_complex_exists(remaining) + self.extract_complex_exists(remaining, Some(&outer)) { return self.plan_exists_as_semi_join_with_input( join_op, @@ -416,118 +443,16 @@ impl super::Planner { nested_sub, nested_neg, nested_rest, + input_writes, ); } - let variable_columns: HashMap = output_columns - .iter() - .enumerate() - .map(|(i, name)| (name.clone(), i)) - .collect(); - let filter_expr = self.convert_expression(remaining)?; - let predicate = ExpressionPredicate::new( - filter_expr, - variable_columns, - Arc::clone(&self.store) as Arc, - ) - .with_transaction_context(self.viewing_epoch, self.transaction_id) - .with_session_context(self.session_context.clone()); - let filter_op = Box::new(FilterOperator::new(join_op, Box::new(predicate))); - return Ok((filter_op, output_columns)); + return self.filter_rest(join_op, output_columns, remaining, input_writes); } Ok((join_op, output_columns)) } - /// Extracts `ParameterScan` variable names from a logical plan, if present. - /// - /// Returns `Some(vars)` when the plan contains a `ParameterScan` node - /// (indicating a correlated subquery from the translator). - fn extract_parameter_scan_vars(plan: &LogicalOperator) -> Option> { - match plan { - LogicalOperator::ParameterScan(ps) => Some(ps.columns.clone()), - LogicalOperator::Filter(f) => Self::extract_parameter_scan_vars(&f.input), - LogicalOperator::Join(j) => Self::extract_parameter_scan_vars(&j.left) - .or_else(|| Self::extract_parameter_scan_vars(&j.right)), - LogicalOperator::NodeScan(s) => s - .input - .as_ref() - .and_then(|i| Self::extract_parameter_scan_vars(i)), - LogicalOperator::Expand(e) => Self::extract_parameter_scan_vars(&e.input), - _ => None, - } - } - - /// Plans a correlated EXISTS/NOT EXISTS using `ApplyOperator` with EXISTS mode. - /// - /// The inner subquery contains a `ParameterScan` for outer variable references. - /// For each outer row, the inner subquery is executed with the outer values - /// injected via `ParameterState`. Semi-join keeps rows where inner has results, - /// anti-join keeps rows where inner has no results. - fn plan_correlated_exists( - &self, - outer_input: &LogicalOperator, - subquery: &LogicalOperator, - param_vars: &[String], - is_negated: bool, - remaining_predicate: Option, - ) -> Result<(Box, Vec)> { - let (left_op, left_columns) = self.plan_operator(outer_input)?; - - // Set up ParameterState for correlated variables - let param_state = std::sync::Arc::new( - grafeo_core::execution::operators::ParameterState::new(param_vars.to_vec()), - ); - let param_col_indices: Vec = param_vars - .iter() - .map(|var| left_columns.iter().position(|c| c == var).unwrap_or(0)) - .collect(); - - // Plan inner subquery with correlated context - *self.correlated_param_state.borrow_mut() = Some(std::sync::Arc::clone(¶m_state)); - let (inner_op, _inner_columns) = self.plan_operator(subquery)?; - *self.correlated_param_state.borrow_mut() = None; - - // Create Apply with EXISTS mode (semi or anti join) - let op = ApplyOperator::new_correlated(left_op, inner_op, param_state, param_col_indices) - .with_exists_mode(!is_negated); - - let output_columns = left_columns; - let mut result: Box = Box::new(op); - - // Handle remaining predicate (from AND splitting) - if let Some(ref remaining) = remaining_predicate { - if let Some((nested_sub, nested_neg, nested_rest)) = - self.extract_complex_exists(remaining) - { - return self.plan_exists_as_semi_join_with_input( - result, - &output_columns, - nested_sub, - nested_neg, - nested_rest, - ); - } - - let variable_columns: HashMap = output_columns - .iter() - .enumerate() - .map(|(i, name)| (name.clone(), i)) - .collect(); - let filter_expr = self.convert_expression(remaining)?; - let predicate = ExpressionPredicate::new( - filter_expr, - variable_columns, - Arc::clone(&self.store) as Arc, - ) - .with_transaction_context(self.viewing_epoch, self.transaction_id) - .with_session_context(self.session_context.clone()); - result = Box::new(FilterOperator::new(result, Box::new(predicate))); - } - - Ok((result, output_columns)) - } - /// Plans an EXISTS/NOT EXISTS semi-join with an already-planned outer input. /// /// Used when the remaining predicate from a prior semi-join contains additional @@ -539,6 +464,7 @@ impl super::Planner { subquery: &LogicalOperator, is_negated: bool, remaining_predicate: Option, + input_writes: bool, ) -> Result<(Box, Vec)> { let (right_op, right_columns) = self.plan_operator(subquery)?; @@ -571,8 +497,9 @@ impl super::Planner { // Recursively handle any further EXISTS in the remaining predicate if let Some(ref remaining) = remaining_predicate { + let outer: HashSet = output_columns.iter().cloned().collect(); if let Some((nested_sub, nested_neg, nested_rest)) = - self.extract_complex_exists(remaining) + self.extract_complex_exists(remaining, Some(&outer)) { return self.plan_exists_as_semi_join_with_input( join_op, @@ -580,202 +507,16 @@ impl super::Planner { nested_sub, nested_neg, nested_rest, + input_writes, ); } - let variable_columns: HashMap = output_columns - .iter() - .enumerate() - .map(|(i, name)| (name.clone(), i)) - .collect(); - let filter_expr = self.convert_expression(remaining)?; - let predicate = ExpressionPredicate::new( - filter_expr, - variable_columns, - Arc::clone(&self.store) as Arc, - ) - .with_transaction_context(self.viewing_epoch, self.transaction_id) - .with_session_context(self.session_context.clone()); - let filter_op = Box::new(FilterOperator::new(join_op, Box::new(predicate))); - return Ok((filter_op, output_columns)); + return self.filter_rest(join_op, output_columns, remaining, input_writes); } Ok((join_op, output_columns)) } - /// Extracts a COUNT subquery comparison from a filter predicate. - /// - /// Recognizes patterns like: - /// - `COUNT { MATCH ... } > 5` - /// - `COUNT { MATCH ... } = 0` - /// - `5 < COUNT { MATCH ... }` (reversed operands) - /// - /// Returns `(subquery, comparison_op, threshold_value, remaining_predicate)`. - fn extract_count_comparison( - predicate: &LogicalExpression, - ) -> Option<( - &LogicalOperator, - BinaryOp, - &LogicalExpression, - Option<&LogicalExpression>, - )> { - match predicate { - LogicalExpression::Binary { left, op, right } => { - // Check for AND-combined: extract COUNT comparison from either side - if *op == BinaryOp::And { - if let Some(result) = Self::extract_count_from_binary(left) { - return Some((result.0, result.1, result.2, Some(right))); - } - if let Some(result) = Self::extract_count_from_binary(right) { - return Some((result.0, result.1, result.2, Some(left))); - } - return None; - } - - // Direct comparison: COUNT { ... } op value - Self::extract_count_from_binary(predicate) - .map(|(sub, op, threshold)| (sub, op, threshold, None)) - } - _ => None, - } - } - - /// Helper: extracts COUNT subquery comparison from a binary expression. - fn extract_count_from_binary( - expr: &LogicalExpression, - ) -> Option<(&LogicalOperator, BinaryOp, &LogicalExpression)> { - if let LogicalExpression::Binary { left, op, right } = expr { - match op { - BinaryOp::Eq - | BinaryOp::Ne - | BinaryOp::Gt - | BinaryOp::Ge - | BinaryOp::Lt - | BinaryOp::Le => { - // COUNT { ... } op literal - if let LogicalExpression::CountSubquery(subplan) = left.as_ref() { - return Some((subplan.as_ref(), *op, right.as_ref())); - } - // literal op COUNT { ... } (flip the operator) - if let LogicalExpression::CountSubquery(subplan) = right.as_ref() { - let flipped = match op { - BinaryOp::Gt => BinaryOp::Lt, - BinaryOp::Ge => BinaryOp::Le, - BinaryOp::Lt => BinaryOp::Gt, - BinaryOp::Le => BinaryOp::Ge, - other => *other, // Eq/Ne are symmetric - }; - return Some((subplan.as_ref(), flipped, left.as_ref())); - } - } - _ => {} - } - } - None - } - - /// Plans a COUNT subquery comparison as Join + Aggregate + Filter. - /// - /// Rewrites `WHERE COUNT { MATCH pattern } > N` into: - /// 1. Inner join on shared variables to get all matches per outer row - /// 2. Aggregate(COUNT) grouped by outer columns - /// 3. Filter(count > N) on the aggregated result - fn plan_count_as_apply( - &self, - outer_input: &LogicalOperator, - subquery: &LogicalOperator, - op: BinaryOp, - threshold: &LogicalExpression, - remaining_predicate: Option<&LogicalExpression>, - ) -> Result<(Box, Vec)> { - let (left_op, left_columns) = self.plan_operator(outer_input)?; - let (right_op, right_columns) = self.plan_operator(subquery)?; - - let output_columns = left_columns.clone(); - - // Find shared variables for equi-join keys - let mut probe_keys = Vec::new(); - let mut build_keys = Vec::new(); - for (right_idx, right_col) in right_columns.iter().enumerate() { - if let Some(left_idx) = left_columns.iter().position(|c| c == right_col) { - probe_keys.push(left_idx); - build_keys.push(right_idx); - } - } - - // Left join to preserve outer rows with no matches (needed for COUNT = 0). - // The join physically outputs both left and right columns. - let mut join_columns = left_columns.clone(); - join_columns.extend(right_columns.iter().cloned()); - let join_schema = self.derive_schema_from_columns(&join_columns); - let join_op: Box = Box::new(HashJoinOperator::new( - left_op, - right_op, - probe_keys, - build_keys, - PhysicalJoinType::Left, - join_schema, - )); - - // Aggregate: COUNT(right_col) grouped by all outer columns. - // Using COUNT on a right-side column so nulls (no match) produce 0 instead of 1. - let count_alias = "_count_subquery_".to_string(); - let mut agg_columns = output_columns.clone(); - agg_columns.push(count_alias.clone()); - - let group_keys: Vec = (0..output_columns.len()).collect(); - // Pick the first right-side column for COUNT (index = left_columns.len()) - let right_col_idx = left_columns.len(); - let agg_exprs = vec![PhysicalAggregateExpr::count(right_col_idx)]; - let agg_schema = self.derive_schema_from_columns(&agg_columns); - - let agg_op: Box = Box::new(HashAggregateOperator::new( - join_op, group_keys, agg_exprs, agg_schema, - )); - - // Filter: _count_ op threshold - let threshold_expr = convert_filter_expression(threshold)?; - let count_var_columns: HashMap = agg_columns - .iter() - .enumerate() - .map(|(i, name)| (name.clone(), i)) - .collect(); - let filter_op_code = convert_binary_op(op)?; - let count_filter = FilterExpression::Binary { - left: Box::new(FilterExpression::Variable(count_alias)), - op: filter_op_code, - right: Box::new(threshold_expr), - }; - - let predicate = ExpressionPredicate::new( - count_filter, - count_var_columns.clone(), - Arc::clone(&self.store) as Arc, - ) - .with_transaction_context(self.viewing_epoch, self.transaction_id) - .with_session_context(self.session_context.clone()); - let mut result_op: Box = - Box::new(FilterOperator::new(agg_op, Box::new(predicate))); - - // If there's a remaining predicate, apply it too - if let Some(remaining) = remaining_predicate { - let remaining_expr = self.convert_expression(remaining)?; - let remaining_predicate = ExpressionPredicate::new( - remaining_expr, - count_var_columns, - Arc::clone(&self.store) as Arc, - ) - .with_transaction_context(self.viewing_epoch, self.transaction_id) - .with_session_context(self.session_context.clone()); - result_op = Box::new(FilterOperator::new( - result_op, - Box::new(remaining_predicate), - )); - } - - Ok((result_op, output_columns)) - } - /// Checks zone maps for a predicate to see if we can skip the scan entirely. /// /// A property comparison is checked against the node zone map only when its @@ -929,15 +670,8 @@ impl super::Planner { .iter() .map(|(p, v)| (p.as_str(), v.clone())) .collect(); - let mut nodes = self.store.find_nodes_by_properties(&conditions_ref); - - // Intersect with label if present - if let Some(label) = &scan_label { - let label_nodes: std::collections::HashSet<_> = - self.store.nodes_by_label(label).into_iter().collect(); - nodes.retain(|n| label_nodes.contains(n)); - } - nodes + // The scan's label is checked on each found node below. + self.store.find_nodes_by_properties(&conditions_ref) } else { // No index but we have a label: scan label first, then check properties. // This is more efficient than ScanOperator → DataChunk → FilterOperator @@ -965,14 +699,10 @@ impl super::Planner { .collect() }; - // MVCC visibility: filter out nodes not visible at the current epoch/tx. - // Without this, rolled-back or uncommitted nodes could leak through. - let epoch = self.viewing_epoch; - if let Some(tx) = self.transaction_id { - matching_nodes.retain(|id| self.store.get_node_versioned(*id, epoch, tx).is_some()); - } else { - matching_nodes.retain(|id| self.store.get_node_at_epoch(*id, epoch).is_some()); - } + // MVCC visibility: filter out nodes not visible at the current epoch/tx + // (rolled-back or uncommitted nodes could leak through otherwise), and + // nodes without the scan's label. + matching_nodes.retain(|&id| self.visible_with_label(id, scan_label.as_deref())); let columns = vec![scan_variable.clone()]; let node_list_op: Box = Box::new(NodeListOperator::new(matching_nodes, 2048)); @@ -1084,20 +814,8 @@ impl super::Planner { } } - // Intersect with the label constraint, if any. - if let Some(label) = &scan_label { - let label_nodes: GrafeoSet<_> = self.store.nodes_by_label(label).into_iter().collect(); - matching_nodes.retain(|n| label_nodes.contains(n)); - } - - // MVCC visibility: drop nodes that aren't visible at the current - // epoch/tx, matching the equality fast path's semantics. - let epoch = self.viewing_epoch; - if let Some(tx) = self.transaction_id { - matching_nodes.retain(|id| self.store.get_node_versioned(*id, epoch, tx).is_some()); - } else { - matching_nodes.retain(|id| self.store.get_node_at_epoch(*id, epoch).is_some()); - } + // MVCC visibility and the label constraint, as in the equality fast path. + matching_nodes.retain(|&id| self.visible_with_label(id, scan_label.as_deref())); // Absorbed-scan PROFILE entry: see `record_absorbed_scan_entry`. self.record_absorbed_scan_entry("NodeScan", &filter.input); @@ -1823,7 +1541,8 @@ fn binding_kind(op: &LogicalOperator, variable: &str) -> Option { } if expand.edge_variable.as_deref() == Some(variable) { // A variable-length expand binds a list of edges. - let single_hop = expand.min_hops == 1 && expand.max_hops == Some(1); + let single_hop = + !expand.quantified && expand.min_hops == 1 && expand.max_hops == Some(1); return single_hop.then_some(BindingKind::Edge); } if expand.path_alias.as_deref() == Some(variable) { @@ -1845,3 +1564,29 @@ fn binding_kind(op: &LogicalOperator, variable: &str) -> Option { _ => None, } } + +/// Adds the parts of an `AND` (the predicate itself when it is not one). +fn split_conjuncts(predicate: &LogicalExpression, out: &mut Vec) { + match predicate { + LogicalExpression::Binary { + left, + op: BinaryOp::And, + right, + } => { + split_conjuncts(left, out); + split_conjuncts(right, out); + } + other => out.push(other.clone()), + } +} + +/// The `AND` of `conjuncts`, or `None` when there are none. +fn join_conjuncts(conjuncts: Vec) -> Option { + conjuncts + .into_iter() + .reduce(|left, right| LogicalExpression::Binary { + left: Box::new(left), + op: BinaryOp::And, + right: Box::new(right), + }) +} diff --git a/crates/grafeo-engine/src/query/planner/lpg/filter_hybrid.rs b/crates/grafeo-engine/src/query/planner/lpg/filter_hybrid.rs index ff871132f..32d741b44 100644 --- a/crates/grafeo-engine/src/query/planner/lpg/filter_hybrid.rs +++ b/crates/grafeo-engine/src/query/planner/lpg/filter_hybrid.rs @@ -6,12 +6,12 @@ #[cfg(feature = "text-index")] use super::{ - Arc, BinaryOp, ExpressionPredicate, FilterOp, FilterOperator, GraphStoreSearch, HashMap, + Arc, BinaryOp, ExpressionPredicate, FilterOperator, GraphStoreSearch, HashMap, LogicalExpression, LogicalOperator, Operator, Result, Value, }; #[cfg(all(feature = "vector-index", feature = "text-index"))] -use super::{HashJoinOperator, PhysicalJoinType}; +use super::{FilterOp, HashJoinOperator, PhysicalJoinType}; // ============================================================================ // Text predicate extraction and pushdown diff --git a/crates/grafeo-engine/src/query/planner/lpg/join.rs b/crates/grafeo-engine/src/query/planner/lpg/join.rs index 504d664bf..d1069aeac 100644 --- a/crates/grafeo-engine/src/query/planner/lpg/join.rs +++ b/crates/grafeo-engine/src/query/planner/lpg/join.rs @@ -2,9 +2,14 @@ use super::{ ApplyOp, ApplyOperator, DistinctOp, Error, ExceptOp, HashJoinOperator, IntersectOp, JoinOp, - JoinType, LeapfrogJoinOperator, LogicalExpression, MultiWayJoinOp, Operator, OtherwiseOp, - PhysicalJoinType, ProjectExpr, ProjectOperator, Result, UnionOp, Value, common, + JoinType, LeapfrogJoinOperator, LogicalExpression, LogicalOperator, MultiWayJoinOp, Operator, + OtherwiseOp, ParameterScanOperator, PhysicalJoinType, ProjectExpr, ProjectOperator, Result, + UnionOp, Value, common, }; +use crate::query::plan::{ + LimitOp, ParameterScanOp, ProjectOp, Projection, ReturnOp, SkipOp, SortKey, SortOp, +}; +use grafeo_common::types::LogicalType; impl super::Planner { /// Plans a JOIN operator. @@ -210,10 +215,7 @@ impl super::Planner { /// same width. User-written Cypher/GQL UNIONNs are checked for matching /// columns in the translators, so padding only applies to internal unions. pub(super) fn plan_union(&self, union: &UnionOp) -> Result<(Box, Vec)> { - let mut planned = Vec::with_capacity(union.inputs.len()); - for input in &union.inputs { - planned.push(self.plan_operator(input)?); - } + let planned = self.plan_branches(&union.inputs)?; // Pad narrower branches with NULL up to the widest branch. let mut unified_columns: Vec = Vec::new(); @@ -264,12 +266,10 @@ impl super::Planner { distinct: &DistinctOp, ) -> Result<(Box, Vec)> { let (input_op, columns) = self.plan_operator(&distinct.input)?; - let schema = self.derive_schema_from_columns(&columns); Ok(common::build_distinct( input_op, columns, distinct.columns.as_deref(), - schema, )) } @@ -278,12 +278,10 @@ impl super::Planner { &self, except: &ExceptOp, ) -> Result<(Box, Vec)> { - let (left_op, columns) = self.plan_operator(&except.left)?; - let (right_op, _) = self.plan_operator(&except.right)?; - let schema = self.derive_schema_from_columns(&columns); - Ok(common::build_except( - left_op, right_op, columns, except.all, schema, - )) + let mut planned = self.plan_branches([except.left.as_ref(), except.right.as_ref()])?; + let (right_op, _) = planned.pop().expect("two branches planned"); + let (left_op, columns) = planned.pop().expect("two branches planned"); + Ok(common::build_except(left_op, right_op, columns, except.all)) } /// Plans an INTERSECT operator. @@ -291,15 +289,15 @@ impl super::Planner { &self, intersect: &IntersectOp, ) -> Result<(Box, Vec)> { - let (left_op, columns) = self.plan_operator(&intersect.left)?; - let (right_op, _) = self.plan_operator(&intersect.right)?; - let schema = self.derive_schema_from_columns(&columns); + let mut planned = + self.plan_branches([intersect.left.as_ref(), intersect.right.as_ref()])?; + let (right_op, _) = planned.pop().expect("two branches planned"); + let (left_op, columns) = planned.pop().expect("two branches planned"); Ok(common::build_intersect( left_op, right_op, columns, intersect.all, - schema, )) } @@ -308,25 +306,85 @@ impl super::Planner { &self, otherwise: &OtherwiseOp, ) -> Result<(Box, Vec)> { - let (left_op, columns) = self.plan_operator(&otherwise.left)?; - let (right_op, _) = self.plan_operator(&otherwise.right)?; + let mut planned = + self.plan_branches([otherwise.left.as_ref(), otherwise.right.as_ref()])?; + let (right_op, _) = planned.pop().expect("two branches planned"); + let (left_op, columns) = planned.pop().expect("two branches planned"); Ok(common::build_otherwise(left_op, right_op, columns)) } + /// Plans the scan that starts a correlated subquery from the outer row. + /// The state holds what the Apply imports (`*` expanded to the outer + /// columns in `plan_apply`); a scan that names some of them (a `UNION` + /// branch that imports less than another) passes on only those, so the + /// others stay free names in its plan. + pub(super) fn plan_parameter_scan( + &self, + scan: &ParameterScanOp, + ) -> Result<(Box, Vec)> { + let state = self + .correlated_param_state + .borrow() + .clone() + .ok_or_else(|| { + Error::Internal("ParameterScan without correlated Apply context".to_string()) + })?; + let columns = state.columns.clone(); + let operator: Box = Box::new(ParameterScanOperator::new(state)); + if scan.columns.iter().any(|name| name == "*") || scan.columns == columns { + return Ok((operator, columns)); + } + let projections = scan + .columns + .iter() + .map(|name| { + columns + .iter() + .position(|column| column == name) + .map(ProjectExpr::Column) + .ok_or_else(|| { + Error::Internal(format!( + "variable '{name}' is not imported by the enclosing Apply" + )) + }) + }) + .collect::>>()?; + let types = vec![LogicalType::Any; projections.len()]; + Ok(( + Box::new(ProjectOperator::new(operator, projections, types)), + scan.columns.clone(), + )) + } + /// Plans an APPLY (lateral join) operator. /// /// When `shared_variables` is non-empty, creates a correlated Apply that /// injects outer row values into the inner plan via [`ParameterState`]. pub(super) fn plan_apply(&self, apply: &ApplyOp) -> Result<(Box, Vec)> { - let (outer_op, outer_columns) = self.plan_operator(&apply.input)?; + // A subquery that comes first runs once, on one empty row. + let (outer_op, outer_columns): (Box, Vec) = + if matches!(apply.input.as_ref(), LogicalOperator::Empty) { + ( + Box::new( + grafeo_core::execution::operators::single_row::SingleRowOperator::new(), + ), + Vec::new(), + ) + } else { + self.plan_operator(&apply.input)? + }; + let output = subquery_output(&apply.subplan); + let subplan = output.as_ref().unwrap_or(&apply.subplan); if apply.shared_variables.is_empty() { // Uncorrelated Apply - let (inner_op, inner_columns) = self.plan_operator(&apply.subplan)?; - // Inner subquery RETURN materializes values (PropertyAccess, NodeResolve, - // aggregates, etc.), so all its output columns are scalar. - for col in &inner_columns { - self.scalar_columns.borrow_mut().insert(col.clone()); + let (inner_op, inner_columns) = self.plan_operator(subplan)?; + // Any other subquery materializes values (PropertyAccess, + // NodeResolve, aggregates, etc.), so its output columns are scalar. + if output.is_none() { + for col in &inner_columns { + self.scalar_columns.borrow_mut().insert(col.clone()); + } } let inner_col_count = inner_columns.len(); let mut columns = outer_columns; @@ -350,24 +408,35 @@ impl super::Planner { grafeo_core::execution::operators::ParameterState::new(shared_vars.clone()), ); - // Find column indices for the shared variables in outer columns + // Find column indices for the shared variables in outer columns (the + // binder has checked that the outer query has them) let param_col_indices: Vec = shared_vars .iter() - .map(|var| outer_columns.iter().position(|c| c == var).unwrap_or(0)) - .collect(); - - // Set the parameter state so the inner plan's ParameterScan can find it - *self.correlated_param_state.borrow_mut() = Some(std::sync::Arc::clone(¶m_state)); - - let (inner_op, inner_columns) = self.plan_operator(&apply.subplan)?; - - // Clear the parameter state after planning the inner operator - *self.correlated_param_state.borrow_mut() = None; - - // Inner subquery RETURN materializes values, so register as scalar + .map(|var| { + outer_columns.iter().position(|c| c == var).ok_or_else(|| { + Error::Internal(format!( + "variable '{var}' imported into CALL is not a column of the outer query" + )) + }) + }) + .collect::>()?; + + // Set the parameter state so the inner plan's ParameterScan can find + // it; the state of an enclosing subquery comes back afterwards, for + // what is planned after this Apply inside that subquery. + let previous = self + .correlated_param_state + .replace(Some(std::sync::Arc::clone(¶m_state))); + let planned = self.plan_operator(subplan); + *self.correlated_param_state.borrow_mut() = previous; + let (inner_op, inner_columns) = planned?; + + // Any other subquery materializes values, so register them as scalar // to prevent the outer RETURN from misinterpreting them as node IDs. - for col in &inner_columns { - self.scalar_columns.borrow_mut().insert(col.clone()); + if output.is_none() { + for col in &inner_columns { + self.scalar_columns.borrow_mut().insert(col.clone()); + } } // Build correlated Apply @@ -382,3 +451,97 @@ impl super::Planner { Ok((Box::new(op), columns)) } } + +/// The plan of a subquery whose `RETURN` passes its values on as `WITH` does: +/// nodes and edges stay references, so the outer query can match from them, +/// compare them and take their ids, and resolves them when it returns them. +/// An `ORDER BY`, `SKIP`, `LIMIT` or `DISTINCT` after the `RETURN` stays on +/// top. `None` when the subquery does not end in such a `RETURN`, or when its +/// `ORDER BY` reads a variable the `RETURN` leaves out (only the `RETURN`'s +/// own planning keeps those for the sort). +fn subquery_output(subplan: &LogicalOperator) -> Option { + match subplan { + LogicalOperator::Return(ret) => return_as_projection(ret), + LogicalOperator::Sort(sort) => { + if let LogicalOperator::Return(ret) = sort.input.as_ref() + && !sort_reads_only_returned(&sort.keys, ret) + { + return None; + } + Some(LogicalOperator::Sort(SortOp { + keys: sort.keys.clone(), + input: Box::new(subquery_output(&sort.input)?), + })) + } + LogicalOperator::Limit(limit) => Some(LogicalOperator::Limit(LimitOp { + count: limit.count.clone(), + input: Box::new(subquery_output(&limit.input)?), + })), + LogicalOperator::Skip(skip) => Some(LogicalOperator::Skip(SkipOp { + count: skip.count.clone(), + input: Box::new(subquery_output(&skip.input)?), + })), + LogicalOperator::Distinct(distinct) => Some(LogicalOperator::Distinct(DistinctOp { + input: Box::new(subquery_output(&distinct.input)?), + columns: distinct.columns.clone(), + })), + // A UNION of such subqueries passes on what each branch returns. + LogicalOperator::Union(union) => Some(LogicalOperator::Union(UnionOp { + inputs: union + .inputs + .iter() + .map(subquery_output) + .collect::>()?, + })), + _ => None, + } +} + +/// Whether every variable the sort `keys` read is a column `ret` returns. +fn sort_reads_only_returned(keys: &[SortKey], ret: &ReturnOp) -> bool { + let returned: Vec<&str> = ret + .items + .iter() + .filter_map(|item| match (&item.alias, &item.expression) { + (Some(alias), _) => Some(alias.as_str()), + (None, LogicalExpression::Variable(name)) => Some(name.as_str()), + _ => None, + }) + .collect(); + let mut read = Vec::new(); + for key in keys { + super::project::collect_vars(&key.expression, &mut read); + } + read.iter().all(|name| returned.contains(&name.as_str())) +} + +/// A `RETURN` as the projection that passes its values on. +fn return_as_projection(ret: &ReturnOp) -> Option { + if ret + .items + .iter() + .any(|item| matches!(&item.expression, LogicalExpression::Variable(name) if name == "*")) + { + return None; + } + let project = LogicalOperator::Project(ProjectOp { + projections: ret + .items + .iter() + .map(|item| Projection { + expression: item.expression.clone(), + alias: item.alias.clone(), + }) + .collect(), + input: ret.input.clone(), + pass_through_input: false, + }); + Some(if ret.distinct { + LogicalOperator::Distinct(DistinctOp { + input: Box::new(project), + columns: None, + }) + } else { + project + }) +} diff --git a/crates/grafeo-engine/src/query/planner/lpg/mod.rs b/crates/grafeo-engine/src/query/planner/lpg/mod.rs index fac8d907d..a5bd3d2d3 100644 --- a/crates/grafeo-engine/src/query/planner/lpg/mod.rs +++ b/crates/grafeo-engine/src/query/planner/lpg/mod.rs @@ -43,7 +43,9 @@ //! per outer row; the join form piggy-backs on the regular hash-join //! infrastructure. The fast path in `expression::convert_expression` //! keeps trivial single-hop EXISTS as inline predicates so small -//! queries stay scan-local. +//! queries stay scan-local; in WHERE only when the pattern starts from a +//! node of the row and shares no other variable with it, because the +//! join form matches every shared variable. //! //! 3. **EXISTS inside OR** (`filter::extract_exists_from_or`): //! semi-joins filter rows and therefore compose incorrectly with the @@ -101,6 +103,7 @@ mod mutation; mod project; mod scan; pub(crate) mod seek; +mod subquery; #[cfg(feature = "algos")] use crate::query::plan::CallProcedureOp; @@ -124,17 +127,17 @@ use grafeo_common::utils::error::{Error, Result}; use grafeo_core::execution::AdaptiveContext; use grafeo_core::execution::operators::{ AddLabelOperator, AggregateExpr as PhysicalAggregateExpr, ApplyOperator, ConstraintValidator, - CreateEdgeOperator, CreateNodeOperator, DeleteEdgeOperator, DeleteNodeOperator, - DistinctOperator, EmptyOperator, EntityKind, ExecutionPathMode, ExpandOperator, ExpandStep, - ExpressionPredicate, FactorizedAggregate, FactorizedAggregateOperator, FilterExpression, - FilterOperator, HashAggregateOperator, HashJoinOperator, HorizontalAggregateOperator, + CreateEdgeOperator, CreateNodeOperator, DeleteEdgeOperator, DeleteNodeOperator, EmptyOperator, + EntityKind, EntityValue, ExecutionPathMode, ExpandOperator, ExpandStep, ExpressionPredicate, + FactorizedAggregate, FactorizedAggregateOperator, FilterExpression, FilterOperator, + HashAggregateOperator, HashJoinOperator, HorizontalAggregateOperator, JoinType as PhysicalJoinType, LazyFactorizedChainOperator, LeapfrogJoinOperator, LoadDataOperator, MapCollectOperator, MergeConfig, MergeOperator, MergeRelationshipConfig, - MergeRelationshipOperator, NestedLoopJoinOperator, NodeListOperator, NullOrder, Operator, + MergeRelationshipOperator, NestedLoopJoinOperator, NodeListOperator, Operator, ParameterScanOperator, ProjectExpr, ProjectOperator, PropertySource, RangeScanOperator, RemoveLabelOperator, ScanOperator, SetPropertyOperator, ShortestPathOperator, SimpleAggregateOperator, SortDirection, SortKey as PhysicalSortKey, SortOperator, - UnionOperator, UnwindOperator, VariableLengthExpandOperator, + UnwindOperator, VariableLengthExpandOperator, }; use grafeo_core::graph::{Direction, GraphStoreMut, GraphStoreSearch}; use std::collections::HashMap; @@ -145,8 +148,8 @@ use crate::query::planner::common::{ expression_to_string, output_column_name, resolved_column_name, }; use crate::query::planner::{ - PhysicalPlan, convert_aggregate_function, convert_binary_op, convert_filter_expression, - convert_unary_op, value_to_logical_type, + PhysicalPlan, convert_aggregate_function, convert_binary_op, convert_unary_op, + value_to_logical_type, }; use crate::transaction::TransactionManager; @@ -158,6 +161,35 @@ struct RangeBounds<'a> { max_inclusive: bool, } +/// How the planned query's rows reach the caller. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Delivery { + /// As one result, read in full before it is returned. + Collected, + /// As a stream, chunk by chunk, with bounded memory. + Streamed, +} + +/// What the planner knows about named columns: which hold scalar values or +/// records, edge IDs, lists of nodes or edges, or group lists. +#[derive(Clone, Default)] +struct ColumnKinds { + scalar: std::collections::HashSet, + edge: std::collections::HashSet, + entity_list: std::collections::HashMap, + group_list: std::collections::HashSet, +} + +impl ColumnKinds { + /// Adds what `other` knows. + fn add(&mut self, other: Self) { + self.scalar.extend(other.scalar); + self.edge.extend(other.edge); + self.entity_list.extend(other.entity_list); + self.group_list.extend(other.group_list); + } +} + /// Converts a logical plan to a physical operator tree for LPG stores. pub struct Planner { /// The graph store (read-only operations). @@ -172,10 +204,15 @@ pub struct Planner { pub(super) viewing_epoch: EpochId, /// Counter for generating unique anonymous edge column names. pub(super) anon_edge_counter: std::cell::Cell, + /// Counter for the column names of subqueries planned per row. + pub(super) subquery_counter: std::cell::Cell, /// Whether to use factorized execution for multi-hop queries. pub(super) factorized_execution: bool, /// Whether a plan without `ORDER BY` returns its rows in random order. - pub(super) shuffle_unordered: bool, + shuffle_unordered: bool, + /// Whether the rows are collected or streamed (a stream is shuffled per + /// chunk). + delivery: Delivery, /// Variables that hold scalar values (from UNWIND/FOR), not node/edge IDs. /// Used by plan_return to assign `LogicalType::Any` instead of `Node`. pub(super) scalar_columns: std::cell::RefCell>, @@ -246,8 +283,10 @@ impl Planner { transaction_id: None, viewing_epoch: epoch, anon_edge_counter: std::cell::Cell::new(0), + subquery_counter: std::cell::Cell::new(0), factorized_execution: true, shuffle_unordered: false, + delivery: Delivery::Collected, scalar_columns: std::cell::RefCell::new(std::collections::HashSet::new()), edge_columns: std::cell::RefCell::new(std::collections::HashSet::new()), entity_list_columns: std::cell::RefCell::new(std::collections::HashMap::new()), @@ -312,8 +351,10 @@ impl Planner { transaction_id, viewing_epoch, anon_edge_counter: std::cell::Cell::new(0), + subquery_counter: std::cell::Cell::new(0), factorized_execution: true, shuffle_unordered: false, + delivery: Delivery::Collected, scalar_columns: std::cell::RefCell::new(std::collections::HashSet::new()), edge_columns: std::cell::RefCell::new(std::collections::HashSet::new()), entity_list_columns: std::cell::RefCell::new(std::collections::HashMap::new()), @@ -372,6 +413,18 @@ impl Planner { Arc::clone(&self.write_counter) } + /// Counts the writes into `counter` instead of a counter of its own, so + /// the writes of a stored procedure's body count for the statement that + /// calls it. + #[must_use] + pub fn with_write_counter( + mut self, + counter: Arc, + ) -> Self { + self.write_counter = counter; + self + } + /// Returns the viewing epoch for this planner. #[must_use] pub fn viewing_epoch(&self) -> EpochId { @@ -405,22 +458,30 @@ impl Planner { self } + /// Plans for a stream: with `shuffle_unordered`, the rows of each chunk + /// are shuffled on their own instead of the whole result, which would + /// have to be read before the first row. + #[must_use] + pub fn for_streaming(mut self) -> Self { + self.delivery = Delivery::Streamed; + self + } + /// The root operator, behind a shuffle when the option is on and the /// plan does not order its rows. fn shuffled_root( &self, logical_plan: &LogicalPlan, operator: Box, - columns: &[String], ) -> Box { - if self.shuffle_unordered && !super::common::orders_rows(&logical_plan.root) { - let schema = self.derive_schema_from_columns(columns); - Box::new(grafeo_core::execution::operators::ShuffleOperator::new( - operator, schema, - )) - } else { - operator + use grafeo_core::execution::operators::ShuffleOperator; + if !self.shuffle_unordered || super::common::orders_rows(&logical_plan.root) { + return operator; } + Box::new(match self.delivery { + Delivery::Streamed => ShuffleOperator::per_chunk(operator), + Delivery::Collected => ShuffleOperator::new(operator), + }) } /// Sets the constraint validator for schema enforcement during mutations. @@ -456,6 +517,91 @@ impl Planner { self } + /// What the planner knows about named columns. + fn column_kinds(&self) -> ColumnKinds { + ColumnKinds { + scalar: self.scalar_columns.borrow().clone(), + edge: self.edge_columns.borrow().clone(), + entity_list: self.entity_list_columns.borrow().clone(), + group_list: self.group_list_variables.borrow().clone(), + } + } + + /// What the named column holds when it holds nodes or edges, classified + /// the way RETURN classifies it: a node or edge list as registered, nothing + /// for a scalar, a path detail or a group list, an edge for an edge column + /// and otherwise a node. + pub(super) fn column_entity(&self, name: &str) -> Option { + if let Some(kind) = self.entity_list_columns.borrow().get(name).copied() { + return Some(kind); + } + if name.starts_with("_path_") + || self.scalar_columns.borrow().contains(name) + || self.group_list_variables.borrow().contains(name) + { + return None; + } + Some(if self.edge_columns.borrow().contains(name) { + EntityValue::Edge + } else { + EntityValue::Node + }) + } + + /// Records that the named column holds `kind`: a node (the default), an + /// edge, a node or edge list, or a scalar value (`None`). + pub(super) fn set_column_entity(&self, name: &str, kind: Option) { + let name = name.to_string(); + self.scalar_columns.borrow_mut().remove(&name); + self.edge_columns.borrow_mut().remove(&name); + self.entity_list_columns.borrow_mut().remove(&name); + match kind { + Some(EntityValue::Node) => {} + Some(EntityValue::Edge) => { + self.edge_columns.borrow_mut().insert(name); + } + Some(list @ (EntityValue::Nodes | EntityValue::Edges)) => { + self.entity_list_columns.borrow_mut().insert(name, list); + } + _ => { + self.scalar_columns.borrow_mut().insert(name); + } + } + } + + /// Replaces what the planner knows about named columns. + fn set_column_kinds(&self, kinds: ColumnKinds) { + *self.scalar_columns.borrow_mut() = kinds.scalar; + *self.edge_columns.borrow_mut() = kinds.edge; + *self.entity_list_columns.borrow_mut() = kinds.entity_list; + *self.group_list_variables.borrow_mut() = kinds.group_list; + } + + /// Plans the branches of a set operation (UNION, EXCEPT, INTERSECT, + /// OTHERWISE), each from what the planner knew before the first: a branch + /// is a query of its own, so what one binds or returns under a name says + /// nothing about that name in the next (an edge `x` in one branch and a + /// node `x` in the other, or a returned record and a bound ID). After the + /// last branch the planner knows what any branch added, as the operators + /// above the set operation read the columns of every branch. + pub(super) fn plan_branches<'a>( + &self, + branches: impl IntoIterator, + ) -> Result, Vec)>> { + let before = self.column_kinds(); + let mut after = ColumnKinds::default(); + let mut planned = Vec::new(); + for (index, branch) in branches.into_iter().enumerate() { + if index > 0 { + self.set_column_kinds(before.clone()); + } + planned.push(self.plan_operator(branch)?); + after.add(self.column_kinds()); + } + self.set_column_kinds(after); + Ok(planned) + } + /// Generates an edge column name from an expand's edge variable (or an /// anonymous fallback) and registers it in `edge_columns` so downstream /// RETURN emits `EdgeResolve` instead of `NodeResolve`. @@ -477,7 +623,8 @@ impl Planner { fn count_expand_chain(op: &LogicalOperator) -> (usize, &LogicalOperator) { match op { LogicalOperator::Expand(expand) => { - let is_single_hop = expand.min_hops == 1 && expand.max_hops == Some(1); + let is_single_hop = + !expand.quantified && expand.min_hops == 1 && expand.max_hops == Some(1); if is_single_hop { let (inner_count, base) = Self::count_expand_chain(&expand.input); @@ -498,7 +645,8 @@ impl Planner { let mut current = op; while let LogicalOperator::Expand(expand) = current { - let is_single_hop = expand.min_hops == 1 && expand.max_hops == Some(1); + let is_single_hop = + !expand.quantified && expand.min_hops == 1 && expand.max_hops == Some(1); if !is_single_hop { break; } @@ -519,7 +667,7 @@ impl Planner { pub fn plan(&self, logical_plan: &LogicalPlan) -> Result { let _span = grafeo_debug_span!("grafeo::query::plan"); let (operator, columns) = self.plan_operator(&logical_plan.root)?; - let operator = self.shuffled_root(logical_plan, operator, &columns); + let operator = self.shuffled_root(logical_plan, operator); Ok(PhysicalPlan { operator, columns, @@ -568,7 +716,7 @@ impl Planner { /// or invalid expressions. pub fn plan_adaptive(&self, logical_plan: &LogicalPlan) -> Result { let (operator, columns) = self.plan_operator(&logical_plan.root)?; - let operator = self.shuffled_root(logical_plan, operator, &columns); + let operator = self.shuffled_root(logical_plan, operator); let mut adaptive_context = AdaptiveContext::new(); self.collect_cardinality_estimates(&logical_plan.root, &mut adaptive_context, 0); @@ -852,22 +1000,7 @@ impl Planner { LogicalOperator::CallProcedure(_) => Err(Error::Internal( "CALL procedures require the 'algos' feature".to_string(), )), - LogicalOperator::ParameterScan(_param_scan) => { - let state = self - .correlated_param_state - .borrow() - .clone() - .ok_or_else(|| { - Error::Internal( - "ParameterScan without correlated Apply context".to_string(), - ) - })?; - // Use the actual column names from the ParameterState (which may - // have been expanded from "*" to real variable names in plan_apply) - let columns = state.columns.clone(); - let operator: Box = Box::new(ParameterScanOperator::new(state)); - Ok((operator, columns)) - } + LogicalOperator::ParameterScan(param_scan) => self.plan_parameter_scan(param_scan), LogicalOperator::MultiWayJoin(mwj) => self.plan_multi_way_join(mwj), LogicalOperator::HorizontalAggregate(ha) => self.plan_horizontal_aggregate(ha), LogicalOperator::LoadData(load) => { @@ -878,6 +1011,8 @@ impl Planner { load.field_terminator, load.variable.clone(), )); + // A loaded row is a value (a map or a list), not a node. + self.set_column_entity(&load.variable, None); Ok((operator, vec![load.variable.clone()])) } LogicalOperator::Empty => Err(Error::Internal("Empty plan".to_string())), @@ -1565,6 +1700,7 @@ mod tests { ], distinct: false, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -1611,6 +1747,7 @@ mod tests { ], distinct: false, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: Some("r".to_string()), @@ -1784,18 +1921,20 @@ mod tests { let physical = planner.plan(&logical).unwrap(); let root = physical.into_operator(); - // Walk down: Limit → Sort → RangeScan, asserting at each step. + // Walk down: Limit → Sort → Project → RangeScan, asserting at each step. let limit_op = root .into_any() .downcast::() .expect("top operator is LimitOperator"); let (after_limit, _cap) = limit_op.into_parts(); + // The Sort also drops the sort-key column (`n_name`) it needed. let sort_op = after_limit .into_any() .downcast::() .expect("operator under Limit must be Sort (Sort blocks pushdown)"); - let (after_sort, _keys) = sort_op.into_parts(); + assert_eq!(sort_op.output_width(), Some(1)); + let (after_sort, _keys, _width) = sort_op.into_parts(); // Sort wraps its input in a Project that materializes the sort // keys. The fold from Filter+NodeScan to RangeScan happens @@ -2555,6 +2694,7 @@ mod tests { ], distinct: false, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -2797,6 +2937,7 @@ mod tests { ], distinct: false, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -2838,6 +2979,7 @@ mod tests { ], distinct: false, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -2881,6 +3023,7 @@ mod tests { ], distinct: false, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -2923,6 +3066,7 @@ mod tests { ], distinct: false, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -2989,6 +3133,7 @@ mod tests { ], distinct: false, input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "b".to_string(), to_variable: "c".to_string(), edge_variable: None, @@ -2997,6 +3142,7 @@ mod tests { min_hops: 1, max_hops: Some(1), input: Box::new(LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -3108,7 +3254,7 @@ mod tests { MergeOp, MergeRelationshipOp, MultiWayJoinOp, OtherwiseOp, ParameterScanOp, RemoveLabelOp, SetPropertyOp, ShortestPathOp, TripleComponent, TripleScanOp, UnionOp, UnwindOp, }; - use grafeo_core::execution::operators::{Operator, SessionContext}; + use grafeo_core::execution::operators::SessionContext; fn full_store() -> Arc { // Richer store so expand and shortest path tests have real data. @@ -3523,6 +3669,7 @@ mod tests { // Register the edge column first via an outgoing expand, then DELETE r. let expand_op = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: Some("r".to_string()), @@ -3693,6 +3840,7 @@ mod tests { let store = full_store(); let planner = Planner::new(Arc::clone(&store) as Arc); let ab = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -3705,6 +3853,7 @@ mod tests { path_mode: PathMode::Walk, }); let bc = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "b".to_string(), to_variable: "c".to_string(), edge_variable: None, @@ -3717,6 +3866,7 @@ mod tests { path_mode: PathMode::Walk, }); let ca = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "c".to_string(), to_variable: "a".to_string(), edge_variable: None, @@ -3743,6 +3893,7 @@ mod tests { let planner = Planner::new(Arc::clone(&store) as Arc); let path = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: Some("r".to_string()), @@ -3815,6 +3966,7 @@ mod tests { fn test_count_expand_chain_variable_length_breaks_chain() { // A variable-length expand (not single-hop) should NOT count in the chain. let var_expand = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: "a".to_string(), to_variable: "b".to_string(), edge_variable: None, @@ -3836,6 +3988,7 @@ mod tests { #[test] fn test_static_result_operator_emits_rows_and_resets() { use grafeo_common::types::Value; + use grafeo_core::execution::operators::Operator; let rows = vec![ vec![Value::Int64(1), Value::String("Vincent".into())], vec![Value::Int64(2), Value::String("Jules".into())], diff --git a/crates/grafeo-engine/src/query/planner/lpg/mutation.rs b/crates/grafeo-engine/src/query/planner/lpg/mutation.rs index 2cccbc65b..067f70deb 100644 --- a/crates/grafeo-engine/src/query/planner/lpg/mutation.rs +++ b/crates/grafeo-engine/src/query/planner/lpg/mutation.rs @@ -3,7 +3,7 @@ use super::{ AddLabelOp, AddLabelOperator, AntiJoinOp, Arc, CreateEdgeOp, CreateEdgeOperator, CreateNodeOp, CreateNodeOperator, DeleteEdgeOp, DeleteEdgeOperator, DeleteNodeOp, DeleteNodeOperator, - Direction, Error, ExpandDirection, ExpressionPredicate, FilterOperator, HashMap, LeftJoinOp, + Direction, EntityValue, Error, ExpandDirection, ExpressionPredicate, HashMap, LeftJoinOp, LogicalExpression, LogicalOperator, LogicalType, MergeConfig, MergeOp, MergeOperator, MergeRelationshipConfig, MergeRelationshipOp, MergeRelationshipOperator, Operator, ProjectExpr, ProjectOperator, PropertySource, RemoveLabelOp, RemoveLabelOperator, Result, SetPropertyOp, @@ -13,6 +13,7 @@ use super::{ #[cfg(feature = "algos")] use super::{CallProcedureOp, StaticResultOperator}; use grafeo_common::utils::error::{QueryError, QueryErrorKind}; +use grafeo_core::execution::operators::{JoinCondition, JoinedRowCondition}; impl super::Planner { /// Plans a CREATE NODE operator. @@ -221,6 +222,32 @@ impl super::Planner { let (right_op, right_columns) = self.plan_operator(&left_join.right)?; let left_types = self.derive_schema_from_columns(&left_columns); let right_types = self.derive_schema_from_columns(&right_columns); + + // A condition that reads both sides (the WHERE of an OPTIONAL MATCH on a + // variable bound before it) decides which pairs are matches, so a left + // row none of whose pairs pass it keeps nulls. It reads the joined row: + // the left columns, then the right ones (a name both sides have reads + // the left one). + let residual = match &left_join.condition { + Some(condition) => { + let filter_expr = self.convert_expression(condition)?; + let mut variable_columns: HashMap = HashMap::new(); + for (i, name) in left_columns.iter().chain(&right_columns).enumerate() { + variable_columns.entry(name.clone()).or_insert(i); + } + let predicate = ExpressionPredicate::new( + filter_expr, + variable_columns, + Arc::clone(&self.store), + ) + .with_transaction_context(self.viewing_epoch, self.transaction_id) + .with_session_context(self.session_context.clone()); + let condition: Box = + Box::new(JoinedRowCondition::new(Box::new(predicate))); + Some(condition) + } + None => None, + }; let (join_op, join_columns, _join_types) = super::common::build_left_join( left_op, right_op, @@ -228,27 +255,9 @@ impl super::Planner { &right_columns, &left_types, &right_types, + residual, ); - // If the LeftJoin carries a cross-side condition (null-safe predicate), - // apply it as a Filter above the join. The condition already incorporates - // IS NULL guards so NULL-padded rows from unmatched optional sides pass through. - if let Some(condition) = &left_join.condition { - let filter_expr = self.convert_expression(condition)?; - let variable_columns: HashMap = join_columns - .iter() - .enumerate() - .map(|(i, name)| (name.clone(), i)) - .collect(); - let predicate = - ExpressionPredicate::new(filter_expr, variable_columns, Arc::clone(&self.store)) - .with_transaction_context(self.viewing_epoch, self.transaction_id) - .with_session_context(self.session_context.clone()); - let filter_op: Box = - Box::new(FilterOperator::new(join_op, Box::new(predicate))); - return Ok((filter_op, join_columns)); - } - Ok((join_op, join_columns)) } @@ -276,6 +285,7 @@ impl super::Planner { ) -> Result<(Box, Vec)> { // Plan the input operator first // Handle Empty specially - use a single-row operator + let unwinds_a_constant = matches!(&*unwind.input, LogicalOperator::Empty); let (input_op, input_columns): (Box, Vec) = if matches!(&*unwind.input, LogicalOperator::Empty) { // For UNWIND without prior MATCH, create a single-row input @@ -318,30 +328,21 @@ impl super::Planner { self.plan_operator(&unwind.input)? }; - // The UNWIND expression should be a list - we need to find/evaluate it - // Handle variable references, property access, and literal lists - - // Find if the expression references an existing column that is itself a list - let list_col_idx = match &unwind.expression { - LogicalExpression::Variable(var) => input_columns.iter().position(|c| c == var), - LogicalExpression::List(_) | LogicalExpression::Literal(_) => { - // Literal list expression - needs to be added as a column - None + // The list is a column of the input (the one row of a constant list, or + // a variable), or an expression evaluated per row in a column of its + // own: a literal, a property, `range(1, n.k)`, `nodes(p)`, ... + let list_col_idx = if unwinds_a_constant { + Some(0) + } else { + match &unwind.expression { + LogicalExpression::Variable(var) => input_columns.iter().position(|c| c == var), + _ => None, } - _ => None, }; - // When the expression needs runtime evaluation (property access, literal list, etc.), - // wrap input in a ProjectOperator that computes the list as an extra column. let (final_input_op, final_input_columns, col_idx) = if let Some(idx) = list_col_idx { (input_op, input_columns, idx) - } else if matches!( - &unwind.expression, - LogicalExpression::List(_) - | LogicalExpression::Literal(Value::List(_)) - | LogicalExpression::Literal(Value::Vector(_)) - | LogicalExpression::Property { .. } - ) { + } else { // Wrap input in a ProjectOperator that adds the list as an extra column let literal_list = self.convert_expression(&unwind.expression)?; let mut proj_exprs: Vec = @@ -371,24 +372,29 @@ impl super::Planner { let mut cols = input_columns; cols.push("__unwind_list__".to_string()); (project_op, cols, list_col) - } else { - // Fallback: assume column 0 contains the list - (input_op, input_columns, 0) }; // Build output columns: all input columns plus the new variable let mut columns = final_input_columns.clone(); columns.push(unwind.variable.clone()); - // Mark the UNWIND variable as scalar (not a node/edge ID) so that - // plan_return uses LogicalType::Any instead of Node for it. - self.scalar_columns - .borrow_mut() - .insert(unwind.variable.clone()); + // The items of a node or edge list (a collected list, `nodes(p)`, + // `relationships(p)`, ...) are nodes or edges, so a property read takes + // the right entity; the items of any other list are values. + let item = match self.entity_value(&unwind.expression) { + Some(EntityValue::Nodes) => Some(EntityValue::Node), + Some(EntityValue::Edges) => Some(EntityValue::Edge), + _ => None, + }; // Build output schema let mut output_schema = self.derive_schema_from_columns(&final_input_columns); - output_schema.push(LogicalType::Any); // The unwound element type is dynamic + output_schema.push(match item { + Some(EntityValue::Node) => LogicalType::Node, + Some(EntityValue::Edge) => LogicalType::Edge, + _ => LogicalType::Any, + }); + self.set_column_entity(&unwind.variable, item); // Add ORDINALITY column (1-based index) if requested let emit_ordinality = unwind.ordinality_var.is_some(); @@ -848,6 +854,7 @@ impl super::Planner { transaction_id: self.transaction_id, viewing_epoch: self.viewing_epoch, catalog: self.catalog.clone(), + write_counter: self.write_counter(), }, )); diff --git a/crates/grafeo-engine/src/query/planner/lpg/project.rs b/crates/grafeo-engine/src/query/planner/lpg/project.rs index 9b5e9ac91..c33fc8ee5 100644 --- a/crates/grafeo-engine/src/query/planner/lpg/project.rs +++ b/crates/grafeo-engine/src/query/planner/lpg/project.rs @@ -5,9 +5,9 @@ use grafeo_core::execution::operators::EntityValue; use super::{ Arc, Error, FilterExpression, GraphStoreSearch, HashMap, LimitOp, LogicalExpression, - LogicalOperator, LogicalType, NullOrder, Operator, PhysicalSortKey, ProjectExpr, - ProjectOperator, Result, ReturnOp, SkipOp, SortDirection, SortOp, SortOperator, SortOrder, - common, output_column_name, resolved_column_name, value_to_logical_type, + LogicalOperator, LogicalType, Operator, PhysicalSortKey, ProjectExpr, ProjectOperator, Result, + ReturnOp, SkipOp, SortDirection, SortOp, SortOperator, SortOrder, common, output_column_name, + resolved_column_name, value_to_logical_type, }; impl super::Planner { @@ -39,8 +39,7 @@ impl super::Planner { // Apply DISTINCT if requested if ret.distinct { - let schema = vec![LogicalType::Any; columns.len()]; - Ok(common::build_distinct(operator, columns, None, schema)) + Ok(common::build_distinct(operator, columns, None)) } else { Ok((operator, columns)) } @@ -53,23 +52,13 @@ impl super::Planner { input_op: Box, input_columns: Vec, ) -> Result<(Box, Vec)> { - // Expand RETURN * wildcard: replace with all user-visible input columns - let expanded_items; - let items = if ret.items.len() == 1 - && matches!(&ret.items[0].expression, LogicalExpression::Variable(n) if n == "*") - { - expanded_items = input_columns - .iter() - .filter(|col| !col.starts_with('_')) // Skip internal columns - .map(|col| crate::query::plan::ReturnItem { - expression: LogicalExpression::Variable(col.clone()), - alias: None, - }) - .collect::>(); - &expanded_items - } else { - &ret.items - }; + let expanded_items = expand_return_star(&ret.items, &input_columns); + let items = expanded_items.as_deref().unwrap_or(&ret.items); + // EXISTS and COUNT subqueries the edge check cannot answer run per row + // first (see `subquery.rs`); the items read their counts. + let (lifted_items, input_op, input_columns) = + self.lift_return_items(items, input_op, input_columns, ret.input.has_mutations())?; + let items = lifted_items.as_deref().unwrap_or(items); // Build variable to column index mapping let variable_columns: HashMap = input_columns @@ -407,6 +396,17 @@ impl super::Planner { } else { self.plan_operator(&project.input)? }; + // EXISTS and COUNT subqueries the edge check cannot answer run per row + // first (see `subquery.rs`); the projections read their counts. + let (lifted_projections, input_op, input_columns) = self.lift_projections( + &project.projections, + input_op, + input_columns, + project.input.has_mutations(), + )?; + let project_projections = lifted_projections + .as_deref() + .unwrap_or(&project.projections); // Build variable to column index mapping let variable_columns: HashMap = input_columns @@ -436,7 +436,7 @@ impl super::Planner { } } - for projection in &project.projections { + for projection in project_projections { let col_name = output_column_name(projection.alias.as_deref(), &projection.expression); match &projection.expression { @@ -497,6 +497,13 @@ impl super::Planner { .borrow_mut() .insert(col_name.clone(), kind); } + // One item of such a list (`head(r)`, `last(nodes(p))`) + // stays a node or an edge, like a pattern variable. + Some(EntityValue::Edge) => { + output_types.push(LogicalType::Edge); + self.edge_columns.borrow_mut().insert(col_name.clone()); + } + Some(EntityValue::Node) => output_types.push(LogicalType::Node), _ => { output_types.push(LogicalType::Any); // Expression results are scalar values @@ -550,7 +557,7 @@ impl super::Planner { /// them (as ids): `relationships(p)`, `nodes(p)` and list columns (the /// variable of a variable-length edge pattern), also through `reverse`, /// `tail` and slices, and one item of such a list (`head`, `last`, `[i]`). - fn entity_value(&self, expression: &LogicalExpression) -> Option { + pub(super) fn entity_value(&self, expression: &LogicalExpression) -> Option { let item = |kind: EntityValue| match kind { EntityValue::Nodes => Some(EntityValue::Node), EntityValue::Edges => Some(EntityValue::Edge), @@ -639,24 +646,20 @@ impl super::Planner { } let (input_op, columns) = plan_result?; - let schema = self.derive_schema_from_columns(&columns); Ok(crate::query::planner::common::build_limit( input_op, columns, limit.count.value(), - schema, )) } /// Plans a SKIP operator. pub(super) fn plan_skip(&self, skip: &SkipOp) -> Result<(Box, Vec)> { let (input_op, columns) = self.plan_operator(&skip.input)?; - let schema = self.derive_schema_from_columns(&columns); Ok(crate::query::planner::common::build_skip( input_op, columns, skip.count.value(), - schema, )) } @@ -694,8 +697,11 @@ impl super::Planner { // Build augmented Return items: original items plus ORDER BY // expressions that reference variables available in the Match but // not in the Return. This includes both property accesses and - // complex expressions (labels(n)[0], type(r), etc.). - let mut augmented_items = ret.items.clone(); + // complex expressions (labels(n)[0], type(r), etc.). `RETURN *` + // is expanded first: `*` is only expanded when it is the sole item. + let return_items = + expand_return_star(&ret.items, &inner_columns).unwrap_or_else(|| ret.items.clone()); + let mut augmented_items = return_items.clone(); let mut extra_columns = Vec::new(); let mut seen: GrafeoSet = GrafeoSet::default(); for key in &sort.keys { @@ -709,7 +715,7 @@ impl super::Planner { if !inner_vars.contains_key(variable) { continue; } - let already_in_return = ret.items.iter().any(|item| { + let already_in_return = return_items.iter().any(|item| { item.alias.as_deref() == Some(variable.as_str()) || matches!( &item.expression, @@ -735,7 +741,7 @@ impl super::Planner { // property access (possibly under an alias). E.g. // RETURN caller.name AS caller ORDER BY caller.name // already has caller.name in the Return items. - let already_in_return = ret.items.iter().any(|item| { + let already_in_return = return_items.iter().any(|item| { matches!( &item.expression, LogicalExpression::Property { @@ -815,7 +821,6 @@ impl super::Planner { } let mut extra_projections: Vec = Vec::new(); let mut next_col_idx = input_columns.len(); - let mut expr_extra_count: usize = 0; for key in &sort.keys { match &key.expression { @@ -856,12 +861,13 @@ impl super::Planner { }); variable_columns.insert(col_name, next_col_idx); next_col_idx += 1; - expr_extra_count += 1; } } } } + let extra_projection_count = extra_projections.len(); + // Track output columns let mut output_columns = input_columns.clone(); @@ -940,37 +946,28 @@ impl super::Planner { SortOrder::Ascending => SortDirection::Ascending, SortOrder::Descending => SortDirection::Descending, }, - null_order: match key.nulls { - Some(crate::query::plan::NullsOrdering::First) => NullOrder::NullsFirst, - Some(crate::query::plan::NullsOrdering::Last) => NullOrder::NullsLast, - None => NullOrder::NullsLast, // default - }, + null_order: common::physical_null_order(key), }) }) .collect::>>()?; - let output_schema = self.derive_schema_from_columns(&output_columns); - let mut operator: Box = - Box::new(SortOperator::new(input_op, physical_keys, output_schema)); + let mut sort = SortOperator::new(input_op, physical_keys); - // Strip extra columns injected for ORDER BY resolution: both pre-Return - // property projections (sort_extra_count) and synthetic __expr_ columns - // for complex expressions like labels(n)[0] or type(r). - let total_extra = sort_extra_count + expr_extra_count; + // The sort drops the columns added for ORDER BY, which come last: the + // pre-Return projections (sort_extra_count) and every projection added + // after it (properties of a RETURN alias such as `e.w` in + // `RETURN r AS e ORDER BY e.w`, and complex expressions like + // labels(n)[0] or type(r)). It keeps the other columns' types: a + // projection here would make them `Any`, and an edge ID in an `Any` + // column reads its properties from the node with that ID. + let total_extra = sort_extra_count + extra_projection_count; if total_extra > 0 { let keep_count = output_columns.len() - total_extra; - let strip_projections: Vec = - (0..keep_count).map(ProjectExpr::Column).collect(); - let strip_types: Vec = (0..keep_count).map(|_| LogicalType::Any).collect(); - operator = Box::new(ProjectOperator::new( - operator, - strip_projections, - strip_types, - )); + sort = sort.with_output_width(keep_count); output_columns.truncate(keep_count); } - Ok((operator, output_columns)) + Ok((Box::new(sort), output_columns)) } /// Resolves a sort expression to a column index, using projected property columns. @@ -989,16 +986,21 @@ impl super::Planner { /// Derives a schema from column names using the planner's type tracking. /// /// Defaults to `Any` (safe for all value types: scalars, maps, property - /// projections, etc.). Columns explicitly tracked in `edge_columns` get - /// `Edge` for compact `Vec` storage. Mutation operators that add + /// projections, etc.). Columns tracked in `edge_columns` get `Edge` for + /// compact `Vec` storage while they hold edge IDs, not once a + /// RETURN has made them records (it marks them scalar): a record in an + /// `Edge` column would be stored as edge 0. Mutation operators that add /// new entity-ID columns (CREATE, MERGE) should append `Node`/`Edge` /// explicitly after calling this for pass-through columns. pub(super) fn derive_schema_from_columns(&self, columns: &[String]) -> Vec { let edges = self.edge_columns.borrow(); + let scalars = self.scalar_columns.borrow(); columns .iter() .map(|name| { - if edges.contains(name) { + // A RETURN marks its outputs scalar: an edge it returns is a + // record from then on, not an edge ID. + if edges.contains(name) && !scalars.contains(name) { LogicalType::Edge } else { LogicalType::Any @@ -1378,17 +1380,40 @@ impl super::Planner { physical_keys = resolve_logical_to_physical_keys(&sort.keys, &actual_columns)?; } - let schema = self.derive_schema_from_columns(&columns); let op: Box = Box::new(grafeo_core::execution::operators::TopKOperator::new( input_op, physical_keys, k, - schema, )); Ok(Some((op, columns))) } } +/// The items of `RETURN *`: every input column a query can name (internal +/// columns start with `_`). `None` for any other RETURN: `*` is expanded +/// only when it is the only item. +fn expand_return_star( + items: &[crate::query::plan::ReturnItem], + input_columns: &[String], +) -> Option> { + let [item] = items else { + return None; + }; + if !matches!(&item.expression, LogicalExpression::Variable(name) if name == "*") { + return None; + } + Some( + input_columns + .iter() + .filter(|column| !column.starts_with('_')) + .map(|column| crate::query::plan::ReturnItem { + expression: LogicalExpression::Variable(column.clone()), + alias: None, + }) + .collect(), + ) +} + /// Predicts the output column names of `op` without planning it. /// /// Returns `None` for shapes that are not predicted; callers then skip the @@ -1444,8 +1469,8 @@ fn register_return_property_sort_aliases( /// For each logical key: /// - Looks up the column index via `common::resolve_expression_to_column`. /// - Maps `SortOrder` → physical `SortDirection`. -/// - Maps `Option` → physical `NullOrder` (default `NullsLast`, -/// matching `SortKey::ascending`'s default). +/// - Maps `Option` → physical `NullOrder` with +/// [`common::physical_null_order`], as `plan_sort` does. /// /// Returns `Err` if any key fails to resolve in `variable_columns`. /// Callers translate that to `Ok(None)` to fall through to the unfused path. @@ -1453,8 +1478,8 @@ fn resolve_logical_to_physical_keys( keys: &[crate::query::plan::SortKey], variable_columns: &HashMap, ) -> Result> { - use crate::query::plan::{NullsOrdering, SortOrder}; - use grafeo_core::execution::operators::{NullOrder, SortDirection, SortKey as PhysSortKey}; + use crate::query::plan::SortOrder; + use grafeo_core::execution::operators::{SortDirection, SortKey as PhysSortKey}; let mut out = Vec::with_capacity(keys.len()); for key in keys { @@ -1469,16 +1494,10 @@ fn resolve_logical_to_physical_keys( SortOrder::Descending => SortDirection::Descending, }; - let null_order = match key.nulls { - Some(NullsOrdering::First) => NullOrder::NullsFirst, - Some(NullsOrdering::Last) => NullOrder::NullsLast, - None => NullOrder::NullsLast, // default, matches plan_sort - }; - out.push(PhysSortKey { column: col, direction, - null_order, + null_order: common::physical_null_order(key), }); } Ok(out) @@ -1494,7 +1513,7 @@ const OPAQUE_SUBQUERY_VARIABLE: &str = "\0subquery"; /// `sort_needs_augmenting_projection` and `plan_sort`'s pre-return projection /// logic to determine whether ORDER BY references variables that the RETURN /// clause has dropped. -fn collect_vars(expr: &LogicalExpression, out: &mut Vec) { +pub(super) fn collect_vars(expr: &LogicalExpression, out: &mut Vec) { match expr { LogicalExpression::Variable(v) | LogicalExpression::Property { variable: v, .. } diff --git a/crates/grafeo-engine/src/query/planner/lpg/scan.rs b/crates/grafeo-engine/src/query/planner/lpg/scan.rs index 3e7e76444..a3e376548 100644 --- a/crates/grafeo-engine/src/query/planner/lpg/scan.rs +++ b/crates/grafeo-engine/src/query/planner/lpg/scan.rs @@ -66,13 +66,18 @@ impl super::Planner { input_columns.push(scan.variable.clone()); // Use nested loop join to combine input rows with scanned nodes - let join_op = Box::new(NestedLoopJoinOperator::new( + let mut join_op = NestedLoopJoinOperator::new( input_op, scan_operator, None, // No join condition (cross join) PhysicalJoinType::Cross, output_schema, - )); + ); + // A scan after a write (`INSERT ... WITH ... MATCH`) sees the write. + if input.has_mutations() { + join_op = join_op.with_left_first(); + } + let join_op = Box::new(join_op); Ok((join_op, input_columns)) } else { diff --git a/crates/grafeo-engine/src/query/planner/lpg/subquery.rs b/crates/grafeo-engine/src/query/planner/lpg/subquery.rs new file mode 100644 index 000000000..2c9b3eb2d --- /dev/null +++ b/crates/grafeo-engine/src/query/planner/lpg/subquery.rs @@ -0,0 +1,958 @@ +//! `EXISTS` and `COUNT` subqueries planned per row of their input. +//! +//! The edge check (`FilterExpression::ExistsSubquery` and `CountSubquery`) +//! answers a subquery that one edge from a node of the row decides. Every +//! other one runs as a correlated `Apply`: the subquery's plan starts from a +//! row of the outer variables it uses, so its patterns continue from the +//! outer nodes and edges and its expressions read the outer values, and an +//! aggregate counts its rows (for `EXISTS`, up to one). The count becomes a +//! column of the row that the expression reads in place of the subquery. + +use std::collections::HashSet; + +use super::{Arc, Error, LogicalExpression, LogicalOperator, Operator, Result, Value}; +use crate::query::plan::{ + AggregateExpr, AggregateFunction, AggregateOp, BinaryOp, LimitOp, MapProjectionEntry, + ParameterScanOp, ProjectOp, Projection, ReturnItem, +}; +use crate::query::planner::common::output_column_name; +use grafeo_common::types::LogicalType; +use grafeo_common::utils::error::{QueryError, QueryErrorKind}; +use grafeo_core::execution::operators::{ + ApplyOperator, JoinType, NestedLoopJoinOperator, ParameterState, +}; + +impl super::Planner { + /// Whether `expression` has an `EXISTS` or `COUNT` subquery that the edge + /// check cannot answer (outside a list comprehension, list predicate or + /// `reduce`, whose subqueries read the item variable). With the row's + /// `columns`, also one that shares no variable with the row (see + /// [`Self::edge_check_answers`]). + pub(super) fn has_subquery_to_lift( + &self, + expression: &LogicalExpression, + columns: Option<&[String]>, + ) -> bool { + let mut expression = expression.clone(); + let mut found = false; + let _ = visit_subqueries(&mut expression, &mut |subquery| { + found |= !self.edge_check_answers(subquery, columns); + Ok(()) + }); + found + } + + /// Plans every `EXISTS` and `COUNT` subquery in `expression` that the edge + /// check cannot answer as a correlated `Apply` over `input`, which adds a + /// column with its count. Returns the expression with each such subquery + /// replaced by a read of that column (`> 0` for `EXISTS`), and the input + /// and columns with the new ones. `input_writes`: whether the input's + /// plan writes (see [`Self::plan_counted_subquery`]). + pub(super) fn lift_subqueries( + &self, + expression: &LogicalExpression, + mut input: Box, + mut columns: Vec, + input_writes: bool, + ) -> Result<(LogicalExpression, Box, Vec)> { + let mut expression = expression.clone(); + visit_subqueries(&mut expression, &mut |subquery| { + if self.edge_check_answers(subquery, Some(&columns)) { + return Ok(()); + } + let column = self.next_subquery_column(&columns); + let outer = std::mem::replace( + &mut input, + Box::new(grafeo_core::execution::operators::EmptyOperator::new( + vec![], + )), + ); + if let LogicalExpression::ValueSubquery(subplan) = subquery { + input = self.plan_value_subquery( + outer, + &columns, + subplan.as_ref().clone(), + &column, + input_writes, + )?; + columns.push(column.clone()); + *subquery = LogicalExpression::Variable(column); + return Ok(()); + } + let (subplan, exists) = match subquery { + LogicalExpression::ExistsSubquery(subplan) => (subplan.as_ref().clone(), true), + LogicalExpression::CountSubquery(subplan) => (subplan.as_ref().clone(), false), + _ => return Ok(()), + }; + input = self.plan_counted_subquery( + outer, + &columns, + subplan, + exists, + &column, + input_writes, + )?; + columns.push(column.clone()); + let count = LogicalExpression::Variable(column); + *subquery = if exists { + LogicalExpression::Binary { + left: Box::new(count), + op: BinaryOp::Gt, + right: Box::new(LogicalExpression::Literal(Value::Int64(0))), + } + } else { + count + }; + Ok(()) + })?; + Ok((expression, input, columns)) + } + + /// Lifts the subqueries of `RETURN` items (see [`Self::lift_subqueries`]), + /// each item keeping its column name. `None` when no item has one. + pub(super) fn lift_return_items( + &self, + items: &[ReturnItem], + mut input: Box, + mut columns: Vec, + input_writes: bool, + ) -> Result<(Option>, Box, Vec)> { + if !items + .iter() + .any(|item| self.has_subquery_to_lift(&item.expression, Some(&columns))) + { + return Ok((None, input, columns)); + } + let mut lifted = Vec::with_capacity(items.len()); + for item in items { + let alias = Some(output_column_name(item.alias.as_deref(), &item.expression)); + let (expression, lifted_input, lifted_columns) = + self.lift_subqueries(&item.expression, input, columns, input_writes)?; + input = lifted_input; + columns = lifted_columns; + lifted.push(ReturnItem { expression, alias }); + } + Ok((Some(lifted), input, columns)) + } + + /// Lifts the subqueries of `WITH` and `LET` projections (see + /// [`Self::lift_subqueries`]), each keeping its column name. `None` when + /// no projection has one. + pub(super) fn lift_projections( + &self, + projections: &[Projection], + mut input: Box, + mut columns: Vec, + input_writes: bool, + ) -> Result<(Option>, Box, Vec)> { + if !projections + .iter() + .any(|projection| self.has_subquery_to_lift(&projection.expression, Some(&columns))) + { + return Ok((None, input, columns)); + } + let mut lifted = Vec::with_capacity(projections.len()); + for projection in projections { + let alias = Some(output_column_name( + projection.alias.as_deref(), + &projection.expression, + )); + let (expression, lifted_input, lifted_columns) = + self.lift_subqueries(&projection.expression, input, columns, input_writes)?; + input = lifted_input; + columns = lifted_columns; + lifted.push(Projection { expression, alias }); + } + Ok((Some(lifted), input, columns)) + } + + /// Whether the edge check answers this `EXISTS` or `COUNT` subquery: one + /// edge (or, for `EXISTS`, one path of one edge type) from a node, which + /// `extract_exists_pattern` recognizes. `COUNT` counts single edges only. + /// With the row's `columns`, the pattern must also share a node or edge + /// with the row: one that shares none has one answer for all rows, which + /// the edge check would find again for each row by walking every edge. + fn edge_check_answers(&self, subquery: &LogicalExpression, columns: Option<&[String]>) -> bool { + let check = match subquery { + LogicalExpression::ExistsSubquery(subplan) => self.extract_exists_pattern(subplan), + LogicalExpression::CountSubquery(subplan) => { + self.extract_exists_pattern(subplan).and_then(|check| { + if check.one_edge { + Ok(check) + } else { + Err(Error::Internal("COUNT over more than one edge".to_string())) + } + }) + } + // A VALUE subquery always runs as a subquery of its own. + LogicalExpression::ValueSubquery(_) => return false, + _ => return true, + }; + check.is_ok_and(|check| { + columns.is_none_or(|columns| { + columns.iter().any(|column| { + *column == check.start_var + || *column == check.end_var + || check.edge_var.as_ref() == Some(column) + }) + }) + }) + } + + /// A name for the column of a lifted subquery, unique in the query and + /// not one of the row's `columns` (a variable may be named so). + fn next_subquery_column(&self, columns: &[String]) -> String { + loop { + let n = self.subquery_counter.get(); + self.subquery_counter.set(n + 1); + let name = format!("__subquery_{n}"); + if !columns.contains(&name) { + return name; + } + } + } + + /// Plans `subplan` once per row of `outer`, counting its rows (up to one + /// when `exists`), as a correlated `Apply` that adds the count as `column`. + /// A subquery that shares nothing with the row is counted once and joined + /// to every row; below a write (`input_writes`) after all the rows are read, + /// so the count sees what they wrote. + fn plan_counted_subquery( + &self, + outer: Box, + outer_columns: &[String], + subplan: LogicalOperator, + exists: bool, + column: &str, + input_writes: bool, + ) -> Result> { + let (seeded, shared) = seed_with_row(outer_columns, subplan)?; + let counted_input = if exists { + LogicalOperator::Limit(LimitOp { + count: 1.into(), + input: Box::new(seeded), + }) + } else { + seeded + }; + let counted = LogicalOperator::Aggregate(AggregateOp { + group_by: Vec::new(), + aggregates: vec![AggregateExpr { + function: AggregateFunction::Count, + expression: None, + expression2: None, + distinct: false, + alias: Some(column.to_string()), + percentile: None, + separator: None, + }], + input: Box::new(counted_input), + having: None, + }); + let added = AddedColumn { + name: column, + logical_type: LogicalType::Int64, + optional: false, + }; + self.join_per_row(outer, outer_columns, &counted, shared, &added, input_writes) + } + + /// Plans a `VALUE` subquery once per row of `outer`: the value of the one + /// column its first row returns, or null when it returns no row, added + /// as `column` (see [`Self::plan_counted_subquery`] for the rest). + fn plan_value_subquery( + &self, + outer: Box, + outer_columns: &[String], + subplan: LogicalOperator, + column: &str, + input_writes: bool, + ) -> Result> { + let returned = returned_column(&subplan).ok_or_else(|| { + Error::Query(QueryError::new( + QueryErrorKind::Semantic, + "A VALUE subquery returns one column", + )) + })?; + let valued = LogicalOperator::Project(ProjectOp { + projections: vec![Projection { + expression: LogicalExpression::Variable(returned), + alias: Some(column.to_string()), + }], + input: Box::new(LogicalOperator::Limit(LimitOp { + count: 1.into(), + input: Box::new(subplan), + })), + pass_through_input: false, + }); + let (seeded, shared) = seed_with_row(outer_columns, valued)?; + let added = AddedColumn { + name: column, + logical_type: LogicalType::Any, + optional: true, + }; + self.join_per_row(outer, outer_columns, &seeded, shared, &added, input_writes) + } + + /// Joins `inner`, which gives the `added` column and at most one row, to + /// each row of `outer`, with the `shared` columns of the row as its + /// parameters. With nothing shared it runs once; below a write + /// (`input_writes`) after all the rows are read, so it sees what they wrote. + fn join_per_row( + &self, + outer: Box, + outer_columns: &[String], + inner: &LogicalOperator, + shared: Vec, + added: &AddedColumn<'_>, + input_writes: bool, + ) -> Result> { + if shared.is_empty() { + let (inner, _) = self.plan_operator(inner)?; + self.scalar_columns + .borrow_mut() + .insert(added.name.to_string()); + let mut schema = self.derive_schema_from_columns(outer_columns); + schema.push(added.logical_type.clone()); + let join_type = if added.optional { + JoinType::Left + } else { + JoinType::Cross + }; + let mut join = NestedLoopJoinOperator::new(outer, inner, None, join_type, schema); + if input_writes { + join = join.with_left_first(); + } + return Ok(Box::new(join)); + } + + let state = Arc::new(ParameterState::new(shared.clone())); + let indices: Vec = shared + .iter() + .filter_map(|name| outer_columns.iter().position(|column| column == name)) + .collect(); + // A subquery inside this one sets its own state while it is planned; + // the one before this subquery comes back afterwards. + let previous = self + .correlated_param_state + .replace(Some(Arc::clone(&state))); + let planned = self.plan_operator(inner); + *self.correlated_param_state.borrow_mut() = previous; + let (inner, _) = planned?; + self.scalar_columns + .borrow_mut() + .insert(added.name.to_string()); + let apply = ApplyOperator::new_correlated(outer, inner, state, indices); + Ok(Box::new(if added.optional { + apply.with_optional(1) + } else { + apply + })) + } +} + +/// The column a subquery planned per row adds to each row. +struct AddedColumn<'a> { + /// Its name. + name: &'a str, + /// Its type. + logical_type: LogicalType, + /// Whether a row the subquery gives no row keeps a null (`VALUE`); a + /// count always gives one. + optional: bool, +} + +/// The outer variables `subplan` uses (those it names, as far as they are +/// columns of the row; all of them when it has an operator whose names are +/// not known here), and `subplan` seeded with them. A pattern through a node +/// or edge of the outer row matches that node or edge, as a later MATCH does: +/// with the parameters at its start, the variables they bring are bound, and +/// the cycle pass turns a pattern variable bound again into a check that it +/// is the same. +fn seed_with_row( + outer_columns: &[String], + subplan: LogicalOperator, +) -> Result<(LogicalOperator, Vec)> { + let shared: Vec = match subplan_variables(&subplan) { + Some(names) => outer_columns + .iter() + .filter(|column| names.contains(*column)) + .cloned() + .collect(), + None => outer_columns + .iter() + .filter(|column| !column.starts_with("__")) + .cloned() + .collect(), + }; + let mut seeded = subplan; + if !shared.is_empty() + && !seed_with_parameters( + &mut seeded, + LogicalOperator::ParameterScan(ParameterScanOp { + columns: shared.clone(), + }), + ) + { + return Err(Error::Internal( + "Unsupported subquery: no pattern to start from the outer row".to_string(), + )); + } + Ok((crate::query::optimizer::close_cycles(seeded), shared)) +} + +/// The name of the one column a subquery's final `RETURN` gives (under its +/// `ORDER BY`, `SKIP`, `LIMIT` and `DISTINCT`, and in every part of a +/// `UNION`), or `None` for another plan or several columns. +fn returned_column(plan: &LogicalOperator) -> Option { + match plan { + LogicalOperator::Union(union) => { + let mut names = union.inputs.iter().map(returned_column); + let first = names.next()??; + names + .all(|name| name.as_ref() == Some(&first)) + .then_some(first) + } + LogicalOperator::Return(ret) => match ret.items.as_slice() { + [item] => Some(output_column_name(item.alias.as_deref(), &item.expression)), + _ => None, + }, + LogicalOperator::Sort(op) => returned_column(&op.input), + LogicalOperator::Limit(op) => returned_column(&op.input), + LogicalOperator::Skip(op) => returned_column(&op.input), + LogicalOperator::Distinct(op) => returned_column(&op.input), + _ => None, + } +} + +/// Calls `f` on every `EXISTS`, `COUNT` and `VALUE` subquery in `expression`, +/// outside the bodies of list comprehensions, list predicates and `reduce`, +/// which read their item variable (not a column of the row). +fn visit_subqueries( + expression: &mut LogicalExpression, + f: &mut dyn FnMut(&mut LogicalExpression) -> Result<()>, +) -> Result<()> { + match expression { + LogicalExpression::ExistsSubquery(_) + | LogicalExpression::CountSubquery(_) + | LogicalExpression::ValueSubquery(_) => f(expression), + LogicalExpression::Binary { left, right, .. } => { + visit_subqueries(left, f)?; + visit_subqueries(right, f) + } + LogicalExpression::Unary { operand, .. } => visit_subqueries(operand, f), + LogicalExpression::FunctionCall { args, .. } | LogicalExpression::List(args) => { + args.iter_mut().try_for_each(|arg| visit_subqueries(arg, f)) + } + LogicalExpression::Map(entries) => entries + .iter_mut() + .try_for_each(|(_, value)| visit_subqueries(value, f)), + LogicalExpression::IndexAccess { base, index } => { + visit_subqueries(base, f)?; + visit_subqueries(index, f) + } + LogicalExpression::MapAccess { base, .. } => visit_subqueries(base, f), + LogicalExpression::SliceAccess { base, start, end } => { + visit_subqueries(base, f)?; + if let Some(start) = start { + visit_subqueries(start, f)?; + } + if let Some(end) = end { + visit_subqueries(end, f)?; + } + Ok(()) + } + LogicalExpression::Case { + operand, + when_clauses, + else_clause, + } => { + if let Some(operand) = operand { + visit_subqueries(operand, f)?; + } + for (condition, result) in when_clauses { + visit_subqueries(condition, f)?; + visit_subqueries(result, f)?; + } + if let Some(else_clause) = else_clause { + visit_subqueries(else_clause, f)?; + } + Ok(()) + } + LogicalExpression::ListComprehension { list_expr, .. } + | LogicalExpression::ListPredicate { list_expr, .. } => visit_subqueries(list_expr, f), + LogicalExpression::Reduce { initial, list, .. } => { + visit_subqueries(initial, f)?; + visit_subqueries(list, f) + } + LogicalExpression::MapProjection { entries, .. } => { + entries.iter_mut().try_for_each(|entry| match entry { + MapProjectionEntry::LiteralEntry(_, value) => visit_subqueries(value, f), + MapProjectionEntry::PropertySelector(_) | MapProjectionEntry::AllProperties => { + Ok(()) + } + }) + } + LogicalExpression::Literal(_) + | LogicalExpression::Variable(_) + | LogicalExpression::Property { .. } + | LogicalExpression::Parameter(_) + | LogicalExpression::Labels(_) + | LogicalExpression::Type(_) + | LogicalExpression::Id(_) + | LogicalExpression::PatternComprehension { .. } => Ok(()), + } +} + +/// Puts `parameters` under the first pattern of `plan`, as the input of its +/// leftmost node or edge scan, so the scan continues from the outer row (a +/// scan of an outer variable reuses its value). A parameter scan the +/// translator joined with the pattern (a join without a condition, which would +/// pair every outer row with every match) gives way to the pattern, seeded the +/// same way; one on its own is replaced. Returns false when the plan has no +/// scan to start from. +fn seed_with_parameters(plan: &mut LogicalOperator, parameters: LogicalOperator) -> bool { + if let LogicalOperator::Join(join) = plan + && (matches!(*join.left, LogicalOperator::ParameterScan(_)) + || matches!(*join.right, LogicalOperator::ParameterScan(_))) + { + let pattern = if matches!(*join.left, LogicalOperator::ParameterScan(_)) { + std::mem::replace(&mut *join.right, LogicalOperator::Empty) + } else { + std::mem::replace(&mut *join.left, LogicalOperator::Empty) + }; + *plan = pattern; + return seed_with_parameters(plan, parameters); + } + match plan { + // The one empty row a subquery that starts with OPTIONAL MATCH + // starts from becomes the outer row. + LogicalOperator::ParameterScan(_) | LogicalOperator::Empty => { + *plan = parameters; + true + } + LogicalOperator::NodeScan(scan) => match &mut scan.input { + Some(input) => seed_with_parameters(input, parameters), + None => { + scan.input = Some(Box::new(parameters)); + true + } + }, + LogicalOperator::EdgeScan(scan) => match &mut scan.input { + Some(input) => seed_with_parameters(input, parameters), + None => { + scan.input = Some(Box::new(parameters)); + true + } + }, + LogicalOperator::Expand(expand) => seed_with_parameters(&mut expand.input, parameters), + LogicalOperator::Filter(filter) => seed_with_parameters(&mut filter.input, parameters), + LogicalOperator::Project(project) => seed_with_parameters(&mut project.input, parameters), + LogicalOperator::Return(ret) => seed_with_parameters(&mut ret.input, parameters), + LogicalOperator::Aggregate(aggregate) => { + seed_with_parameters(&mut aggregate.input, parameters) + } + LogicalOperator::Limit(limit) => seed_with_parameters(&mut limit.input, parameters), + LogicalOperator::Skip(skip) => seed_with_parameters(&mut skip.input, parameters), + LogicalOperator::Sort(sort) => seed_with_parameters(&mut sort.input, parameters), + LogicalOperator::Distinct(distinct) => { + seed_with_parameters(&mut distinct.input, parameters) + } + LogicalOperator::Bind(bind) => seed_with_parameters(&mut bind.input, parameters), + LogicalOperator::Join(join) => seed_with_parameters(&mut join.left, parameters), + LogicalOperator::LeftJoin(join) => seed_with_parameters(&mut join.left, parameters), + LogicalOperator::AntiJoin(join) => seed_with_parameters(&mut join.left, parameters), + LogicalOperator::Unwind(unwind) => { + if matches!(*unwind.input, LogicalOperator::Empty) { + *unwind.input = parameters; + true + } else { + seed_with_parameters(&mut unwind.input, parameters) + } + } + _ => false, + } +} + +/// Whether a subquery reads a value of the outer row other than through a +/// node or edge its patterns share with it: a variable its expressions use +/// that its patterns do not bind (`{id: s.id}`, `WHERE x.id = s.id`, an +/// `UNWIND` variable, a parameter scan of outer variables). Such a subquery +/// runs per row; one tied to the row by shared pattern variables alone can be +/// a semi-join on them. One that starts with OPTIONAL MATCH is not tied by +/// its patterns (its row of nulls binds none of them), so it runs per row too. +pub(super) fn reads_outer_values(subplan: &LogicalOperator) -> bool { + if starts_with_optional_match(subplan) { + return true; + } + let Some(used) = subplan_variables(subplan) else { + return true; + }; + let mut bound = HashSet::new(); + bound_names(subplan, &mut bound); + used.iter().any(|name| !bound.contains(name)) +} + +/// Whether `plan` starts with an OPTIONAL MATCH: a left join of one empty row. +fn starts_with_optional_match(plan: &LogicalOperator) -> bool { + match plan { + LogicalOperator::LeftJoin(join) => { + matches!(join.left.as_ref(), LogicalOperator::Empty) + || starts_with_optional_match(&join.left) + } + LogicalOperator::Join(join) => starts_with_optional_match(&join.left), + LogicalOperator::Filter(op) => starts_with_optional_match(&op.input), + LogicalOperator::Project(op) => starts_with_optional_match(&op.input), + LogicalOperator::Return(op) => starts_with_optional_match(&op.input), + LogicalOperator::Aggregate(op) => starts_with_optional_match(&op.input), + LogicalOperator::Limit(op) => starts_with_optional_match(&op.input), + LogicalOperator::Skip(op) => starts_with_optional_match(&op.input), + LogicalOperator::Sort(op) => starts_with_optional_match(&op.input), + LogicalOperator::Distinct(op) => starts_with_optional_match(&op.input), + LogicalOperator::Expand(op) => starts_with_optional_match(&op.input), + LogicalOperator::NodeScan(op) => { + op.input.as_deref().is_some_and(starts_with_optional_match) + } + LogicalOperator::EdgeScan(op) => { + op.input.as_deref().is_some_and(starts_with_optional_match) + } + _ => false, + } +} + +/// The names `plan` binds itself: its pattern variables, projection and +/// aggregate aliases, and the variables of `UNWIND` and `LET`. +fn bound_names(plan: &LogicalOperator, names: &mut HashSet) { + match plan { + LogicalOperator::NodeScan(scan) => { + names.insert(scan.variable.clone()); + } + LogicalOperator::EdgeScan(scan) => { + names.insert(scan.variable.clone()); + } + LogicalOperator::Expand(expand) => { + names.insert(expand.from_variable.clone()); + names.insert(expand.to_variable.clone()); + names.extend(expand.edge_variable.iter().cloned()); + names.extend(expand.path_alias.iter().cloned()); + } + LogicalOperator::Project(project) => { + names.extend(project.projections.iter().filter_map(|p| p.alias.clone())); + } + LogicalOperator::Return(ret) => { + names.extend(ret.items.iter().filter_map(|item| item.alias.clone())); + } + // A group key only reads a name: one the subquery binds is bound by + // its pattern, and one from the outer row stays an outer value. + LogicalOperator::Aggregate(aggregate) => { + names.extend(aggregate.aggregates.iter().filter_map(|a| a.alias.clone())); + } + LogicalOperator::Unwind(unwind) => { + names.insert(unwind.variable.clone()); + names.extend(unwind.ordinality_var.iter().cloned()); + names.extend(unwind.offset_var.iter().cloned()); + } + LogicalOperator::Bind(bind) => { + names.insert(bind.variable.clone()); + } + _ => {} + } + for child in plan.children() { + bound_names(child, names); + } +} + +/// Every variable name `plan` binds or reads, also in the subqueries of its +/// expressions; `None` when it has an operator this does not look into. +fn subplan_variables(plan: &LogicalOperator) -> Option> { + let mut names = HashSet::new(); + plan_names(plan, &mut names).then_some(names) +} + +fn plan_names(plan: &LogicalOperator, names: &mut HashSet) -> bool { + match plan { + LogicalOperator::NodeScan(scan) => { + names.insert(scan.variable.clone()); + scan.input + .as_deref() + .is_none_or(|input| plan_names(input, names)) + } + LogicalOperator::EdgeScan(scan) => { + names.insert(scan.variable.clone()); + scan.input + .as_deref() + .is_none_or(|input| plan_names(input, names)) + } + LogicalOperator::Expand(expand) => { + names.insert(expand.from_variable.clone()); + names.insert(expand.to_variable.clone()); + names.extend(expand.edge_variable.iter().cloned()); + names.extend(expand.path_alias.iter().cloned()); + plan_names(&expand.input, names) + } + LogicalOperator::Filter(filter) => { + expression_names(&filter.predicate, names) && plan_names(&filter.input, names) + } + LogicalOperator::Project(project) => { + project + .projections + .iter() + .all(|projection| expression_names(&projection.expression, names)) + && plan_names(&project.input, names) + } + LogicalOperator::Return(ret) => { + ret.items + .iter() + .all(|item| expression_names(&item.expression, names)) + && plan_names(&ret.input, names) + } + LogicalOperator::Aggregate(aggregate) => { + aggregate + .group_by + .iter() + .all(|key| expression_names(key, names)) + && aggregate.aggregates.iter().all(|aggregate| { + aggregate + .expression + .iter() + .chain(&aggregate.expression2) + .all(|expression| expression_names(expression, names)) + }) + && aggregate + .having + .as_ref() + .is_none_or(|having| expression_names(having, names)) + && plan_names(&aggregate.input, names) + } + LogicalOperator::Sort(sort) => { + sort.keys + .iter() + .all(|key| expression_names(&key.expression, names)) + && plan_names(&sort.input, names) + } + LogicalOperator::Limit(limit) => plan_names(&limit.input, names), + LogicalOperator::Skip(skip) => plan_names(&skip.input, names), + LogicalOperator::Distinct(distinct) => plan_names(&distinct.input, names), + LogicalOperator::Unwind(unwind) => { + names.insert(unwind.variable.clone()); + expression_names(&unwind.expression, names) && plan_names(&unwind.input, names) + } + LogicalOperator::Bind(bind) => { + names.insert(bind.variable.clone()); + expression_names(&bind.expression, names) && plan_names(&bind.input, names) + } + LogicalOperator::Join(join) => { + join.conditions.iter().all(|condition| { + expression_names(&condition.left, names) + && expression_names(&condition.right, names) + }) && plan_names(&join.left, names) + && plan_names(&join.right, names) + } + LogicalOperator::LeftJoin(join) => { + join.condition + .as_ref() + .is_none_or(|condition| expression_names(condition, names)) + && plan_names(&join.left, names) + && plan_names(&join.right, names) + } + LogicalOperator::AntiJoin(join) => { + plan_names(&join.left, names) && plan_names(&join.right, names) + } + LogicalOperator::Apply(apply) => { + names.extend(apply.shared_variables.iter().cloned()); + plan_names(&apply.input, names) && plan_names(&apply.subplan, names) + } + LogicalOperator::ParameterScan(scan) => { + names.extend(scan.columns.iter().cloned()); + true + } + LogicalOperator::Union(union) => union.inputs.iter().all(|input| plan_names(input, names)), + LogicalOperator::Empty => true, + _ => false, + } +} + +fn expression_names(expression: &LogicalExpression, names: &mut HashSet) -> bool { + match expression { + LogicalExpression::Variable(name) + | LogicalExpression::Labels(name) + | LogicalExpression::Type(name) + | LogicalExpression::Id(name) => { + names.insert(name.clone()); + true + } + LogicalExpression::Property { variable, .. } => { + names.insert(variable.clone()); + true + } + LogicalExpression::Literal(_) | LogicalExpression::Parameter(_) => true, + LogicalExpression::Binary { left, right, .. } => { + expression_names(left, names) && expression_names(right, names) + } + LogicalExpression::Unary { operand, .. } => expression_names(operand, names), + LogicalExpression::FunctionCall { args, .. } | LogicalExpression::List(args) => { + args.iter().all(|arg| expression_names(arg, names)) + } + LogicalExpression::Map(entries) => entries + .iter() + .all(|(_, value)| expression_names(value, names)), + LogicalExpression::IndexAccess { base, index } => { + expression_names(base, names) && expression_names(index, names) + } + LogicalExpression::MapAccess { base, .. } => expression_names(base, names), + LogicalExpression::SliceAccess { base, start, end } => { + expression_names(base, names) + && start.as_deref().is_none_or(|e| expression_names(e, names)) + && end.as_deref().is_none_or(|e| expression_names(e, names)) + } + LogicalExpression::Case { + operand, + when_clauses, + else_clause, + } => { + operand + .as_deref() + .is_none_or(|e| expression_names(e, names)) + && when_clauses.iter().all(|(condition, result)| { + expression_names(condition, names) && expression_names(result, names) + }) + && else_clause + .as_deref() + .is_none_or(|e| expression_names(e, names)) + } + LogicalExpression::ListComprehension { + list_expr, + filter_expr, + map_expr, + .. + } => { + expression_names(list_expr, names) + && filter_expr + .as_deref() + .is_none_or(|e| expression_names(e, names)) + && expression_names(map_expr, names) + } + LogicalExpression::ListPredicate { + list_expr, + predicate, + .. + } => expression_names(list_expr, names) && expression_names(predicate, names), + LogicalExpression::Reduce { + initial, + list, + expression, + .. + } => { + expression_names(initial, names) + && expression_names(list, names) + && expression_names(expression, names) + } + LogicalExpression::MapProjection { base, entries } => { + names.insert(base.clone()); + entries.iter().all(|entry| match entry { + MapProjectionEntry::LiteralEntry(_, value) => expression_names(value, names), + MapProjectionEntry::PropertySelector(_) | MapProjectionEntry::AllProperties => true, + }) + } + LogicalExpression::ExistsSubquery(subplan) + | LogicalExpression::CountSubquery(subplan) + | LogicalExpression::ValueSubquery(subplan) => plan_names(subplan, names), + LogicalExpression::PatternComprehension { + subplan, + projection, + } => plan_names(subplan, names) && expression_names(projection, names), + } +} + +#[cfg(test)] +mod tests { + use super::*; + use crate::query::plan::{AggregateExpr, AggregateFunction, AggregateOp, NodeScanOp}; + + /// A subquery grouping by a name its patterns do not bind (here `k`, an + /// outer `UNWIND` variable) reads that outer value, so it runs per row; + /// grouping by its own pattern variable does not. + #[test] + fn a_group_key_from_the_outer_row_is_an_outer_value() { + let grouped_by = |key: &str| { + LogicalOperator::Aggregate(AggregateOp { + group_by: vec![LogicalExpression::Variable(key.into())], + aggregates: vec![AggregateExpr { + function: AggregateFunction::Count, + expression: Some(LogicalExpression::Variable("p".into())), + expression2: None, + distinct: false, + alias: Some("c".into()), + percentile: None, + separator: None, + }], + input: Box::new(LogicalOperator::NodeScan(NodeScanOp { + variable: "p".into(), + label: None, + input: None, + })), + having: None, + }) + }; + assert!(reads_outer_values(&grouped_by("k"))); + assert!(!reads_outer_values(&grouped_by("p"))); + } + + /// A VALUE subquery returns the one column of its RETURN, also through + /// ordering and in every part of a UNION (which must agree). + #[test] + fn the_returned_column_of_a_value_subquery() { + use crate::query::plan::{ReturnOp, UnionOp}; + + let returning = |alias: &str| { + LogicalOperator::Return(ReturnOp { + items: vec![ReturnItem { + expression: LogicalExpression::Variable("f".into()), + alias: Some(alias.into()), + }], + distinct: false, + input: Box::new(LogicalOperator::Empty), + }) + }; + let union = |a: &str, b: &str| { + LogicalOperator::Union(UnionOp { + inputs: vec![returning(a), returning(b)], + }) + }; + assert_eq!(returned_column(&returning("n")), Some("n".to_string())); + assert_eq!(returned_column(&union("n", "n")), Some("n".to_string())); + assert_eq!(returned_column(&union("n", "m")), None); + } + + /// A subquery that starts with OPTIONAL MATCH runs per row (its row of + /// nulls binds no shared variable), also under a node or edge scan. + #[test] + fn a_subquery_that_starts_with_optional_match_runs_per_row() { + use crate::query::plan::{EdgeScanOp, LeftJoinOp}; + + let optional = || { + Box::new(LogicalOperator::LeftJoin(LeftJoinOp { + left: Box::new(LogicalOperator::Empty), + right: Box::new(LogicalOperator::NodeScan(NodeScanOp { + variable: "p".into(), + label: None, + input: None, + })), + condition: None, + })) + }; + let under_a_node_scan = LogicalOperator::NodeScan(NodeScanOp { + variable: "q".into(), + label: None, + input: Some(optional()), + }); + let under_an_edge_scan = LogicalOperator::EdgeScan(EdgeScanOp { + variable: "r".into(), + edge_types: Vec::new(), + input: Some(optional()), + }); + assert!(reads_outer_values(&optional())); + assert!(reads_outer_values(&under_a_node_scan)); + assert!(reads_outer_values(&under_an_edge_scan)); + } +} diff --git a/crates/grafeo-engine/src/query/planner/rdf/mod.rs b/crates/grafeo-engine/src/query/planner/rdf/mod.rs index 831130929..4d056db74 100644 --- a/crates/grafeo-engine/src/query/planner/rdf/mod.rs +++ b/crates/grafeo-engine/src/query/planner/rdf/mod.rs @@ -318,7 +318,6 @@ impl RdfPlanner { if self.shuffle_unordered && !super::common::orders_rows(&logical_plan.root) { Box::new(grafeo_core::execution::operators::ShuffleOperator::new( operator, - vec![LogicalType::Any; columns.len()], )) } else { operator @@ -633,12 +632,7 @@ impl RdfPlanner { ) -> Result<(Box, Vec, Vec)> { use crate::query::planner::common; let (input_op, columns, types) = self.plan_operator(&distinct.input)?; - let (op, cols) = common::build_distinct( - input_op, - columns, - distinct.columns.as_deref(), - types.clone(), - ); + let (op, cols) = common::build_distinct(input_op, columns, distinct.columns.as_deref()); Ok((op, cols, types)) } @@ -649,7 +643,7 @@ impl RdfPlanner { ) -> Result<(Box, Vec, Vec)> { use crate::query::planner::common; let (input_op, columns, types) = self.plan_operator(&limit.input)?; - let (op, cols) = common::build_limit(input_op, columns, limit.count.value(), types.clone()); + let (op, cols) = common::build_limit(input_op, columns, limit.count.value()); Ok((op, cols, types)) } @@ -660,7 +654,7 @@ impl RdfPlanner { ) -> Result<(Box, Vec, Vec)> { use crate::query::planner::common; let (input_op, columns, types) = self.plan_operator(&skip.input)?; - let (op, cols) = common::build_skip(input_op, columns, skip.count.value(), types.clone()); + let (op, cols) = common::build_skip(input_op, columns, skip.count.value()); Ok((op, cols, types)) } @@ -671,7 +665,7 @@ impl RdfPlanner { ) -> Result<(Box, Vec, Vec)> { use crate::query::plan::SortOrder; use grafeo_core::execution::operators::{ - FilterExpression, NullOrder, ProjectExpr, ProjectOperator, SortDirection, SortKey, + FilterExpression, ProjectExpr, ProjectOperator, SortDirection, SortKey, }; let (mut input_op, columns, types) = self.plan_operator(&sort.input)?; @@ -732,13 +726,17 @@ impl RdfPlanner { SortOrder::Ascending => SortDirection::Ascending, SortOrder::Descending => SortDirection::Descending, }, - null_order: NullOrder::NullsLast, + null_order: super::common::physical_null_order(key), }) }) .collect::>>()?; - let operator = Box::new(SortOperator::new(input_op, physical_keys, types.clone())); - Ok((operator, columns, types)) + let mut sort = SortOperator::new(input_op, physical_keys); + // The columns computed for ORDER BY come last and are not returned. + if !expression_projections.is_empty() { + sort = sort.with_output_width(columns.len()); + } + Ok((Box::new(sort), columns, types)) } /// Plans a PROJECT operator. @@ -1215,6 +1213,7 @@ impl RdfPlanner { &right_columns, &left_types, &right_types, + None, )) } @@ -6571,6 +6570,82 @@ mod tests { ); } + #[test] + fn test_plan_sort_honors_explicit_null_order() { + use crate::query::plan::{LeftJoinOp, NullsOrdering, SortKey, SortOp, SortOrder}; + let store = Arc::new(RdfStore::new()); + for (person, name, age) in [ + ("alix", "Alix", Some("30")), + ("gus", "Gus", None), + ("vincent", "Vincent", Some("25")), + ] { + let subject = Term::iri(format!("http://example.org/{person}")); + store.insert(Triple::new( + subject.clone(), + Term::iri("http://xmlns.com/foaf/0.1/name"), + Term::literal(name), + )); + if let Some(age) = age { + store.insert(Triple::new( + subject, + Term::iri("http://xmlns.com/foaf/0.1/age"), + Term::literal(age), + )); + } + } + let scan = |predicate: &str, object: &str| { + LogicalOperator::TripleScan(TripleScanOp { + subject: TripleComponent::Variable("s".to_string()), + predicate: TripleComponent::Iri(format!("http://xmlns.com/foaf/0.1/{predicate}")), + object: TripleComponent::Variable(object.to_string()), + graph: None, + input: None, + dataset: None, + }) + }; + // The ages in sort order, `None` for the person without one. + let ages = |order: SortOrder, nulls: NullsOrdering| { + let sort = LogicalOperator::Sort(SortOp { + keys: vec![SortKey { + expression: LogicalExpression::Variable("age".to_string()), + order, + nulls: Some(nulls), + }], + input: Box::new(LogicalOperator::LeftJoin(LeftJoinOp { + left: Box::new(scan("name", "name")), + right: Box::new(scan("age", "age")), + condition: None, + })), + }); + let physical = RdfPlanner::new(Arc::clone(&store)) + .plan(&LogicalPlan::new(sort)) + .unwrap(); + let column = physical.columns.iter().position(|c| c == "age").unwrap(); + let mut op = physical.operator; + let mut ages = Vec::new(); + while let Some(chunk) = op.next().unwrap() { + let values = chunk.column(column).unwrap(); + for row in chunk.selected_indices() { + ages.push(match values.get_value(row) { + Some(Value::String(age)) => Some(age.to_string()), + _ => None, + }); + } + } + ages + }; + let some = |age: &str| Some(age.to_string()); + + assert_eq!( + ages(SortOrder::Ascending, NullsOrdering::First), + vec![None, some("25"), some("30")] + ); + assert_eq!( + ages(SortOrder::Descending, NullsOrdering::Last), + vec![some("30"), some("25"), None] + ); + } + #[test] fn test_plan_insert_triple_concrete() { let store = Arc::new(RdfStore::new()); diff --git a/crates/grafeo-engine/src/query/processor.rs b/crates/grafeo-engine/src/query/processor.rs index 9981ed784..86f112a6f 100644 --- a/crates/grafeo-engine/src/query/processor.rs +++ b/crates/grafeo-engine/src/query/processor.rs @@ -297,7 +297,9 @@ impl QueryProcessor { // 2. Substitute parameters if provided (merge defaults from the plan first) let has_defaults = !logical_plan.default_params.is_empty(); - if params.is_some() || has_defaults { + // A parameter nobody supplied fails here, before planning; only an + // EXPLAIN without parameters shows the plan with them unresolved. + if params.is_some() || has_defaults || !logical_plan.explain { let merged = if has_defaults { let mut merged = logical_plan.default_params.clone(); if let Some(params) = params { @@ -396,8 +398,9 @@ impl QueryProcessor { } #[allow(unreachable_patterns)] _ => Err(Error::Internal(format!( - "Language {:?} is not an LPG language", - language + "Language {:?} is not an LPG language ({} bytes of query)", + language, + query.len() ))), } } @@ -482,7 +485,9 @@ impl QueryProcessor { // 2. Substitute parameters if provided (merge defaults from the plan first) let has_defaults = !logical_plan.default_params.is_empty(); - if params.is_some() || has_defaults { + // A parameter nobody supplied fails here, before planning; only an + // EXPLAIN without parameters shows the plan with them unresolved. + if params.is_some() || has_defaults || !logical_plan.explain { let merged = if has_defaults { let mut merged = logical_plan.default_params.clone(); if let Some(params) = params { @@ -778,7 +783,18 @@ fn substitute_in_operator(op: &mut LogicalOperator, params: &QueryParams) -> Res } LogicalOperator::Return(ret) => { for item in &mut ret.items { + // An unaliased column that reads a parameter is named after + // the query text (`$x`), not after the value replacing it. + let name = item.alias.is_none().then(|| { + crate::query::planner::common::output_column_name(None, &item.expression) + }); substitute_in_expression(&mut item.expression, params)?; + if let Some(name) = name + && name + != crate::query::planner::common::output_column_name(None, &item.expression) + { + item.alias = Some(name); + } } substitute_in_operator(&mut ret.input, params)?; } @@ -1053,7 +1069,10 @@ fn substitute_in_expression(expr: &mut LogicalExpression, params: &QueryParams) if let Some(value) = params.get(name) { *expr = LogicalExpression::Literal(value.clone()); } else { - return Err(Error::Internal(format!("Missing parameter: ${}", name))); + return Err(Error::Query(grafeo_common::utils::error::QueryError::new( + grafeo_common::utils::error::QueryErrorKind::Semantic, + format!("Missing parameter: ${name}"), + ))); } } LogicalExpression::Binary { left, right, .. } => { diff --git a/crates/grafeo-engine/src/query/translators/common.rs b/crates/grafeo-engine/src/query/translators/common.rs index 3a1ea4ce7..9db54db73 100644 --- a/crates/grafeo-engine/src/query/translators/common.rs +++ b/crates/grafeo-engine/src/query/translators/common.rs @@ -9,11 +9,73 @@ use std::sync::atomic::{AtomicU32, Ordering}; use crate::query::plan::{ AggregateFunction, BinaryOp, CountExpr, DistinctOp, FilterOp, LeftJoinOp, LimitOp, - LogicalExpression, LogicalOperator, ReturnItem, ReturnOp, SkipOp, SortKey, SortOp, UnaryOp, + LogicalExpression, LogicalOperator, ReturnItem, ReturnOp, SkipOp, SortKey, SortOp, }; use grafeo_common::types::Value; use grafeo_common::utils::error::{Error, QueryError, QueryErrorKind, Result}; +/// Expands the `RETURN *` that ends a `CALL` subquery into the variables the +/// subquery binds itself, in name order: the variables of the outer row +/// (`outer`) stay where they are, and internal names (`_...`) are not +/// returned. Fails when those variables are not known, so that the subquery +/// names what it returns. +pub(crate) fn expand_subquery_return_star( + subplan: &mut LogicalOperator, + outer: Option<&HashSet>, +) -> Result<()> { + let Some(ret) = final_return_mut(subplan) else { + return Ok(()); + }; + let [item] = ret.items.as_slice() else { + return Ok(()); + }; + if !matches!(&item.expression, LogicalExpression::Variable(name) if name == "*") { + return Ok(()); + } + let (Some(outer), Some(bound)) = (outer, ret.input.bound_variables(outer)) else { + return Err(Error::Query(QueryError::new( + QueryErrorKind::Semantic, + "RETURN * in this CALL subquery cannot tell which variables it binds: return them by name", + ))); + }; + let mut names: Vec = bound + .into_iter() + .filter(|name| !name.starts_with('_') && !outer.contains(name)) + .collect(); + names.sort(); + ret.items = names + .into_iter() + .map(|name| ReturnItem { + expression: LogicalExpression::Variable(name), + alias: None, + }) + .collect(); + Ok(()) +} + +/// The `RETURN` that ends `plan`, under the `ORDER BY`, `SKIP`, `LIMIT` or +/// `DISTINCT` that follow it. +fn final_return_mut(plan: &mut LogicalOperator) -> Option<&mut ReturnOp> { + match plan { + LogicalOperator::Return(ret) => Some(ret), + LogicalOperator::Sort(op) => final_return_mut(&mut op.input), + LogicalOperator::Limit(op) => final_return_mut(&mut op.input), + LogicalOperator::Skip(op) => final_return_mut(&mut op.input), + LogicalOperator::Distinct(op) => final_return_mut(&mut op.input), + _ => None, + } +} + +/// The error for a `WITH` item that is an expression without a name. As in +/// openCypher, later clauses refer to what a `WITH` passes on by name, and a +/// property read such as `n.name` does not keep `n`. +pub(crate) fn unaliased_with_expression() -> Error { + Error::Query(QueryError::new( + QueryErrorKind::Semantic, + "Expression in WITH must be aliased (use AS)", + )) +} + /// Returns true if the function name is a recognized aggregate function. pub(crate) fn is_aggregate_function(name: &str) -> bool { matches!( @@ -193,6 +255,7 @@ impl VarGen { } /// Returns the current counter value without incrementing. + #[cfg(any(feature = "gremlin", test))] pub fn current(&self) -> u32 { self.counter.load(Ordering::Relaxed) } @@ -277,9 +340,8 @@ pub(crate) fn map_access(base: LogicalExpression, key: &str) -> Result Result bool { match expr { LogicalExpression::Variable(_) | LogicalExpression::Parameter(_) | LogicalExpression::Property { .. } | LogicalExpression::Map(_) + | LogicalExpression::MapProjection { .. } | LogicalExpression::MapAccess { .. } => true, LogicalExpression::IndexAccess { base, .. } => { matches!(**base, LogicalExpression::List(_)) || can_be_map(base) @@ -370,25 +433,34 @@ pub(crate) fn collect_expression_variables(expr: &LogicalExpression, vars: &mut collect_expression_variables(else_expr, vars); } } + // The variable a comprehension, list predicate or `reduce` binds is + // its own: only the other names its body uses come from outside. LogicalExpression::ListComprehension { + variable, list_expr, filter_expr, map_expr, - .. } => { collect_expression_variables(list_expr, vars); + let mut body = HashSet::new(); if let Some(filter) = filter_expr { - collect_expression_variables(filter, vars); + collect_expression_variables(filter, &mut body); } - collect_expression_variables(map_expr, vars); + collect_expression_variables(map_expr, &mut body); + body.remove(variable); + vars.extend(body); } LogicalExpression::ListPredicate { + variable, list_expr, predicate, .. } => { collect_expression_variables(list_expr, vars); - collect_expression_variables(predicate, vars); + let mut body = HashSet::new(); + collect_expression_variables(predicate, &mut body); + body.remove(variable); + vars.extend(body); } LogicalExpression::MapProjection { base, entries } => { vars.insert(base.clone()); @@ -399,14 +471,19 @@ pub(crate) fn collect_expression_variables(expr: &LogicalExpression, vars: &mut } } LogicalExpression::Reduce { + accumulator, initial, + variable, list, expression, - .. } => { collect_expression_variables(initial, vars); collect_expression_variables(list, vars); - collect_expression_variables(expression, vars); + let mut body = HashSet::new(); + collect_expression_variables(expression, &mut body); + body.remove(accumulator); + body.remove(variable); + vars.extend(body); } LogicalExpression::PatternComprehension { projection, .. } => { collect_expression_variables(projection, vars); @@ -601,24 +678,144 @@ pub(crate) fn collect_operator_variables(op: &LogicalOperator, vars: &mut HashSe } } +/// The left join of an OPTIONAL MATCH: `right` matched for each row of +/// `left`, with nulls where it has no match. A filter in `right` that reads a +/// variable only `left` binds (GQL's `(c WHERE c.age > a.age)`) is a condition +/// of the join: it decides which matches count, so it moves there. +pub(crate) fn optional_join(left: LogicalOperator, right: LogicalOperator) -> LogicalOperator { + let mut left_vars = HashSet::new(); + collect_operator_variables(&left, &mut left_vars); + let mut right_vars = HashSet::new(); + collect_operator_variables(&right, &mut right_vars); + let mut moved = Vec::new(); + let right = take_left_reading_filters(right, &left_vars, &right_vars, &mut moved); + LogicalOperator::LeftJoin(LeftJoinOp { + left: Box::new(left), + right: Box::new(right), + condition: join_conjuncts(moved), + }) +} + +/// Removes from the filters of `plan` (down its pattern) the conjuncts that +/// read a variable `left_vars` has and `right_vars` does not, adding them to +/// `moved`. +fn take_left_reading_filters( + plan: LogicalOperator, + left_vars: &HashSet, + right_vars: &HashSet, + moved: &mut Vec, +) -> LogicalOperator { + match plan { + LogicalOperator::Filter(mut filter) => { + let input = take_left_reading_filters(*filter.input, left_vars, right_vars, moved); + let mut kept = Vec::new(); + for conjunct in split_and(filter.predicate) { + let mut read = HashSet::new(); + collect_expression_variables(&conjunct, &mut read); + if read + .iter() + .any(|name| left_vars.contains(name) && !right_vars.contains(name)) + { + moved.push(conjunct); + } else { + kept.push(conjunct); + } + } + match join_conjuncts(kept) { + Some(predicate) => { + filter.predicate = predicate; + filter.input = Box::new(input); + LogicalOperator::Filter(filter) + } + None => input, + } + } + LogicalOperator::Expand(mut expand) => { + expand.input = Box::new(take_left_reading_filters( + *expand.input, + left_vars, + right_vars, + moved, + )); + LogicalOperator::Expand(expand) + } + LogicalOperator::NodeScan(mut scan) => { + scan.input = scan.input.map(|input| { + Box::new(take_left_reading_filters( + *input, left_vars, right_vars, moved, + )) + }); + LogicalOperator::NodeScan(scan) + } + // The patterns of a comma list each have their filters. + LogicalOperator::Join(mut join) => { + join.left = Box::new(take_left_reading_filters( + *join.left, left_vars, right_vars, moved, + )); + join.right = Box::new(take_left_reading_filters( + *join.right, + left_vars, + right_vars, + moved, + )); + LogicalOperator::Join(join) + } + other => other, + } +} + +/// The conjuncts of `predicate` (`a AND b AND c` gives three). +fn split_and(predicate: LogicalExpression) -> Vec { + match predicate { + LogicalExpression::Binary { + left, + op: BinaryOp::And, + right, + } => { + let mut conjuncts = split_and(*left); + conjuncts.extend(split_and(*right)); + conjuncts + } + other => vec![other], + } +} + +/// The conjunction of `conjuncts`, `None` for none. +fn join_conjuncts(conjuncts: Vec) -> Option { + conjuncts + .into_iter() + .reduce(|acc, conjunct| LogicalExpression::Binary { + left: Box::new(acc), + op: BinaryOp::And, + right: Box::new(conjunct), + }) +} + /// Builds a LeftJoin with properly classified WHERE predicates. /// -/// Given a WHERE predicate that follows an OPTIONAL MATCH, this function: +/// Given a WHERE predicate that follows an OPTIONAL MATCH (whose join is +/// `left_join`), this function: /// 1. Collects variables from both sides /// 2. Classifies predicates into left-only, right-only, and cross-side /// 3. Pushes right-only predicates as a Filter on the right input -/// 4. Stores cross-side predicates in `LeftJoinOp.condition` +/// 4. Adds cross-side predicates to `LeftJoinOp.condition`, which decides +/// which pairs of rows are matches (a left row without one keeps nulls) /// 5. Returns the LeftJoin and any remaining post-filters to apply above pub(crate) fn build_left_join_with_predicates( - left: LogicalOperator, - right: LogicalOperator, + left_join: LeftJoinOp, predicate: Option, ) -> (LogicalOperator, Option) { + let LeftJoinOp { + left, + right, + condition, + } = left_join; + let (left, right) = (*left, *right); let Some(predicate) = predicate else { let join = LogicalOperator::LeftJoin(LeftJoinOp { left: Box::new(left), right: Box::new(right), - condition: None, + condition, }); return (join, None); }; @@ -648,61 +845,14 @@ pub(crate) fn build_left_join_with_predicates( wrap_filter(right, right_pred) }; - // Build null-safe condition for cross-side predicates. - // For each cross predicate P referencing right-only variable R, wrap it as: - // (R IS NULL) OR P - // This preserves NULL-padded rows (unmatched optional side) while evaluating P - // correctly when the right side matched. - let cross_condition = if classified.cross_filters.is_empty() { - None - } else { - // Collect all right-only variable names for the IS NULL sentinel. - let right_only_vars: Vec = right_vars - .iter() - .filter(|v| !left_vars.contains(*v)) - .cloned() - .collect(); - - let null_safe: Vec = classified - .cross_filters + // The cross-side predicates join the condition the join had: together + // they decide which pairs of rows are matches. + let cross_condition = join_conjuncts( + condition .into_iter() - .map(|pred| { - // Pick the first right-only variable referenced in this predicate - // as the NULL sentinel. Falling back to the first right-only var - // overall is safe: if the right side produced no row, all right - // columns are NULL so any of them serves as the sentinel. - let mut pred_vars = HashSet::new(); - collect_expression_variables(&pred, &mut pred_vars); - let sentinel = pred_vars - .iter() - .find(|v| right_vars.contains(*v) && !left_vars.contains(*v)) - .or_else(|| right_only_vars.first()) - .cloned() - .unwrap_or_default(); - - let is_null = LogicalExpression::Unary { - op: UnaryOp::IsNull, - operand: Box::new(LogicalExpression::Variable(sentinel)), - }; - LogicalExpression::Binary { - left: Box::new(is_null), - op: BinaryOp::Or, - right: Box::new(pred), - } - }) - .collect(); - - Some( - null_safe - .into_iter() - .reduce(|acc, expr| LogicalExpression::Binary { - left: Box::new(acc), - op: BinaryOp::And, - right: Box::new(expr), - }) - .expect("non-empty cross_filters"), - ) - }; + .chain(classified.cross_filters) + .collect(), + ); let join = LogicalOperator::LeftJoin(LeftJoinOp { left: Box::new(left), diff --git a/crates/grafeo-engine/src/query/translators/cypher.rs b/crates/grafeo-engine/src/query/translators/cypher.rs index f78eec2a3..1be079e65 100644 --- a/crates/grafeo-engine/src/query/translators/cypher.rs +++ b/crates/grafeo-engine/src/query/translators/cypher.rs @@ -4,14 +4,15 @@ //! that can be optimized and executed. use super::common::{ - build_left_join_with_predicates, check_branch_columns, combine_with_and, has_all_labels, - is_aggregate_function, to_aggregate_function, wrap_distinct, wrap_filter, wrap_limit, - wrap_return, wrap_skip, wrap_sort, + build_left_join_with_predicates, check_branch_columns, collect_expression_variables, + combine_with_and, expand_subquery_return_star, has_all_labels, is_aggregate_function, + optional_join, to_aggregate_function, wrap_distinct, wrap_filter, wrap_limit, wrap_return, + wrap_skip, wrap_sort, }; use crate::query::plan::{ AddLabelOp, AggregateExpr, AggregateFunction, AggregateOp, ApplyOp, BinaryOp, CallProcedureOp, CountExpr, CreateEdgeOp, CreateNodeOp, DeleteEdgeOp, DeleteNodeOp, ExpandDirection, ExpandOp, - JoinCondition, JoinOp, JoinType, LeftJoinOp, ListPredicateKind, LoadDataFormat, LoadDataOp, + JoinCondition, JoinOp, JoinType, ListPredicateKind, LoadDataFormat, LoadDataOp, LogicalExpression, LogicalOperator, LogicalPlan, MapProjectionEntry, MergeOp, MergeRelationshipOp, NodeScanOp, ParameterScanOp, PathMode, ProcedureYield, ProjectOp, Projection, RemoveLabelOp, ReturnItem, SetPropertyOp, ShortestPathOp, SortKey, SortOrder, @@ -181,6 +182,22 @@ impl CypherTranslator { } fn translate_query(&self, query: &ast::Query) -> Result { + // As in Neo4j, the rows of a CALL subquery that returns some are not + // the result of a query: a RETURN after it says what is. + if let Some(ast::Clause::CallSubquery { query: inner, .. }) = query.clauses.last() + && inner + .clauses + .iter() + .any(|clause| matches!(clause, ast::Clause::Return(_))) + { + return Err(Error::Query(QueryError::new( + QueryErrorKind::Semantic, + concat!( + "Query cannot conclude with CALL (must be a RETURN clause, an update clause, ", + "a unit subquery call, or a procedure call with no YIELD)" + ), + ))); + } let mut plan: Option = None; for clause in &query.clauses { @@ -218,9 +235,12 @@ impl CypherTranslator { ast::Clause::Set(set_clause) => self.translate_set(set_clause, input), ast::Clause::Remove(remove_clause) => self.translate_remove(remove_clause, input), ast::Clause::Call(call) => self.translate_call_clause(call, input), - ast::Clause::CallSubquery(inner_query) => { - self.translate_call_subquery(inner_query, input) - } + ast::Clause::CallSubquery { + query, + scope, + unions, + union_all, + } => self.translate_call_subquery(query, unions, *union_all, scope.as_deref(), input), ast::Clause::ForEach(foreach) => self.translate_foreach(foreach, input), ast::Clause::LoadCsv(load_csv) => self.translate_load_csv(load_csv), } @@ -266,65 +286,181 @@ impl CypherTranslator { /// Translates `CALL { subquery }` to an Apply operator. /// - /// When the inner subquery starts with `WITH ` and there is an outer - /// input, the WITH items are treated as variable imports from the outer scope. - /// The imported variable names are recorded in `ApplyOp.shared_variables` so - /// the planner can wire them through `ParameterState`. + /// The subquery sees the outer variables its variable scope clause names + /// (`CALL (a, b) { ... }`, all of them for `(*)`, none for `()`), or + /// without one, the variables its importing `WITH` names. It starts from a + /// `ParameterScan` of them, and they are recorded in + /// `ApplyOp.shared_variables` so the planner can wire them through + /// `ParameterState`. Parts joined by `UNION` each import their own; the + /// Apply imports all of them, and each part's scan names its own. fn translate_call_subquery( &self, inner: &ast::Query, + unions: &[ast::Query], + union_all: bool, + scope: Option<&[String]>, input: Option, ) -> Result { - // Detect importing WITH: if the first clause is WITH and we have outer input, - // extract the imported variable names and start the inner plan from a - // ParameterScan instead of Empty. + let outer_names = match &input { + Some(outer) => outer.bound_variables(None), + None => Some(HashSet::new()), + }; + let mut shared_variables: Vec = Vec::new(); + let mut parts = Vec::with_capacity(1 + unions.len()); + for part in std::iter::once(inner).chain(unions) { + let (plan, imported) = self.translate_call_subquery_part( + part, + scope, + input.is_some(), + outer_names.as_ref(), + )?; + for name in imported { + if !shared_variables.contains(&name) { + shared_variables.push(name); + } + } + parts.push(plan); + } + if shared_variables.iter().any(|name| name == "*") { + shared_variables = vec!["*".to_string()]; + } + let subplan = if parts.len() == 1 { + parts.remove(0) + } else { + check_branch_columns("UNION", &parts)?; + let union = LogicalOperator::Union(UnionOp { inputs: parts }); + if union_all { + union + } else { + wrap_distinct(union) + } + }; + + // A CALL that comes first runs once, on one empty row. + Ok(LogicalOperator::Apply(ApplyOp { + input: Box::new(input.unwrap_or(LogicalOperator::Empty)), + subplan: Box::new(subplan), + shared_variables, + optional: false, + })) + } + + /// Translates one part of a `CALL` subquery (the whole body, or one side + /// of a `UNION` in it): its plan and the outer variables it imports. + fn translate_call_subquery_part( + &self, + inner: &ast::Query, + scope: Option<&[String]>, + has_input: bool, + outer_names: Option<&HashSet>, + ) -> Result<(LogicalOperator, Vec)> { let mut shared_variables = Vec::new(); - let mut inner_plan: Option = None; let mut clauses_iter = inner.clauses.iter(); - if input.is_some() - && let Some(ast::Clause::With(with_clause)) = inner.clauses.first() - { - if with_clause.is_wildcard { - // WITH * imports all outer variables - shared_variables.push("*".to_string()); - } else { - for item in &with_clause.items { - if let ast::Expression::Variable(name) = &item.expression { - let var_name = item.alias.as_deref().unwrap_or(name); - shared_variables.push(var_name.to_string()); - } + match scope { + // After a scope clause, a WITH is an ordinary WITH. With no outer + // row, `(*)` imports nothing, and named variables are reported as + // undefined by the binder. + Some(names) => { + if has_input || names.iter().any(|name| name != "*") { + shared_variables = names.to_vec(); } } - if !shared_variables.is_empty() { - // Skip the importing WITH and start from a ParameterScan - clauses_iter.next(); - inner_plan = Some(LogicalOperator::ParameterScan(ParameterScanOp { - columns: shared_variables.clone(), - })); + // Without one, an importing WITH names what the subquery sees and + // is replaced by the ParameterScan. + None => { + if has_input + && let Some(ast::Clause::With(with_clause)) = inner.clauses.first() + && let Some(imported) = + self.importing_with(with_clause, inner.clauses.get(1))? + { + shared_variables = imported; + clauses_iter.next(); + } } } + let mut inner_plan = (!shared_variables.is_empty()).then(|| { + LogicalOperator::ParameterScan(ParameterScanOp { + columns: shared_variables.clone(), + }) + }); // Translate the remaining inner subquery clauses for clause in clauses_iter { inner_plan = Some(self.translate_clause(clause, inner_plan)?); } - let inner_plan = inner_plan.ok_or_else(|| { + let mut inner_plan = inner_plan.ok_or_else(|| { Error::Query(QueryError::new( QueryErrorKind::Semantic, "CALL subquery requires at least one clause", )) })?; + expand_subquery_return_star(&mut inner_plan, outer_names)?; + Ok((inner_plan, shared_variables)) + } - match input { - Some(outer) => Ok(LogicalOperator::Apply(ApplyOp { - input: Box::new(outer), - subplan: Box::new(inner_plan), - shared_variables, - optional: false, - })), - None => Ok(inner_plan), + /// Reads the first `WITH` of a `CALL` subquery: the outer variables it + /// imports (`*` for `WITH *`), or `None` when it names no variable and is + /// an ordinary `WITH`. As in openCypher, an importing `WITH` only lists + /// variables; an alias, an expression, `WHERE`, `DISTINCT`, or an + /// `ORDER BY`, `SKIP` or `LIMIT` after it (the `next` clause) is an error + /// (a second `WITH` can do those). + fn importing_with( + &self, + with_clause: &ast::WithClause, + next: Option<&ast::Clause>, + ) -> Result>> { + let mut imported = Vec::new(); + let mut names_variables = with_clause.is_wildcard; + let mut only_names = true; + if with_clause.is_wildcard { + imported.push("*".to_string()); + } + for item in &with_clause.items { + match &item.expression { + ast::Expression::Variable(name) + if item.alias.as_ref().is_none_or(|alias| alias == name) => + { + imported.push(name.clone()); + names_variables = true; + } + expression => { + only_names = false; + let mut variables = HashSet::new(); + collect_expression_variables( + &self.translate_expression(expression)?, + &mut variables, + ); + names_variables |= !variables.is_empty(); + } + } + } + if !names_variables { + return Ok(None); } + let not_allowed = if !only_names { + Some("Aliasing or expressions are not supported.") + } else if with_clause.where_clause.is_some() { + Some("WHERE is not allowed.") + } else if with_clause.distinct { + Some("DISTINCT is not allowed.") + } else { + match next { + Some(ast::Clause::OrderBy(_)) => Some("ORDER BY is not allowed."), + Some(ast::Clause::Skip(_)) => Some("SKIP is not allowed."), + Some(ast::Clause::Limit(_)) => Some("LIMIT is not allowed."), + _ => None, + } + }; + if let Some(reason) = not_allowed { + const IMPORTING_WITH: &str = + "Importing WITH should consist only of simple references to outside variables."; + return Err(Error::Query(QueryError::new( + QueryErrorKind::Semantic, + format!("{IMPORTING_WITH} {reason}"), + ))); + } + Ok(Some(imported)) } /// Translates `FOREACH (var IN list | clauses)` to Unwind + mutation pipeline. @@ -465,22 +601,14 @@ impl CypherTranslator { match_clause: &ast::MatchClause, input: Option, ) -> Result { - // OPTIONAL MATCH uses LEFT JOIN semantics - let input = input.ok_or_else(|| { - Error::Query(QueryError::new( - QueryErrorKind::Semantic, - "OPTIONAL MATCH requires input", - )) - })?; + // OPTIONAL MATCH uses LEFT JOIN semantics; one that comes first + // joins one empty row, so no match is one row of nulls. + let input = input.unwrap_or(LogicalOperator::Empty); // Build the right side with proper shared variable joins let right = self.translate_comma_patterns(&match_clause.patterns, None)?; - Ok(LogicalOperator::LeftJoin(LeftJoinOp { - left: Box::new(input), - right: Box::new(right), - condition: None, - })) + Ok(optional_join(input, right)) } fn translate_pattern( @@ -800,6 +928,7 @@ impl CypherTranslator { }; let expand = LogicalOperator::Expand(ExpandOp { + quantified: rel.length.is_some(), from_variable, to_variable: expand_target.clone(), edge_variable, @@ -891,8 +1020,7 @@ impl CypherTranslator { // predicate so right-side references become join conditions rather // than post-filters (which would incorrectly eliminate NULL rows). if let LogicalOperator::LeftJoin(left_join) = input { - let (join, post_filter) = - build_left_join_with_predicates(*left_join.left, *left_join.right, Some(predicate)); + let (join, post_filter) = build_left_join_with_predicates(left_join, Some(predicate)); if let Some(pf) = post_filter { Ok(wrap_filter(join, pf)) } else { @@ -928,6 +1056,12 @@ impl CypherTranslator { return Ok(plan); } + if with_clause.items.iter().any(|item| { + item.alias.is_none() && !matches!(item.expression, ast::Expression::Variable(_)) + }) { + return Err(super::common::unaliased_with_expression()); + } + // Check if WITH contains aggregate functions (e.g. WITH collect(n) AS people) let has_aggregates = with_clause .items @@ -935,8 +1069,10 @@ impl CypherTranslator { .any(|item| contains_aggregate(&item.expression)); let mut plan = if has_aggregates { - let (aggregates, group_by, post_return) = + let (mut aggregates, mut group_by, post_return) = self.extract_aggregates_and_groups_from_items(&with_clause.items)?; + let input = + self.lift_aggregate_pattern_comprehensions(input, &mut aggregates, &mut group_by)?; let agg_op = LogicalOperator::Aggregate(AggregateOp { group_by, @@ -973,7 +1109,13 @@ impl CypherTranslator { }) .collect::>()?; - // Rewrite pattern comprehensions into Apply + Aggregate(Collect) + // Rewrite pattern comprehensions into Apply + Aggregate(Collect): + // the ones inside an expression here, the item ones below. + let mut projections = projections; + let input = self.lift_nested_pattern_comprehensions( + input, + projections.iter_mut().map(|p| &mut p.expression), + )?; let has_pattern_comp = projections.iter().any(|p| { matches!( &p.expression, @@ -1323,8 +1465,10 @@ impl CypherTranslator { }; // With aliases (e.g. `n.city AS city`) the post-Return renames the // columns, which ORDER BY alias resolution and result naming need. - let (aggregates, group_by, post_return) = + let (mut aggregates, mut group_by, post_return) = self.extract_aggregates_and_groups_from_items(items)?; + let input = + self.lift_aggregate_pattern_comprehensions(input, &mut aggregates, &mut group_by)?; // Register aggregate output column names so ORDER BY can // reference them. Group-by columns use expression_to_string @@ -1386,7 +1530,13 @@ impl CypherTranslator { .collect::>()?, }; - // Rewrite pattern comprehensions into Apply + Aggregate(Collect) + // Rewrite pattern comprehensions into Apply + Aggregate(Collect): + // the ones inside an expression here, the item ones below. + let mut items = items; + let input = self.lift_nested_pattern_comprehensions( + input, + items.iter_mut().map(|item| &mut item.expression), + )?; let has_pattern_comp = items.iter().any(|item| { matches!( &item.expression, @@ -2299,7 +2449,8 @@ impl CypherTranslator { } } - /// Translates the inner query of an EXISTS subquery to a `LogicalOperator`. + /// Translates the inner query of an EXISTS or COUNT subquery to a + /// `LogicalOperator`. fn translate_exists_subquery(&self, query: &ast::Query) -> Result { let mut plan: Option = None; @@ -2308,13 +2459,16 @@ impl CypherTranslator { ast::Clause::Match(m) => { plan = Some(self.translate_match(m, plan)?); } + ast::Clause::OptionalMatch(m) => { + plan = Some(self.translate_optional_match(m, plan)?); + } ast::Clause::Where(w) => { plan = Some(self.translate_where(w, plan)?); } _ => { return Err(Error::Query(QueryError::new( QueryErrorKind::Semantic, - "EXISTS subquery only supports MATCH and WHERE clauses", + "EXISTS and COUNT subqueries only support MATCH, OPTIONAL MATCH and WHERE clauses", ))); } } @@ -2474,6 +2628,138 @@ impl CypherTranslator { } } + /// Rewrites the pattern comprehensions in the arguments and group keys of + /// an aggregation into `Apply`s over its input (see + /// [`rewrite_pattern_comprehensions`](Self::rewrite_pattern_comprehensions)), + /// so `sum(size([(b)-->(c) | c]))` aggregates their lists. + fn lift_aggregate_pattern_comprehensions( + &self, + input: LogicalOperator, + aggregates: &mut [AggregateExpr], + group_by: &mut [LogicalExpression], + ) -> Result { + let expressions = aggregates + .iter_mut() + .flat_map(|aggregate| { + [&mut aggregate.expression, &mut aggregate.expression2] + .into_iter() + .flatten() + }) + .chain(group_by.iter_mut()); + let mut lifted = Vec::new(); + for expression in expressions { + self.take_pattern_comprehensions(expression, &mut lifted); + } + if lifted.is_empty() { + return Ok(input); + } + Ok(self.rewrite_pattern_comprehensions(input, lifted)?.0) + } + + /// Rewrites the pattern comprehensions nested inside `expressions` (not + /// one that is a whole expression, which the item rewrite handles) into + /// `Apply`s over `input`. + fn lift_nested_pattern_comprehensions<'e>( + &self, + input: LogicalOperator, + expressions: impl Iterator, + ) -> Result { + let mut lifted = Vec::new(); + for expression in expressions { + if !matches!(expression, LogicalExpression::PatternComprehension { .. }) { + self.take_pattern_comprehensions(expression, &mut lifted); + } + } + if lifted.is_empty() { + return Ok(input); + } + Ok(self.rewrite_pattern_comprehensions(input, lifted)?.0) + } + + /// Replaces each pattern comprehension in `expression` with a variable of + /// its own and adds it to `lifted` as an item that collects into that + /// variable. Comprehension and predicate bodies are left alone: they can + /// read their own iteration variable. + fn take_pattern_comprehensions( + &self, + expression: &mut LogicalExpression, + lifted: &mut Vec, + ) { + match expression { + LogicalExpression::PatternComprehension { .. } => { + let alias = self.next_anon_var(); + let comprehension = + std::mem::replace(expression, LogicalExpression::Variable(alias.clone())); + lifted.push(ReturnItem { + expression: comprehension, + alias: Some(alias), + }); + } + LogicalExpression::Binary { left, right, .. } => { + self.take_pattern_comprehensions(left, lifted); + self.take_pattern_comprehensions(right, lifted); + } + LogicalExpression::Unary { operand, .. } => { + self.take_pattern_comprehensions(operand, lifted); + } + LogicalExpression::FunctionCall { args, .. } | LogicalExpression::List(args) => { + for arg in args { + self.take_pattern_comprehensions(arg, lifted); + } + } + LogicalExpression::Map(entries) => { + for (_, value) in entries { + self.take_pattern_comprehensions(value, lifted); + } + } + LogicalExpression::IndexAccess { base, index } => { + self.take_pattern_comprehensions(base, lifted); + self.take_pattern_comprehensions(index, lifted); + } + LogicalExpression::MapAccess { base, .. } => { + self.take_pattern_comprehensions(base, lifted); + } + LogicalExpression::SliceAccess { base, start, end } => { + self.take_pattern_comprehensions(base, lifted); + for bound in [start, end].into_iter().flatten() { + self.take_pattern_comprehensions(bound, lifted); + } + } + LogicalExpression::Case { + operand, + when_clauses, + else_clause, + } => { + if let Some(operand) = operand { + self.take_pattern_comprehensions(operand, lifted); + } + for (condition, result) in when_clauses { + self.take_pattern_comprehensions(condition, lifted); + self.take_pattern_comprehensions(result, lifted); + } + if let Some(else_clause) = else_clause { + self.take_pattern_comprehensions(else_clause, lifted); + } + } + LogicalExpression::ListComprehension { list_expr, .. } + | LogicalExpression::ListPredicate { list_expr, .. } => { + self.take_pattern_comprehensions(list_expr, lifted); + } + LogicalExpression::Reduce { initial, list, .. } => { + self.take_pattern_comprehensions(initial, lifted); + self.take_pattern_comprehensions(list, lifted); + } + LogicalExpression::MapProjection { entries, .. } => { + for entry in entries { + if let MapProjectionEntry::LiteralEntry(_, value) = entry { + self.take_pattern_comprehensions(value, lifted); + } + } + } + _ => {} + } + } + /// Rewrites pattern comprehensions in return items into Apply + Aggregate. /// /// For each `PatternComprehension` found in the items: diff --git a/crates/grafeo-engine/src/query/translators/gql/expression.rs b/crates/grafeo-engine/src/query/translators/gql/expression.rs index 5a1c93f03..c84e2c420 100644 --- a/crates/grafeo-engine/src/query/translators/gql/expression.rs +++ b/crates/grafeo-engine/src/query/translators/gql/expression.rs @@ -178,11 +178,34 @@ impl GqlTranslator { } ast::Expression::ValueSubquery { query } => { // VALUE { subquery } returns a scalar from the inner query. - // If the inner RETURN is a count() aggregate over an edge pattern, - // use CountSubquery (optimized path that handles correlation). - // Otherwise, translate the full query and use ValueSubquery + Apply. - if Self::is_count_aggregate_return(&query.return_clause) { - let inner_plan = self.translate_subquery_to_operator(query)?; + // A count() return becomes a CountSubquery over the rows it + // counts: those where the argument is not null, one per value + // with DISTINCT. Otherwise, translate the full query and use + // ValueSubquery + Apply. + if let Some((argument, distinct)) = + Self::count_aggregate_return(&query.return_clause) + { + let mut inner_plan = self.translate_subquery_to_operator(query)?; + if let Some(argument) = argument { + let argument = self.translate_expression(argument)?; + inner_plan = wrap_filter( + inner_plan, + LogicalExpression::Unary { + op: UnaryOp::IsNotNull, + operand: Box::new(argument.clone()), + }, + ); + if distinct { + inner_plan = wrap_distinct(LogicalOperator::Project(ProjectOp { + projections: vec![Projection { + expression: argument, + alias: Some("__counted".to_string()), + }], + input: Box::new(inner_plan), + pass_through_input: false, + })); + } + } Ok(LogicalExpression::CountSubquery(Box::new(inner_plan))) } else { let inner_logical_plan = self.translate_query(query)?; diff --git a/crates/grafeo-engine/src/query/translators/gql/mod.rs b/crates/grafeo-engine/src/query/translators/gql/mod.rs index 4525ef94d..6084c3d11 100644 --- a/crates/grafeo-engine/src/query/translators/gql/mod.rs +++ b/crates/grafeo-engine/src/query/translators/gql/mod.rs @@ -9,8 +9,9 @@ mod pattern; use std::collections::{HashMap, HashSet}; use super::common::{ - build_left_join_with_predicates, check_branch_columns, combine_with_and, flatten_and_conjuncts, - has_all_labels, is_aggregate_function, is_binary_set_function, join_and_conjuncts, + build_left_join_with_predicates, check_branch_columns, collect_expression_variables, + combine_with_and, expand_subquery_return_star, flatten_and_conjuncts, has_all_labels, + is_aggregate_function, is_binary_set_function, join_and_conjuncts, optional_join, references_any, to_aggregate_function, wrap_distinct, wrap_filter, wrap_limit, wrap_return, wrap_skip, wrap_sort, }; @@ -20,8 +21,8 @@ use crate::query::plan::{ ExpandDirection, ExpandOp, HorizontalAggregateOp, IntersectOp, JoinCondition, JoinOp, JoinType, LeftJoinOp, LoadDataFormat, LoadDataOp, LogicalExpression, LogicalOperator, LogicalPlan, MergeOp, MergeRelationshipOp, NodeScanOp, NullsOrdering, OtherwiseOp, ParameterScanOp, - PathMode, ProcedureYield, ProjectOp, Projection, RemoveLabelOp, ReturnItem, SetPropertyOp, - ShortestPathOp, SortKey, SortOrder, UnaryOp, UnionOp, UnwindOp, + PathMode, ProcedureYield, ProjectOp, Projection, RemoveLabelOp, ReturnItem, ReturnOp, + SetPropertyOp, ShortestPathOp, SortKey, SortOrder, UnaryOp, UnionOp, UnwindOp, }; #[cfg(test)] use crate::query::plan::{FilterOp, LimitOp, SkipOp}; @@ -79,12 +80,199 @@ struct GqlTranslator { /// Edge variables from variable-length expand patterns (group-list variables). /// Maps edge variable name to the path alias used for `_path_edges_{alias}` lookup. group_list_variables: std::cell::RefCell>, + /// The variables of the row the `CALL` subquery being translated runs + /// for (`None` outside one, or when they are not known): what a nested + /// subquery's `RETURN *` leaves out. + call_scope: std::cell::RefCell>>, +} + +/// The rows a query passes to the one after `NEXT`: its final `RETURN` as a +/// `WITH` (a projection), under its `ORDER BY`, `SKIP` and `LIMIT`. An +/// unaliased item is a column named after it; `RETURN *` passes every column +/// on. Without `DISTINCT` the rows are ordered before the projection, so an +/// `ORDER BY` key can read what the `RETURN` leaves out (and reads an alias +/// through its expression). A plan that ends otherwise (an aggregation) +/// passes its rows on as they are. +fn return_as_with(plan: LogicalOperator) -> LogicalOperator { + match plan { + LogicalOperator::Return(ret) => { + if ret.items.iter().any( + |item| matches!(&item.expression, LogicalExpression::Variable(name) if name == "*"), + ) { + return if ret.distinct { + wrap_distinct(*ret.input) + } else { + *ret.input + }; + } + let projections = ret + .items + .into_iter() + .map(|item| { + let alias = match (&item.alias, &item.expression) { + (Some(alias), _) => Some(alias.clone()), + (None, LogicalExpression::Variable(_)) => None, + (None, expression) => Some( + crate::query::planner::common::expression_to_string(expression), + ), + }; + Projection { + expression: item.expression, + alias, + } + }) + .collect(); + let project = LogicalOperator::Project(ProjectOp { + projections, + input: ret.input, + pass_through_input: false, + }); + if ret.distinct { + wrap_distinct(project) + } else { + project + } + } + LogicalOperator::Sort(mut sort) => match *sort.input { + LogicalOperator::Return(ret) if !ret.distinct => { + match keys_before_return(&sort.keys, &ret) { + Some(keys) => { + sort.keys = keys; + sort.input = ret.input; + return_as_with(LogicalOperator::Return(ReturnOp { + input: Box::new(LogicalOperator::Sort(sort)), + ..ret + })) + } + // A key reads an alias that cannot be replaced: order the + // projected rows (the key then reads only what they hold). + None => { + sort.input = Box::new(return_as_with(LogicalOperator::Return(ret))); + LogicalOperator::Sort(sort) + } + } + } + input => { + sort.input = Box::new(return_as_with(input)); + LogicalOperator::Sort(sort) + } + }, + LogicalOperator::Skip(mut skip) => { + skip.input = Box::new(return_as_with(*skip.input)); + LogicalOperator::Skip(skip) + } + LogicalOperator::Limit(mut limit) => { + limit.input = Box::new(return_as_with(*limit.input)); + LogicalOperator::Limit(limit) + } + LogicalOperator::Distinct(mut distinct) => { + distinct.input = Box::new(return_as_with(*distinct.input)); + LogicalOperator::Distinct(distinct) + } + other => other, + } +} + +/// The sort `keys` of a `RETURN` rewritten to read the `RETURN`'s input: each +/// alias replaced by its expression (a property of an alias that is a +/// variable by that variable's property). `None` when a key still reads an +/// alias, such as one inside a `CASE`, or a property of an alias that is not +/// a variable. +fn keys_before_return(keys: &[SortKey], ret: &ReturnOp) -> Option> { + let aliases: Vec<(String, LogicalExpression)> = ret + .items + .iter() + .filter_map(|item| { + item.alias + .as_ref() + .map(|alias| (alias.clone(), item.expression.clone())) + }) + .collect(); + let input_names = ret.input.bound_variables(None); + keys.iter() + .map(|key| { + let expression = + GqlTranslator::substitute_let_bindings(key.expression.clone(), &aliases); + let mut read = HashSet::new(); + collect_expression_variables(&expression, &mut read); + let reads_an_alias = read.iter().any(|name| { + aliases.iter().any(|(alias, _)| alias == name) + && input_names + .as_ref() + .is_none_or(|names| !names.contains(name)) + }); + (!reads_an_alias).then(|| SortKey { + expression, + ..key.clone() + }) + }) + .collect() +} + +/// Combines two queries with a set operator, or with `NEXT` (the right one +/// runs for each row of the left one; a right side that is a query reads the +/// left one's rows instead, see `translate_composite_query`). +fn combine_queries( + op: ast::CompositeOp, + left: LogicalOperator, + right: LogicalOperator, +) -> Result { + Ok(match op { + ast::CompositeOp::Union | ast::CompositeOp::UnionAll => { + let inputs = vec![left, right]; + check_branch_columns("UNION", &inputs)?; + let union_op = LogicalOperator::Union(UnionOp { inputs }); + if op == ast::CompositeOp::UnionAll { + union_op + } else { + wrap_distinct(union_op) + } + } + ast::CompositeOp::Except | ast::CompositeOp::ExceptAll => { + let branches = [left, right]; + check_branch_columns("EXCEPT", &branches)?; + let [left, right] = branches; + LogicalOperator::Except(ExceptOp { + left: Box::new(left), + right: Box::new(right), + all: matches!(op, ast::CompositeOp::ExceptAll), + }) + } + ast::CompositeOp::Intersect | ast::CompositeOp::IntersectAll => { + let branches = [left, right]; + check_branch_columns("INTERSECT", &branches)?; + let [left, right] = branches; + LogicalOperator::Intersect(IntersectOp { + left: Box::new(left), + right: Box::new(right), + all: matches!(op, ast::CompositeOp::IntersectAll), + }) + } + ast::CompositeOp::Otherwise => { + let branches = [left, right]; + check_branch_columns("OTHERWISE", &branches)?; + let [left, right] = branches; + LogicalOperator::Otherwise(OtherwiseOp { + left: Box::new(left), + right: Box::new(right), + }) + } + // NEXT (linear composition): output of left feeds as input to right. + // Translate as Apply: for each row from left, execute right with bound variables. + ast::CompositeOp::Next => LogicalOperator::Apply(ApplyOp { + input: Box::new(left), + subplan: Box::new(right), + shared_variables: Vec::new(), + optional: false, + }), + }) } impl GqlTranslator { fn new() -> Self { Self { group_list_variables: std::cell::RefCell::new(HashMap::new()), + call_scope: std::cell::RefCell::new(None), } } @@ -141,64 +329,20 @@ impl GqlTranslator { right: &ast::Statement, ) -> Result { let left_plan = self.translate_statement(left)?; - let right_plan = self.translate_statement(right)?; - - match op { - ast::CompositeOp::Union | ast::CompositeOp::UnionAll => { - let inputs = vec![left_plan.root, right_plan.root]; - check_branch_columns("UNION", &inputs)?; - let union_op = LogicalOperator::Union(UnionOp { inputs }); - let root = if op == ast::CompositeOp::UnionAll { - union_op - } else { - wrap_distinct(union_op) - }; - Ok(LogicalPlan::new(root)) - } - ast::CompositeOp::Except | ast::CompositeOp::ExceptAll => { - let branches = [left_plan.root, right_plan.root]; - check_branch_columns("EXCEPT", &branches)?; - let [left, right] = branches; - let root = LogicalOperator::Except(ExceptOp { - left: Box::new(left), - right: Box::new(right), - all: matches!(op, ast::CompositeOp::ExceptAll), - }); - Ok(LogicalPlan::new(root)) - } - ast::CompositeOp::Intersect | ast::CompositeOp::IntersectAll => { - let branches = [left_plan.root, right_plan.root]; - check_branch_columns("INTERSECT", &branches)?; - let [left, right] = branches; - let root = LogicalOperator::Intersect(IntersectOp { - left: Box::new(left), - right: Box::new(right), - all: matches!(op, ast::CompositeOp::IntersectAll), - }); - Ok(LogicalPlan::new(root)) - } - ast::CompositeOp::Otherwise => { - let branches = [left_plan.root, right_plan.root]; - check_branch_columns("OTHERWISE", &branches)?; - let [left, right] = branches; - let root = LogicalOperator::Otherwise(OtherwiseOp { - left: Box::new(left), - right: Box::new(right), - }); - Ok(LogicalPlan::new(root)) - } - ast::CompositeOp::Next => { - // NEXT (linear composition): output of left feeds as input to right. - // Translate as Apply: for each row from left, execute right with bound variables. - let root = LogicalOperator::Apply(ApplyOp { - input: Box::new(left_plan.root), - subplan: Box::new(right_plan.root), - shared_variables: Vec::new(), - optional: false, - }); - Ok(LogicalPlan::new(root)) - } + // NEXT: the query after it reads the rows the one before returns, as + // a query reads the rows of a WITH. + if op == ast::CompositeOp::Next + && let ast::Statement::Query(right_query) = right + { + let input = return_as_with(left_plan.root); + return self.translate_query_from(right_query, input); } + let right_plan = self.translate_statement(right)?; + Ok(LogicalPlan::new(combine_queries( + op, + left_plan.root, + right_plan.root, + )?)) } fn translate_call(&self, call: &ast::CallStatement) -> Result { @@ -408,7 +552,17 @@ impl GqlTranslator { } fn translate_query(&self, query: &ast::QueryStatement) -> Result { - let mut plan = LogicalOperator::Empty; + self.translate_query_from(query, LogicalOperator::Empty) + } + + /// Translates `query` on the rows of `input`: `Empty` for a query of its + /// own, the outer row's variables for a `CALL` subquery. + fn translate_query_from( + &self, + query: &ast::QueryStatement, + input: LogicalOperator, + ) -> Result { + let mut plan = input; let mut where_applied = false; // Process clauses in source order for correct variable scoping. @@ -424,7 +578,9 @@ impl GqlTranslator { ast::QueryClause::Create(_) | ast::QueryClause::Delete(_) | ast::QueryClause::Set(_) + | ast::QueryClause::Remove(_) | ast::QueryClause::Merge(_) + | ast::QueryClause::With(_) ) { if let Some(where_clause) = &query.where_clause { @@ -441,22 +597,14 @@ impl GqlTranslator { // an implicit unit table so unmatched patterns produce // a single row of NULLs instead of zero rows. let match_plan = self.translate_match(match_clause)?; - plan = LogicalOperator::LeftJoin(LeftJoinOp { - left: Box::new(LogicalOperator::Empty), - right: Box::new(match_plan), - condition: None, - }); + plan = optional_join(LogicalOperator::Empty, match_plan); } else if matches!(plan, LogicalOperator::Empty) { // No prior input: standard MATCH plan = self.translate_match(match_clause)?; } else if match_clause.optional { // OPTIONAL MATCH: left join (prior vars on left, match on right) let match_plan = self.translate_match(match_clause)?; - plan = LogicalOperator::LeftJoin(LeftJoinOp { - left: Box::new(plan), - right: Box::new(match_plan), - condition: None, - }); + plan = optional_join(plan, match_plan); } else { // Non-optional MATCH after prior clauses (UNWIND, etc.) // Pass current plan as input so the MATCH's NodeScan creates @@ -550,8 +698,25 @@ impl GqlTranslator { pass_through_input: true, }); } - ast::QueryClause::InlineCall { subquery, optional } => { - plan = self.translate_inline_call(subquery, plan, *optional)?; + ast::QueryClause::Remove(remove_clause) => { + plan = Self::apply_remove(plan, remove_clause); + } + ast::QueryClause::With(with_clause) => { + plan = self.apply_with(plan, with_clause)?; + } + ast::QueryClause::InlineCall { + subquery, + combined, + optional, + scope, + } => { + plan = self.translate_inline_call( + subquery, + combined, + plan, + *optional, + scope.as_deref(), + )?; } ast::QueryClause::CallProcedure(call_stmt) => { // CALL procedure(...) within a query context @@ -588,19 +753,11 @@ impl GqlTranslator { for match_clause in &query.match_clauses { let match_plan = self.translate_match(match_clause)?; if matches!(plan, LogicalOperator::Empty) && match_clause.optional { - plan = LogicalOperator::LeftJoin(LeftJoinOp { - left: Box::new(LogicalOperator::Empty), - right: Box::new(match_plan), - condition: None, - }); + plan = optional_join(LogicalOperator::Empty, match_plan); } else if matches!(plan, LogicalOperator::Empty) { plan = match_plan; } else if match_clause.optional { - plan = LogicalOperator::LeftJoin(LeftJoinOp { - left: Box::new(plan), - right: Box::new(match_plan), - condition: None, - }); + plan = optional_join(plan, match_plan); } else { plan = LogicalOperator::Join(JoinOp { left: Box::new(plan), @@ -679,144 +836,27 @@ impl GqlTranslator { } } - // REMOVE clauses (not yet in ordered_clauses, always process) - for remove_clause in &query.remove_clauses { - for label_op in &remove_clause.label_operations { - plan = LogicalOperator::RemoveLabel(RemoveLabelOp { - variable: label_op.variable.clone(), - labels: label_op.labels.clone(), - input: Box::new(plan), - }); - } - for (variable, property) in &remove_clause.property_removals { - plan = LogicalOperator::SetProperty(SetPropertyOp { - variable: variable.clone(), - properties: vec![(property.clone(), LogicalExpression::Literal(Value::Null))], - replace: false, - is_edge: false, - input: Box::new(plan), - }); + // REMOVE clauses not among the ordered clauses (statements built + // without them) apply here, after the rest. + if !query + .ordered_clauses + .iter() + .any(|clause| matches!(clause, ast::QueryClause::Remove(_))) + { + for remove_clause in &query.remove_clauses { + plan = Self::apply_remove(plan, remove_clause); } } - // Handle WITH clauses (projection for query chaining) - for with_clause in &query.with_clauses { - if !with_clause.is_wildcard { - // Check if WITH contains aggregate functions (e.g. WITH count(n) AS cnt) - let has_aggregates = with_clause - .items - .iter() - .any(|item| contains_aggregate(&item.expression)); - - if has_aggregates { - let (aggregates, auto_group_by, post_return) = - self.extract_aggregates_and_groups(&with_clause.items, false)?; - - // Split the WHERE into HAVING (aggregate-referencing - // conjuncts) and a post-aggregate filter (the rest). - // This handles mixed predicates like - // `WHERE a.name = 'Alix' AND cnt > 2` correctly. - let aggregate_aliases: Vec = - aggregates.iter().filter_map(|a| a.alias.clone()).collect(); - let (having, post_agg_filter) = - if let Some(where_clause) = &with_clause.where_clause { - let pred = self.translate_expression(&where_clause.expression)?; - let conjuncts = flatten_and_conjuncts(&pred); - let (having_parts, filter_parts): (Vec<_>, Vec<_>) = conjuncts - .into_iter() - .partition(|c| references_any(c, &aggregate_aliases)); - ( - join_and_conjuncts(having_parts.into_iter().cloned().collect()), - join_and_conjuncts(filter_parts.into_iter().cloned().collect()), - ) - } else { - (None, None) - }; - - plan = LogicalOperator::Aggregate(AggregateOp { - group_by: auto_group_by, - aggregates, - input: Box::new(plan), - having, - }); - - // Apply post-aggregate projection if aggregates were wrapped - // in expressions (e.g. WITH count(n) + 1 AS cnt_plus_one) - if let Some(post_items) = post_return { - let post_projections: Vec = post_items - .into_iter() - .map(|item| Projection { - expression: item.expression, - alias: item.alias, - }) - .collect(); - plan = LogicalOperator::Project(ProjectOp { - projections: post_projections, - input: Box::new(plan), - pass_through_input: false, - }); - } - - // Apply non-aggregate WHERE conjuncts as a post-aggregate filter. - if let Some(filter_pred) = post_agg_filter { - plan = wrap_filter(plan, filter_pred); - } - } else { - let projections: Vec = with_clause - .items - .iter() - .map(|item| { - Ok(Projection { - expression: self.translate_expression(&item.expression)?, - alias: item.alias.clone(), - }) - }) - .collect::>()?; - - plan = LogicalOperator::Project(ProjectOp { - projections, - input: Box::new(plan), - pass_through_input: false, - }); - } - } - // WITH * skips projection: all variables pass through unchanged - - // Handle LET bindings attached to this WITH clause. - // LET adds new columns without replacing existing ones. - if !with_clause.let_bindings.is_empty() { - let mut let_projections = Vec::new(); - for (name, expr) in &with_clause.let_bindings { - let logical_expr = self.translate_expression(expr)?; - let_projections.push(Projection { - expression: logical_expr, - alias: Some(name.clone()), - }); - } - plan = LogicalOperator::Project(ProjectOp { - projections: let_projections, - input: Box::new(plan), - pass_through_input: true, - }); - } - - // Apply WHERE filter if present in WITH clause. - // For aggregate WITH clauses, the WHERE was already split into - // HAVING + post-aggregate filter above, so skip here. - if let Some(where_clause) = &with_clause.where_clause { - let has_agg = with_clause - .items - .iter() - .any(|item| contains_aggregate(&item.expression)); - if !has_agg { - let predicate = self.translate_expression(&where_clause.expression)?; - plan = wrap_filter(plan, predicate); - } - } - - // Handle DISTINCT - if with_clause.distinct { - plan = wrap_distinct(plan); + // WITH clauses not among the ordered clauses (statements built + // without them) apply here, after the rest. + if !query + .ordered_clauses + .iter() + .any(|clause| matches!(clause, ast::QueryClause::With(_))) + { + for with_clause in &query.with_clauses { + plan = self.apply_with(plan, with_clause)?; } } @@ -973,7 +1013,7 @@ impl GqlTranslator { } else { // Apply RETURN first (closest to input), then Sort wraps it. // This ensures RETURN aliases are visible to ORDER BY in the binder. - let mut return_items = if query.return_clause.is_wildcard { + let return_items = if query.return_clause.is_wildcard { // RETURN *: emit a wildcard marker that the planner expands vec![ReturnItem { expression: LogicalExpression::Variable("*".into()), @@ -993,27 +1033,6 @@ impl GqlTranslator { .collect::>>()? }; - // Lift VALUE subqueries: wrap with Apply and replace expression - // with a Variable reference to the inner plan's output column. - for item in &mut return_items { - if let LogicalExpression::ValueSubquery(inner_plan) = &item.expression { - // Determine the output column name from the inner plan's RETURN - let col_name = - Self::extract_return_column_name(inner_plan).unwrap_or_else(|| { - item.alias.clone().unwrap_or_else(|| "__value".to_string()) - }); - - plan = LogicalOperator::Apply(ApplyOp { - input: Box::new(plan), - subplan: inner_plan.clone(), - shared_variables: vec![], - optional: false, - }); - - item.expression = LogicalExpression::Variable(col_name); - } - } - plan = wrap_return(plan, return_items, query.return_clause.distinct); // Apply ORDER BY (wraps Return so aliases are visible) @@ -1081,6 +1100,18 @@ impl GqlTranslator { } expr } + // A property of a binding that is a variable reads that + // variable's property. + LogicalExpression::Property { + ref variable, + ref property, + } => match bindings.iter().find(|(name, _)| name == variable) { + Some((_, LogicalExpression::Variable(target))) => LogicalExpression::Property { + variable: target.clone(), + property: property.clone(), + }, + _ => expr, + }, LogicalExpression::Binary { left, op, right } => LogicalExpression::Binary { left: Box::new(Self::substitute_let_bindings(*left, bindings)), op, @@ -1313,173 +1344,217 @@ impl GqlTranslator { }) } - /// as imports from the outer scope: a `ParameterScan` replaces `Empty` as - /// the inner plan root and `shared_variables` is populated so the planner - /// can wire them through `ParameterState`. - fn translate_inline_call( + /// Applies a REMOVE clause to `plan`: its label removals, then its + /// property removals. + fn apply_remove( + mut plan: LogicalOperator, + remove_clause: &ast::RemoveClause, + ) -> LogicalOperator { + for label_op in &remove_clause.label_operations { + plan = LogicalOperator::RemoveLabel(RemoveLabelOp { + variable: label_op.variable.clone(), + labels: label_op.labels.clone(), + input: Box::new(plan), + }); + } + for (variable, property) in &remove_clause.property_removals { + plan = LogicalOperator::SetProperty(SetPropertyOp { + variable: variable.clone(), + properties: vec![(property.clone(), LogicalExpression::Literal(Value::Null))], + replace: false, + is_edge: false, + input: Box::new(plan), + }); + } + plan + } + + /// Applies a WITH clause to `plan`: its projection (or aggregation), the + /// LET bindings attached to it, its WHERE and DISTINCT. The clauses after + /// it read the rows it passes on. + fn apply_with( &self, - subquery: &ast::QueryStatement, - outer: LogicalOperator, - optional: bool, + mut plan: LogicalOperator, + with_clause: &ast::WithClause, ) -> Result { - let has_outer = !matches!(outer, LogicalOperator::Empty); - - // Detect importing WITH: extract shared variable names and skip it. - let mut shared_variables = Vec::new(); - let skip_with = if has_outer && !subquery.with_clauses.is_empty() { - let first_with = &subquery.with_clauses[0]; - if first_with.is_wildcard { - shared_variables.push("*".to_string()); - true - } else { - for item in &first_with.items { - if let ast::Expression::Variable(name) = &item.expression { - let var_name = item.alias.as_deref().unwrap_or(name); - shared_variables.push(var_name.to_string()); - } - } - !shared_variables.is_empty() + if !with_clause.is_wildcard { + if with_clause.items.iter().any(|item| { + item.alias.is_none() && !matches!(item.expression, ast::Expression::Variable(_)) + }) { + return Err(super::common::unaliased_with_expression()); } - } else { - false - }; - // Build the inner plan: start from ParameterScan when importing variables. - let inner_plan = if skip_with && !shared_variables.is_empty() { - // Translate the subquery but override the first WITH clause: - // start from ParameterScan instead of Empty, skip the importing WITH. - let mut plan = LogicalOperator::ParameterScan(ParameterScanOp { - columns: shared_variables.clone(), - }); + // Check if WITH contains aggregate functions (e.g. WITH count(n) AS cnt) + let has_aggregates = with_clause + .items + .iter() + .any(|item| contains_aggregate(&item.expression)); - // Process MATCH clauses - for match_clause in &subquery.match_clauses { - if match_clause.optional { - let match_plan = self.translate_match(match_clause)?; - plan = LogicalOperator::LeftJoin(LeftJoinOp { - left: Box::new(plan), - right: Box::new(match_plan), - condition: None, - }); - } else { - let input = std::mem::replace(&mut plan, LogicalOperator::Empty); - plan = self.translate_match_with_input(match_clause, Some(input))?; - } - } + if has_aggregates { + let (aggregates, auto_group_by, post_return) = + self.extract_aggregates_and_groups(&with_clause.items, false)?; + + // Split the WHERE into HAVING (aggregate-referencing + // conjuncts) and a post-aggregate filter (the rest). + // This handles mixed predicates like + // `WHERE a.name = 'Alix' AND cnt > 2` correctly. + let aggregate_aliases: Vec = + aggregates.iter().filter_map(|a| a.alias.clone()).collect(); + let (having, post_agg_filter) = + if let Some(where_clause) = &with_clause.where_clause { + let pred = self.translate_expression(&where_clause.expression)?; + let conjuncts = flatten_and_conjuncts(&pred); + let (having_parts, filter_parts): (Vec<_>, Vec<_>) = conjuncts + .into_iter() + .partition(|c| references_any(c, &aggregate_aliases)); + ( + join_and_conjuncts(having_parts.into_iter().cloned().collect()), + join_and_conjuncts(filter_parts.into_iter().cloned().collect()), + ) + } else { + (None, None) + }; - // Apply WHERE filter - if let Some(where_clause) = &subquery.where_clause { - let predicate = self.translate_expression(&where_clause.expression)?; - plan = wrap_filter(plan, predicate); - } + plan = LogicalOperator::Aggregate(AggregateOp { + group_by: auto_group_by, + aggregates, + input: Box::new(plan), + having, + }); - // Process remaining WITH clauses (skip the first importing one) - for with_clause in subquery.with_clauses.iter().skip(1) { - if !with_clause.is_wildcard { - let projections: Vec = with_clause - .items - .iter() - .map(|item| { - Ok(Projection { - expression: self.translate_expression(&item.expression)?, - alias: item.alias.clone(), - }) + // Apply post-aggregate projection if aggregates were wrapped + // in expressions (e.g. WITH count(n) + 1 AS cnt_plus_one) + if let Some(post_items) = post_return { + let post_projections: Vec = post_items + .into_iter() + .map(|item| Projection { + expression: item.expression, + alias: item.alias, }) - .collect::>()?; + .collect(); plan = LogicalOperator::Project(ProjectOp { - projections, + projections: post_projections, input: Box::new(plan), pass_through_input: false, }); } - // Handle LET bindings in inline call WITH clause - if !with_clause.let_bindings.is_empty() { - let mut let_projections = Vec::new(); - for (name, expr) in &with_clause.let_bindings { - let logical_expr = self.translate_expression(expr)?; - let_projections.push(Projection { - expression: logical_expr, - alias: Some(name.clone()), - }); - } - plan = LogicalOperator::Project(ProjectOp { - projections: let_projections, - input: Box::new(plan), - pass_through_input: true, - }); - } - if let Some(wc) = &with_clause.where_clause { - let predicate = self.translate_expression(&wc.expression)?; - plan = wrap_filter(plan, predicate); - } - } - - // Translate RETURN clause - let has_aggregates = !subquery.return_clause.is_wildcard - && subquery - .return_clause - .items - .iter() - .any(|item| contains_aggregate(&item.expression)); - if has_aggregates { - let (aggregates, auto_group_by, post_return) = self.extract_aggregates_and_groups( - &subquery.return_clause.items, - !subquery.return_clause.group_by.is_empty(), - )?; - let group_by = if subquery.return_clause.group_by.is_empty() { - auto_group_by - } else { - subquery - .return_clause - .group_by - .iter() - .map(|e| self.translate_expression(e)) - .collect::>>()? - }; - let agg_op = LogicalOperator::Aggregate(AggregateOp { - group_by, - aggregates, - input: Box::new(plan), - having: None, - }); - plan = if let Some(return_items) = post_return { - wrap_return(agg_op, return_items, subquery.return_clause.distinct) - } else { - agg_op - }; + // Apply non-aggregate WHERE conjuncts as a post-aggregate filter. + if let Some(filter_pred) = post_agg_filter { + plan = wrap_filter(plan, filter_pred); + } } else { - let return_items = subquery - .return_clause + let projections: Vec = with_clause .items .iter() .map(|item| { - Ok(ReturnItem { + Ok(Projection { expression: self.translate_expression(&item.expression)?, alias: item.alias.clone(), }) }) - .collect::>>()?; - plan = wrap_return(plan, return_items, subquery.return_clause.distinct); + .collect::>()?; + + plan = LogicalOperator::Project(ProjectOp { + projections, + input: Box::new(plan), + pass_through_input: false, + }); } - plan - } else { - // No importing WITH: translate the entire subquery independently - self.translate_query(subquery)?.root - }; + } + // WITH * skips projection: all variables pass through unchanged - // Wire the inner plan to the outer plan - if has_outer { - Ok(LogicalOperator::Apply(ApplyOp { - input: Box::new(outer), - subplan: Box::new(inner_plan), - shared_variables, - optional, - })) - } else { - // No outer input: just use the inner plan directly - Ok(inner_plan) + // Handle LET bindings attached to this WITH clause. + // LET adds new columns without replacing existing ones. + if !with_clause.let_bindings.is_empty() { + let mut let_projections = Vec::new(); + for (name, expr) in &with_clause.let_bindings { + let logical_expr = self.translate_expression(expr)?; + let_projections.push(Projection { + expression: logical_expr, + alias: Some(name.clone()), + }); + } + plan = LogicalOperator::Project(ProjectOp { + projections: let_projections, + input: Box::new(plan), + pass_through_input: true, + }); + } + + // Apply WHERE filter if present in WITH clause. + // For aggregate WITH clauses, the WHERE was already split into + // HAVING + post-aggregate filter above, so skip here. + if let Some(where_clause) = &with_clause.where_clause { + let has_agg = with_clause + .items + .iter() + .any(|item| contains_aggregate(&item.expression)); + if !has_agg { + let predicate = self.translate_expression(&where_clause.expression)?; + plan = wrap_filter(plan, predicate); + } } + + // Handle DISTINCT + if with_clause.distinct { + plan = wrap_distinct(plan); + } + Ok(plan) + } + + /// Translates `CALL { subquery }` to an `Apply` that runs the subquery for + /// each row of `outer`. As in GQL, the subquery sees the outer row's + /// variables: all of them, or the ones its variable scope clause names + /// (`CALL (a, b) { ... }`; none for `CALL () { ... }`). It starts from a + /// `ParameterScan` of them, which the planner fills for each row through + /// `ParameterState`, so a `WITH` in it is an ordinary `WITH`. + fn translate_inline_call( + &self, + subquery: &ast::QueryStatement, + combined: &[(ast::CompositeOp, ast::QueryStatement)], + outer: LogicalOperator, + optional: bool, + scope: Option<&[String]>, + ) -> Result { + // A CALL that comes first has no outer row to see: it runs once, on + // one empty row (`Empty`). A scope clause there names variables the + // binder reports as undefined. + let shared_variables: Vec = match scope { + Some(names) => names.to_vec(), + None if matches!(outer, LogicalOperator::Empty) => Vec::new(), + None => vec!["*".to_string()], + }; + let input = if shared_variables.is_empty() { + LogicalOperator::Empty + } else { + LogicalOperator::ParameterScan(ParameterScanOp { + columns: shared_variables.clone(), + }) + }; + // The outer row's variables: what a `RETURN *` of the subquery leaves + // out, and the scope of a CALL nested in it. + let outer_names = outer.bound_variables(self.call_scope.borrow().as_ref()); + let enclosing = self.call_scope.replace(outer_names.clone()); + // Each query combined in the body starts from the same outer row. + let translate_part = |part: &ast::QueryStatement| -> Result { + let mut plan = self.translate_query_from(part, input.clone())?.root; + expand_subquery_return_star(&mut plan, outer_names.as_ref())?; + Ok(plan) + }; + let inner = translate_part(subquery).and_then(|first| { + combined.iter().try_fold(first, |plan, (op, part)| { + combine_queries(*op, plan, translate_part(part)?) + }) + }); + self.call_scope.replace(enclosing); + let inner_plan = inner?; + Ok(LogicalOperator::Apply(ApplyOp { + input: Box::new(outer), + subplan: Box::new(inner_plan), + shared_variables, + optional, + })) } fn translate_match(&self, match_clause: &ast::MatchClause) -> Result { @@ -1497,8 +1572,7 @@ impl GqlTranslator { predicate: LogicalExpression, ) -> LogicalOperator { if let LogicalOperator::LeftJoin(left_join) = plan { - let (join, post_filter) = - build_left_join_with_predicates(*left_join.left, *left_join.right, Some(predicate)); + let (join, post_filter) = build_left_join_with_predicates(left_join, Some(predicate)); if let Some(pf) = post_filter { wrap_filter(join, pf) } else { @@ -1934,16 +2008,14 @@ impl GqlTranslator { let mut inner_defined = std::collections::HashSet::new(); for match_clause in &query.match_clauses { Self::collect_pattern_variables(&match_clause.patterns, &mut inner_defined); - let match_plan = self.translate_match(match_clause)?; - plan = if matches!(plan, LogicalOperator::Empty) { - match_plan + // Each MATCH goes on from the ones before it, as in the outer + // query, so a variable in two clauses is the same node or edge. + plan = if match_clause.optional { + optional_join(plan, self.translate_match(match_clause)?) + } else if matches!(plan, LogicalOperator::Empty) { + self.translate_match(match_clause)? } else { - LogicalOperator::Join(JoinOp { - left: Box::new(plan), - right: Box::new(match_plan), - join_type: JoinType::Cross, - conditions: vec![], - }) + self.translate_match_with_input(match_clause, Some(plan))? }; } @@ -1972,39 +2044,21 @@ impl GqlTranslator { Ok(plan) } - /// Returns true if the RETURN clause is a single count() aggregate. - fn is_count_aggregate_return(ret: &ast::ReturnClause) -> bool { - if ret.items.len() != 1 { - return false; - } - matches!( - &ret.items[0].expression, - ast::Expression::FunctionCall { name, .. } if name.eq_ignore_ascii_case("count") - ) - } - - /// Extracts the first output column name from a Return operator in a logical plan. - fn extract_return_column_name(plan: &LogicalOperator) -> Option { - match plan { - LogicalOperator::Return(ret) => { - let item = ret.items.first()?; - if let Some(alias) = &item.alias { - Some(alias.clone()) - } else { - // Derive name from expression - match &item.expression { - LogicalExpression::Variable(name) => Some(name.clone()), - LogicalExpression::Property { variable, property } => { - Some(format!("{variable}.{property}")) - } - _ => None, - } - } + /// If the RETURN clause is a single `count()` aggregate of at most one + /// argument, returns that argument (`None` for `count(*)`) and whether it + /// is counted DISTINCT. + fn count_aggregate_return(ret: &ast::ReturnClause) -> Option<(Option<&ast::Expression>, bool)> { + let [item] = ret.items.as_slice() else { + return None; + }; + match &item.expression { + ast::Expression::FunctionCall { + name, + args, + distinct, + } if name.eq_ignore_ascii_case("count") && args.len() <= 1 => { + Some((args.first(), *distinct)) } - // Walk through wrapping operators to find the Return - LogicalOperator::Sort(s) => Self::extract_return_column_name(&s.input), - LogicalOperator::Limit(l) => Self::extract_return_column_name(&l.input), - LogicalOperator::Distinct(d) => Self::extract_return_column_name(&d.input), _ => None, } } diff --git a/crates/grafeo-engine/src/query/translators/gql/pattern.rs b/crates/grafeo-engine/src/query/translators/gql/pattern.rs index a9b6cbb9f..023c053ad 100644 --- a/crates/grafeo-engine/src/query/translators/gql/pattern.rs +++ b/crates/grafeo-engine/src/query/translators/gql/pattern.rs @@ -517,7 +517,11 @@ impl GqlTranslator { let (min_hops, max_hops) = edge_hop_bounds(edge); - let is_variable_length = min_hops != 1 || max_hops.is_none() || max_hops != Some(1); + // A quantifier (`{1,1}` too) makes the edge variable a group + // variable: the list of the path's edges. An edge without one is a + // single hop, so the quantified edges are the variable-length ones. + let quantified = edge.min_hops.is_some() || edge.max_hops.is_some(); + let is_variable_length = quantified; // For variable-length edges with a named edge variable, auto-generate // a path alias if none exists, so path detail columns are available @@ -559,6 +563,7 @@ impl GqlTranslator { let property_path = expand_path_alias.clone(); plan = LogicalOperator::Expand(ExpandOp { + quantified, from_variable: current_source, to_variable: expand_target.clone(), edge_variable: edge_var, diff --git a/crates/grafeo-engine/src/query/translators/graphql.rs b/crates/grafeo-engine/src/query/translators/graphql.rs index 16e41bada..209d5ef79 100644 --- a/crates/grafeo-engine/src/query/translators/graphql.rs +++ b/crates/grafeo-engine/src/query/translators/graphql.rs @@ -845,6 +845,7 @@ impl GraphQLTranslator { // The field name is the edge type: preserve original case to match how edges are stored let mut plan = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: from_var.to_string(), to_variable: to_var.clone(), edge_variable: None, diff --git a/crates/grafeo-engine/src/query/translators/gremlin.rs b/crates/grafeo-engine/src/query/translators/gremlin.rs index aa5a49f0b..3958f6cdb 100644 --- a/crates/grafeo-engine/src/query/translators/gremlin.rs +++ b/crates/grafeo-engine/src/query/translators/gremlin.rs @@ -596,6 +596,7 @@ impl GremlinTranslator { let target_var = self.var_gen.next(); plan = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: var, to_variable: target_var, edge_variable: Some(edge_var.clone()), @@ -650,6 +651,7 @@ impl GremlinTranslator { let target_var = self.var_gen.next(); let edge_types = labels.clone(); let plan = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: current_var.to_string(), to_variable: target_var.clone(), edge_variable: None, @@ -667,6 +669,7 @@ impl GremlinTranslator { let target_var = self.var_gen.next(); let edge_types = labels.clone(); let plan = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: current_var.to_string(), to_variable: target_var.clone(), edge_variable: None, @@ -684,6 +687,7 @@ impl GremlinTranslator { let target_var = self.var_gen.next(); let edge_types = labels.clone(); let plan = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: current_var.to_string(), to_variable: target_var.clone(), edge_variable: None, @@ -702,6 +706,7 @@ impl GremlinTranslator { let target_var = self.var_gen.next(); let edge_types = labels.clone(); let plan = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: current_var.to_string(), to_variable: target_var, edge_variable: Some(edge_var.clone()), @@ -720,6 +725,7 @@ impl GremlinTranslator { let target_var = self.var_gen.next(); let edge_types = labels.clone(); let plan = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: current_var.to_string(), to_variable: target_var, edge_variable: Some(edge_var.clone()), @@ -738,6 +744,7 @@ impl GremlinTranslator { let target_var = self.var_gen.next(); let edge_types = labels.clone(); let plan = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: current_var.to_string(), to_variable: target_var, edge_variable: Some(edge_var.clone()), @@ -1853,6 +1860,7 @@ impl GremlinTranslator { let target_var = self.var_gen.next(); let plan = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: current_var.to_string(), to_variable: target_var.clone(), edge_variable: None, @@ -2273,6 +2281,7 @@ impl GremlinTranslator { let target_var = self.var_gen.next(); // Create an existence subquery via Expand + count > 0 let expand = LogicalOperator::Expand(ExpandOp { + quantified: false, from_variable: current_var.to_string(), to_variable: target_var, edge_variable: None, diff --git a/crates/grafeo-engine/src/query/translators/mod.rs b/crates/grafeo-engine/src/query/translators/mod.rs index a14847e27..eaa1d4ff5 100644 --- a/crates/grafeo-engine/src/query/translators/mod.rs +++ b/crates/grafeo-engine/src/query/translators/mod.rs @@ -3,6 +3,14 @@ //! Each submodule translates a parsed AST from a specific query language //! (GQL, Cypher, SPARQL, etc.) into the shared [`LogicalPlan`](crate::query::plan::LogicalPlan) IR. +#[cfg(any( + feature = "gql", + feature = "cypher", + feature = "sparql", + feature = "gremlin", + feature = "graphql", + feature = "sql-pgq" +))] pub(crate) mod common; #[cfg(feature = "gql")] diff --git a/crates/grafeo-engine/src/query/translators/sql_pgq.rs b/crates/grafeo-engine/src/query/translators/sql_pgq.rs index fdbdcc307..66697cde5 100644 --- a/crates/grafeo-engine/src/query/translators/sql_pgq.rs +++ b/crates/grafeo-engine/src/query/translators/sql_pgq.rs @@ -4,8 +4,10 @@ //! representation. The inner MATCH clause reuses GQL AST types, so pattern //! translation follows the GQL translator pattern. +use std::collections::HashSet; + use super::common::{ - check_branch_columns, combine_with_and, has_all_labels, is_aggregate_function, + VarGen, check_branch_columns, combine_with_and, has_all_labels, is_aggregate_function, to_aggregate_function, wrap_filter, wrap_limit, wrap_return, wrap_skip, wrap_sort, }; use crate::query::plan::{ @@ -52,7 +54,7 @@ pub fn translate(query: &str) -> Result { }; let statement = sql_pgq::parse(actual_query)?; - let translator = SqlPgqTranslator::new(); + let translator = SqlPgqTranslator::new(&statement); let mut plan = translator.translate_statement(&statement)?; plan.explain = explain; plan.profile = profile; @@ -60,11 +62,39 @@ pub fn translate(query: &str) -> Result { } /// SQL/PGQ AST to logical plan translator. -struct SqlPgqTranslator; +struct SqlPgqTranslator { + /// Names for anonymous nodes: each is a variable of its own, so two of + /// them in one pattern are not taken for the same node. + anonymous: VarGen, + /// The names the statement itself uses, which no anonymous node gets. + named: HashSet, +} impl SqlPgqTranslator { - fn new() -> Self { - Self + fn new(statement: &ast::Statement) -> Self { + let mut named = HashSet::new(); + match statement { + ast::Statement::Select(select) => select_names(select, &mut named), + ast::Statement::SetOperation(set_op) => { + select_names(&set_op.left, &mut named); + select_names(&set_op.right, &mut named); + } + ast::Statement::CreatePropertyGraph(_) | ast::Statement::Call(_) => {} + } + Self { + anonymous: VarGen::new(), + named, + } + } + + /// A name for an anonymous node that no variable of the statement has. + fn anonymous_variable(&self) -> String { + loop { + let name = self.anonymous.next(); + if !self.named.contains(&name) { + return name; + } + } } fn translate_statement(&self, stmt: &ast::Statement) -> Result { @@ -575,7 +605,10 @@ impl SqlPgqTranslator { node: &ast::NodePattern, input: Option, ) -> Result { - let variable = node.variable.clone().unwrap_or_else(|| "_anon".to_string()); + let variable = node + .variable + .clone() + .unwrap_or_else(|| self.anonymous_variable()); let label = node.labels.first().cloned(); let mut plan = LogicalOperator::NodeScan(NodeScanOp { @@ -625,7 +658,7 @@ impl SqlPgqTranslator { .target .variable .clone() - .unwrap_or_else(|| "_anon".to_string()); + .unwrap_or_else(|| self.anonymous_variable()); let direction = match edge.direction { ast::EdgeDirection::Outgoing => ExpandDirection::Outgoing, @@ -633,22 +666,24 @@ impl SqlPgqTranslator { ast::EdgeDirection::Undirected => ExpandDirection::Both, }; - let (min_hops, max_hops) = if edge.min_hops.is_some() || edge.max_hops.is_some() { + // A quantified edge (`*1..1` too) binds the list of the path's edges. + let quantified = edge.min_hops.is_some() || edge.max_hops.is_some(); + let (min_hops, max_hops) = if quantified { (edge.min_hops.unwrap_or(1), edge.max_hops) } else { (1, Some(1)) }; - // Set path_alias for variable-length patterns so path functions work - let is_variable_length = - min_hops != 1 || max_hops.is_none() || max_hops.is_some_and(|m| m != 1); - let path_alias = if is_variable_length { + // Set path_alias for variable-length patterns so path functions work: + // the quantified ones (an edge without a quantifier is a single hop). + let path_alias = if quantified { edge_variable.clone() } else { None }; let expand = LogicalOperator::Expand(ExpandOp { + quantified, from_variable, to_variable: to_variable.clone(), edge_variable, @@ -1286,6 +1321,47 @@ impl SqlPgqTranslator { } } +/// Adds the names a SELECT gives its pattern variables, paths and columns. +fn select_names(select: &ast::SelectStatement, names: &mut HashSet) { + let graph_table = &select.graph_table; + for clause in std::iter::once(&graph_table.match_clause).chain(&graph_table.optional_matches) { + for aliased in &clause.patterns { + names.extend(aliased.alias.iter().cloned()); + pattern_names(&aliased.pattern, names); + } + } + names.extend( + graph_table + .columns + .items + .iter() + .map(|item| item.alias.clone()), + ); + if let ast::SelectList::Columns(items) = &select.select_list { + names.extend(items.iter().filter_map(|item| item.alias.clone())); + } +} + +/// Adds the variables a pattern names. +fn pattern_names(pattern: &ast::Pattern, names: &mut HashSet) { + match pattern { + ast::Pattern::Node(node) => names.extend(node.variable.iter().cloned()), + ast::Pattern::Path(path) => { + names.extend(path.source.variable.iter().cloned()); + for edge in &path.edges { + names.extend(edge.variable.iter().cloned()); + names.extend(edge.target.variable.iter().cloned()); + } + } + ast::Pattern::Quantified { pattern, .. } => pattern_names(pattern, names), + ast::Pattern::Union(patterns) | ast::Pattern::MultisetUnion(patterns) => { + for pattern in patterns { + pattern_names(pattern, names); + } + } + } +} + #[cfg(test)] mod tests { use super::*; diff --git a/crates/grafeo-engine/src/session/mod.rs b/crates/grafeo-engine/src/session/mod.rs index f3be9d3fd..a6c73c616 100644 --- a/crates/grafeo-engine/src/session/mod.rs +++ b/crates/grafeo-engine/src/session/mod.rs @@ -15,11 +15,11 @@ use std::time::{Duration, Instant}; #[cfg(feature = "lpg")] use grafeo_common::grafeo_debug_span; +use grafeo_common::grafeo_info_span; #[cfg(feature = "lpg")] use grafeo_common::types::{EdgeId, NodeId}; use grafeo_common::types::{EpochId, PropertyKey, TransactionId, Value}; use grafeo_common::utils::error::Result; -use grafeo_common::{grafeo_info_span, grafeo_warn}; #[cfg(feature = "lpg")] use grafeo_core::graph::Direction; #[cfg(feature = "lpg")] @@ -114,6 +114,7 @@ pub(crate) struct SessionConfig { pub query_timeout: Option, pub max_property_size: Option, /// Buffer manager for memory-aware query execution. + #[cfg(feature = "spill")] pub buffer_manager: Option>, pub commit_counter: Arc, pub gc_interval: usize, @@ -183,6 +184,7 @@ pub struct Session { /// Maximum size in bytes for a single property value. max_property_size: Option, /// Buffer manager for memory-aware execution (spill decisions). + #[cfg(feature = "spill")] buffer_manager: Option>, /// Shared commit counter for triggering auto-GC. commit_counter: Arc, @@ -311,6 +313,7 @@ impl Session { graph_model: cfg.graph_model, query_timeout: cfg.query_timeout, max_property_size: cfg.max_property_size, + #[cfg(feature = "spill")] buffer_manager: cfg.buffer_manager, commit_counter: cfg.commit_counter, gc_interval: cfg.gc_interval, @@ -381,7 +384,7 @@ impl Session { && self.current_transaction.lock().is_none() && let Err(e) = wal.flush_implicit() { - grafeo_warn!("Session: failed to write WAL records: {}", e); + grafeo_common::grafeo_warn!("Session: failed to write WAL records: {}", e); } } @@ -450,6 +453,7 @@ impl Session { graph_model: cfg.graph_model, query_timeout: cfg.query_timeout, max_property_size: cfg.max_property_size, + #[cfg(feature = "spill")] buffer_manager: cfg.buffer_manager, commit_counter: cfg.commit_counter, gc_interval: cfg.gc_interval, @@ -580,7 +584,9 @@ impl Session { } named_store as Arc } - None => Arc::clone(&self.graph_store), + // Dropped meanwhile: no data, never the default graph's (the + // graph check before a statement reports the drop). + None => Arc::new(grafeo_core::graph::NullGraphStore) as Arc, }, #[cfg(not(feature = "lpg"))] Some(_) => Arc::clone(&self.graph_store), @@ -603,29 +609,38 @@ impl Session { #[cfg(feature = "lpg")] Some(name) => match self.store.graph(name) { Some(named_store) => { - let mut store: Arc = Arc::clone(&named_store) as _; + let store: Arc = Arc::clone(&named_store) as _; #[cfg(feature = "wal")] - if let Some(wal) = &self.wal { - store = Arc::new(crate::database::wal_store::WalGraphStore::new_for_graph( - named_store, - Arc::clone(wal), - name.to_string(), - )); - } + let store: Arc = match &self.wal { + Some(wal) => { + Arc::new(crate::database::wal_store::WalGraphStore::new_for_graph( + named_store, + Arc::clone(wal), + name.to_string(), + )) + } + None => store, + }; #[cfg(feature = "cdc")] - if let Some(ref pending) = self.cdc_pending_events { - store = Arc::new(crate::database::cdc_store::CdcGraphStore::wrap( - store, - Arc::clone(&self.cdc_log), - Arc::clone(pending), - )); - } + let store: Arc = match &self.cdc_pending_events { + Some(pending) => Arc::new( + crate::database::cdc_store::CdcGraphStore::wrap( + store, + Arc::clone(&self.cdc_log), + Arc::clone(pending), + ) + .for_graph(name.to_string()), + ), + None => store, + }; Some(store) } - None => self.graph_store_mut.as_ref().map(Arc::clone), + // Dropped meanwhile: nothing to write to, never the default + // graph (see `store_for_key`). + None => None, }, #[cfg(not(feature = "lpg"))] Some(_) => self.graph_store_mut.as_ref().map(Arc::clone), @@ -1240,7 +1255,7 @@ impl Session { if let Some(ref wal) = self.wal && let Err(e) = wal.wal().log_batch(&records) { - grafeo_warn!("Failed to log schema change to WAL: {}", e); + grafeo_common::grafeo_warn!("Failed to log schema change to WAL: {}", e); } } @@ -2858,7 +2873,7 @@ impl Session { // optimizer and planner see the values and no cached plan keeps them. let cache_key = CacheKey::with_graph(query, QueryLanguage::Gql, self.current_graph()); let parsed = params.and_then(|_| self.query_cache.get_parsed(&cache_key)); - let logical_plan = match parsed { + let mut logical_plan = match parsed { Some(plan) => plan, None => match gql::translate_full(query)? { gql::GqlTranslationResult::SessionCommand(cmd) => { @@ -2905,6 +2920,7 @@ impl Session { )); } + let params = params_to_fill(&mut logical_plan, params)?; let optimized_plan = if let Some(params) = params { self.optimize_with_params(logical_plan, params)? } else if let Some(cached_plan) = self.query_cache.get_optimized(&cache_key) { @@ -2931,6 +2947,8 @@ impl Session { // EXPLAIN: annotate pushdown hints and return the plan tree if optimized_plan.explain { use crate::query::processor::{annotate_pushdown_hints, explain_result}; + #[cfg(feature = "lpg")] + self.check_graph_access(optimized_plan.root.has_mutations())?; let mut plan = optimized_plan; annotate_pushdown_hints(&mut plan.root, active.as_ref()); return Ok(explain_result(&plan)); @@ -3142,6 +3160,7 @@ impl Session { plan } }; + self.check_graph_access(false)?; // Cache + bind + optimize (same path as execute). let cache_key = CacheKey::with_graph(query, QueryLanguage::Gql, self.current_graph()); @@ -3170,12 +3189,14 @@ impl Session { let active = self.active_store(); let has_active_tx = self.current_transaction.lock().is_some(); let (viewing_epoch, transaction_id) = self.get_transaction_context(); - let planner = self.create_planner_for_store_with_read_only( - Arc::clone(&active), - viewing_epoch, - transaction_id, - !has_active_tx, - ); + let planner = self + .create_planner_for_store_with_read_only( + Arc::clone(&active), + viewing_epoch, + transaction_id, + !has_active_tx, + ) + .for_streaming(); let physical_plan = planner.plan(&optimized_plan)?; let columns = physical_plan.columns.clone(); @@ -3364,7 +3385,7 @@ impl Session { // A parameterized statement reuses its parsed plan (see execute_gql). let cache_key = CacheKey::with_graph(query, QueryLanguage::Cypher, self.current_graph()); let parsed = params.and_then(|_| self.query_cache.get_parsed(&cache_key)); - let logical_plan = match parsed { + let mut logical_plan = match parsed { Some(plan) => plan, // Schema DDL and SHOW commands run before the normal query path. None => match cypher::translate_full(query)? { @@ -3406,6 +3427,7 @@ impl Session { }, }; + let params = params_to_fill(&mut logical_plan, params)?; let optimized_plan = if let Some(params) = params { self.optimize_with_params(logical_plan, params)? } else if let Some(cached_plan) = self.query_cache.get_optimized(&cache_key) { @@ -3437,6 +3459,8 @@ impl Session { // EXPLAIN if optimized_plan.explain { use crate::query::processor::{annotate_pushdown_hints, explain_result}; + #[cfg(feature = "lpg")] + self.check_graph_access(optimized_plan.root.has_mutations())?; let mut plan = optimized_plan; annotate_pushdown_hints(&mut plan.root, active.as_ref()); return Ok(explain_result(&plan)); @@ -3535,13 +3559,19 @@ impl Session { /// ``` #[cfg(feature = "gremlin")] pub fn execute_gremlin(&self, query: &str) -> Result { - use crate::query::{binder::Binder, optimizer::Optimizer, translators::gremlin}; + use crate::query::{ + binder::Binder, optimizer::Optimizer, processor::substitute_params, + translators::gremlin, + }; #[cfg(all(feature = "metrics", not(target_arch = "wasm32")))] let start_time = Instant::now(); // Parse and translate the query to a logical plan - let logical_plan = gremlin::translate(query)?; + let mut logical_plan = gremlin::translate(query)?; + + // No parameters are supplied, so one the query uses is missing. + substitute_params(&mut logical_plan, &std::collections::HashMap::new())?; // Semantic validation let mut binder = Binder::new(); @@ -3682,11 +3712,10 @@ impl Session { let mut logical_plan = graphql::translate(query)?; - // Substitute default parameter values from variable declarations - if !logical_plan.default_params.is_empty() { - let defaults = logical_plan.default_params.clone(); - substitute_params(&mut logical_plan, &defaults)?; - } + // Substitute default parameter values from variable declarations; a + // variable without a default is missing. + let defaults = logical_plan.default_params.clone(); + substitute_params(&mut logical_plan, &defaults)?; let mut binder = Binder::new(); let _binding_context = binder.bind(&logical_plan)?; @@ -3838,7 +3867,7 @@ impl Session { // A parameterized statement reuses its parsed plan (see execute_gql). let cache_key = CacheKey::with_graph(query, QueryLanguage::SqlPgq, self.current_graph()); let parsed = params.and_then(|_| self.query_cache.get_parsed(&cache_key)); - let logical_plan = match parsed { + let mut logical_plan = match parsed { Some(plan) => plan, None => { // Parse and translate (always needed to check for DDL) @@ -3871,6 +3900,7 @@ impl Session { } }; + let params = params_to_fill(&mut logical_plan, params)?; let optimized_plan = if let Some(params) = params { self.optimize_with_params(logical_plan, params)? } else if let Some(cached_plan) = self.query_cache.get_optimized(&cache_key) { @@ -4095,7 +4125,7 @@ impl Session { if let Some(ref wal) = self.wal && let Err(e) = wal.flush_implicit() { - grafeo_warn!("Session: failed to write WAL records: {}", e); + grafeo_common::grafeo_warn!("Session: failed to write WAL records: {}", e); } let transaction_id = if let Some(level) = isolation_level { @@ -4251,7 +4281,7 @@ impl Session { epoch: commit_epoch, }, ]) { - grafeo_warn!("Failed to write transaction to WAL: {}", e); + grafeo_common::grafeo_warn!("Failed to write transaction to WAL: {}", e); } } @@ -4688,8 +4718,7 @@ impl Session { where F: FnOnce() -> Result, { - self.check_active_graph()?; - self.check_graph_grant(has_mutations)?; + self.check_graph_access(has_mutations)?; if has_mutations { self.check_writable()?; } @@ -4724,6 +4753,62 @@ impl Session { } } + /// Runs `body`, which may run several statements, as one write: in a + /// transaction of its own when none is open (whatever the auto-commit + /// setting), otherwise inside the open one. An error undoes everything + /// `body` wrote; an open transaction goes on. + #[cfg(all(feature = "lpg", feature = "gql"))] + pub(crate) fn as_one_write(&self, body: impl FnOnce() -> Result) -> Result { + if self.current_transaction.lock().is_some() { + return self.with_auto_commit(true, body); + } + self.begin_transaction_inner(false, None)?; + match self.with_auto_commit(true, body) { + Ok(result) => { + self.commit_inner()?; + Ok(result) + } + Err(error) => { + // The body's error is the one to report, as in + // `with_auto_commit`. + let _ = self.rollback_inner(); + Err(error) + } + } + } + + /// Fails when the selected graph is gone or this identity has no grant + /// for it (see `check_active_graph` and `check_graph_grant`). Every + /// statement checks this in `with_auto_commit`; `EXPLAIN`, which shows a + /// plan without running it, and streamed queries check it themselves. + #[cfg(feature = "lpg")] + fn check_graph_access(&self, writes: bool) -> Result<()> { + self.check_active_graph()?; + if writes { + self.check_reads_the_present()?; + } + self.check_graph_grant(writes) + } + + /// Fails while the session reads at an earlier epoch + /// ([`set_viewing_epoch`](Self::set_viewing_epoch), `execute_at_epoch`): a + /// write there would change the past. + fn check_reads_the_present(&self) -> Result<()> { + match *self.viewing_epoch_override.lock() { + Some(epoch) => Err(grafeo_common::utils::error::Error::Query( + grafeo_common::utils::error::QueryError::new( + grafeo_common::utils::error::QueryErrorKind::Semantic, + format!( + "cannot write while the session reads at an earlier epoch ({}): \ + clear the viewing epoch first", + epoch.as_u64() + ), + ), + )), + None => Ok(()), + } + } + /// Fails when this identity has per-graph grants and none covers the /// selected graph at the level the statement needs: any grant to read, a /// read-write one to write. `use_graph` selects a graph without the check @@ -5042,6 +5127,7 @@ impl Session { ) -> Result { use grafeo_core::execution::operators::GraphWriter; + self.check_reads_the_present()?; self.with_auto_commit(true, || { let key = self.active_graph_storage_key(); if self.current_transaction.lock().is_some() { @@ -5606,7 +5692,8 @@ impl Session { // ── Change Data Capture ───────────────────────────────────────────── - /// Returns the full change history for an entity (node or edge). + /// Returns the full change history for an entity (node or edge) of the + /// session's current graph. /// /// # Errors /// @@ -5617,10 +5704,13 @@ impl Session { entity_id: impl Into, ) -> Result> { self.require_permission(crate::auth::StatementKind::Read)?; - Ok(self.cdc_log.history(entity_id.into())) + Ok(self + .cdc_log + .history_in(self.active_graph_storage_key().as_deref(), entity_id.into())) } - /// Returns change events for an entity since the given epoch. + /// Returns change events for an entity of the session's current graph + /// since the given epoch. /// /// # Errors /// @@ -5632,10 +5722,15 @@ impl Session { since_epoch: EpochId, ) -> Result> { self.require_permission(crate::auth::StatementKind::Read)?; - Ok(self.cdc_log.history_since(entity_id.into(), since_epoch)) + Ok(self.cdc_log.history_since_in( + self.active_graph_storage_key().as_deref(), + entity_id.into(), + since_epoch, + )) } - /// Returns all change events across all entities in an epoch range. + /// Returns all change events across all entities and graphs in an epoch + /// range; each event names its graph. /// /// # Errors /// @@ -5673,6 +5768,33 @@ impl Drop for Session { } } +/// The parameter values to fill into `plan`: `None` for an empty map when the +/// plan has no defaults either, after checking that the plan names no +/// parameter (it fails like any missing value). A statement without +/// parameters then uses its cached plan, like the same call without a map. +#[cfg(any(feature = "gql", feature = "cypher", feature = "sql-pgq"))] +fn params_to_fill<'a>( + plan: &mut crate::query::plan::LogicalPlan, + params: Option<&'a std::collections::HashMap>, +) -> Result>> { + match params { + Some(values) if values.is_empty() && plan.default_params.is_empty() => { + crate::query::processor::substitute_params(plan, values)?; + Ok(None) + } + // An EXPLAIN without parameters shows the plan with them unresolved. + None if plan.explain && plan.default_params.is_empty() => Ok(None), + // No parameters: the plan's defaults fill what they can, and a + // parameter nobody supplied fails here, before planning. + None => { + let defaults = plan.default_params.clone(); + crate::query::processor::substitute_params(plan, &defaults)?; + Ok(None) + } + other => Ok(other), + } +} + #[cfg(test)] mod tests { use super::parse_default_literal; @@ -7140,4 +7262,21 @@ mod tests { assert_eq!(result.rows[0][0], Value::from("social")); } } + + /// A selected graph that was dropped meanwhile resolves to no data and no + /// writable store, never to the default graph's: a statement that passed + /// its graph check just before the drop must not read or write there. + #[cfg(feature = "lpg")] + #[test] + fn a_dropped_selected_graph_resolves_to_nothing() { + let db = GrafeoDB::new_in_memory(); + db.execute("INSERT (:Person {name: 'Alix'})").unwrap(); + db.create_graph("model").unwrap(); + let session = db.session(); + session.use_graph("model"); + assert!(db.drop_graph("model")); + + assert_eq!(session.active_store().node_count(), 0); + assert!(session.active_write_store().is_none()); + } } diff --git a/crates/grafeo-engine/src/session/rdf.rs b/crates/grafeo-engine/src/session/rdf.rs index f82ff33e4..2b86de526 100644 --- a/crates/grafeo-engine/src/session/rdf.rs +++ b/crates/grafeo-engine/src/session/rdf.rs @@ -56,6 +56,7 @@ impl Session { graph_model: cfg.graph_model, query_timeout: cfg.query_timeout, max_property_size: cfg.max_property_size, + #[cfg(feature = "spill")] buffer_manager: cfg.buffer_manager, commit_counter: cfg.commit_counter, gc_interval: cfg.gc_interval, diff --git a/crates/grafeo-engine/tests/auth_permissions.rs b/crates/grafeo-engine/tests/auth_permissions.rs index d3350ad35..7ec4d8efb 100644 --- a/crates/grafeo-engine/tests/auth_permissions.rs +++ b/crates/grafeo-engine/tests/auth_permissions.rs @@ -403,7 +403,8 @@ fn cypher_execute_language_write_with_readonly_fails() { /// `use_graph` selects a graph without the grant check `USE GRAPH` makes: /// statements and direct writes check the grant themselves, reads needing -/// any grant for the graph and writes a read-write one. +/// any grant for the graph and writes a read-write one. `EXPLAIN` needs what +/// the statement it shows would need. #[test] fn per_graph_grants_hold_after_use_graph() { use grafeo_engine::auth::Grant; @@ -428,15 +429,41 @@ fn per_graph_grants_hold_after_use_graph() { ); denied(session.execute("INSERT (:X)").unwrap_err().to_string()); denied(session.create_node(&["X"]).unwrap_err().to_string()); + denied( + session + .execute("EXPLAIN MATCH (n) RETURN n") + .unwrap_err() + .to_string(), + ); + #[cfg(feature = "cypher")] + denied( + session + .execute_cypher("EXPLAIN MATCH (n) RETURN n") + .unwrap_err() + .to_string(), + ); + let Err(error) = session.execute_streaming("MATCH (n) RETURN n") else { + panic!("a stream of a graph without a grant"); + }; + denied(error.to_string()); session.use_graph("readonly"); session.execute("MATCH (n) RETURN count(n)").unwrap(); denied(session.execute("INSERT (:X)").unwrap_err().to_string()); denied(session.create_node(&["X"]).unwrap_err().to_string()); + session.execute("EXPLAIN MATCH (n) RETURN n").unwrap(); + assert!(session.execute_streaming("MATCH (n) RETURN n").is_ok()); + denied( + session + .execute("EXPLAIN INSERT (:X)") + .unwrap_err() + .to_string(), + ); session.use_graph("open"); session.execute("INSERT (:X)").unwrap(); session.create_node(&["X"]).unwrap(); + session.execute("EXPLAIN INSERT (:X)").unwrap(); let admin = db.session(); for (graph, expected) in [("open", 2), ("readonly", 0), ("closed", 0)] { diff --git a/crates/grafeo-engine/tests/bound_edges.rs b/crates/grafeo-engine/tests/bound_edges.rs new file mode 100644 index 000000000..440f6e9e2 --- /dev/null +++ b/crates/grafeo-engine/tests/bound_edges.rs @@ -0,0 +1,272 @@ +//! A pattern that names an edge bound before matches that edge only: the +//! rows of the same pattern with a fresh edge filtered to equal it. +//! +//! ```bash +//! cargo test -p grafeo-engine --all-features --test bound_edges +//! ``` + +use grafeo_common::types::Value; +use grafeo_engine::GrafeoDB; + +/// Alix, Gus, Vincent and Mia, and Amsterdam. Edges with `w` 1 to 6: KNOWS +/// Alix->Gus (1), Gus->Vincent (2), Vincent->Alix (3), a second KNOWS +/// Alix->Gus (4), LIVES_IN Alix->Amsterdam (5) and a LIKES loop on Mia (6). +/// The parallel edges tell a check on the edge from one on its endpoints. +fn graph() -> GrafeoDB { + let db = GrafeoDB::new_in_memory(); + db.execute( + "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), \ + (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), \ + (ams:City {name: 'Amsterdam'}), \ + (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), \ + (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), \ + (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)", + ) + .unwrap(); + db +} + +/// The rows of `result` as text, sorted. +fn rows( + result: grafeo_common::utils::error::Result, +) -> Vec> { + let mut rows: Vec> = result + .unwrap() + .rows() + .iter() + .map(|row| { + row.iter() + .map(|value| match value { + Value::String(text) => text.to_string(), + Value::Int64(number) => number.to_string(), + Value::Null => "null".to_string(), + other => panic!("unexpected value {other:?}"), + }) + .collect() + }) + .collect(); + rows.sort(); + rows +} + +/// `rows` for literal text cells. +fn expected(rows: &[&[&str]]) -> Vec> { + let mut rows: Vec> = rows + .iter() + .map(|row| row.iter().map(|cell| (*cell).to_string()).collect()) + .collect(); + rows.sort(); + rows +} + +/// Runs `query` as GQL and, with the `cypher` feature, as Cypher, and checks +/// both against `want`. +fn assert_both(db: &GrafeoDB, query: &str, want: &[&[&str]]) { + assert_eq!(rows(db.execute(query)), expected(want), "GQL: {query}"); + #[cfg(feature = "cypher")] + assert_eq!( + rows(db.execute_cypher(query)), + expected(want), + "Cypher: {query}" + ); +} + +#[test] +fn a_later_match_through_a_bound_edge_matches_that_edge() { + let db = graph(); + assert_both( + &db, + "MATCH ()-[r]->() MATCH (x)-[r]->(y) RETURN r.w, x.name, y.name", + &[ + &["1", "Alix", "Gus"], + &["2", "Gus", "Vincent"], + &["3", "Vincent", "Alix"], + &["4", "Alix", "Gus"], + &["5", "Alix", "Amsterdam"], + &["6", "Mia", "Mia"], + ], + ); + assert_both( + &db, + "MATCH ()-[r]->() MATCH ()-[r]->() RETURN count(*)", + &[&["6"]], + ); +} + +/// With both ends bound too, the parallel KNOWS edges Alix->Gus each match +/// once, not twice. +#[test] +fn a_bound_edge_between_bound_nodes_matches_once() { + let db = graph(); + assert_both( + &db, + "MATCH (a)-[r]->(b) MATCH (a)-[r]->(b) RETURN r.w", + &[&["1"], &["2"], &["3"], &["4"], &["5"], &["6"]], + ); + assert_both( + &db, + "MATCH (a)-[r]->(b) MATCH (a)-[r]->(c) RETURN r.w, c.name", + &[ + &["1", "Gus"], + &["2", "Vincent"], + &["3", "Alix"], + &["4", "Gus"], + &["5", "Amsterdam"], + &["6", "Mia"], + ], + ); +} + +/// The bound edge keeps its direction: read backwards its ends swap, and +/// from its own source backwards only the loop matches. +#[test] +fn a_bound_edge_keeps_its_direction() { + let db = graph(); + assert_both( + &db, + "MATCH (a)-[r]->(b) MATCH (x)<-[r]-(y) RETURN r.w, x.name, y.name", + &[ + &["1", "Gus", "Alix"], + &["2", "Vincent", "Gus"], + &["3", "Alix", "Vincent"], + &["4", "Gus", "Alix"], + &["5", "Amsterdam", "Alix"], + &["6", "Mia", "Mia"], + ], + ); + assert_both( + &db, + "MATCH (a)-[r]->(b) MATCH (a)<-[r]-(b) RETURN r.w", + &[&["6"]], + ); + assert_both( + &db, + "MATCH (a)-[r:KNOWS {w: 2}]->(b) MATCH (x)-[r]-(y) RETURN x.name, y.name", + &[&["Gus", "Vincent"], &["Vincent", "Gus"]], + ); +} + +/// A type or a property map on the later pattern applies to the bound edge. +#[test] +fn a_type_or_property_on_the_later_pattern_checks_the_bound_edge() { + let db = graph(); + assert_both( + &db, + "MATCH ()-[r:KNOWS]->() MATCH ()-[r:LIVES_IN]->() RETURN count(*)", + &[&["0"]], + ); + assert_both( + &db, + "MATCH ()-[r]->() MATCH ()-[r:KNOWS]->() RETURN r.w", + &[&["1"], &["2"], &["3"], &["4"]], + ); + assert_both( + &db, + "MATCH ()-[r]->() MATCH (x)-[r {w: 3}]->(y) RETURN x.name, y.name", + &[&["Vincent", "Alix"]], + ); +} + +/// The same edge twice in one path can only be a loop. +#[test] +fn the_same_edge_twice_in_a_path_is_a_loop() { + let db = graph(); + assert_eq!( + rows(db.execute("MATCH (a)-[r]->(b)-[r]->(c) RETURN a.name, b.name, c.name")), + expected(&[&["Mia", "Mia", "Mia"]]) + ); +} + +/// A named path through the bound edge has that edge only. +#[test] +fn a_named_path_through_a_bound_edge() { + let db = graph(); + assert_both( + &db, + "MATCH ()-[r:KNOWS]->() MATCH p = (x)-[r]->(y) RETURN r.w, length(p)", + &[&["1", "1"], &["2", "1"], &["3", "1"], &["4", "1"]], + ); +} + +/// A variable-length pattern that names the edge list of an earlier one +/// matches that list: each two-hop KNOWS walk once. +#[test] +fn a_bound_edge_list_matches_the_same_walk() { + let db = graph(); + assert_both(&db, "MATCH ()-[r:KNOWS*2]->() RETURN count(*)", &[&["5"]]); + assert_both( + &db, + "MATCH ()-[r:KNOWS*2]->() MATCH (x)-[r:KNOWS*2]->(y) RETURN count(*)", + &[&["5"]], + ); +} + +/// The edge stays bound through WITH, ORDER BY and LIMIT. +#[cfg(feature = "cypher")] +#[test] +fn a_bound_edge_after_an_ordered_cut() { + let db = graph(); + assert_eq!( + rows(db.execute_cypher( + "MATCH ()-[r]->() WITH r ORDER BY r.w LIMIT 2 MATCH (x)-[r]->(y) \ + RETURN r.w, x.name, y.name" + )), + expected(&[&["1", "Alix", "Gus"], &["2", "Gus", "Vincent"]]) + ); +} + +/// A subquery that imports the edge, by name or with `WITH *`, matches it. +#[test] +fn a_subquery_matches_the_edge_it_imports() { + let db = graph(); + let want: &[&[&str]] = &[ + &["1", "Gus"], + &["2", "Vincent"], + &["3", "Alix"], + &["4", "Gus"], + &["5", "Amsterdam"], + &["6", "Mia"], + ]; + assert_both( + &db, + "MATCH ()-[r]->() CALL { WITH r MATCH (x)-[r]->(y) RETURN y.name AS target } \ + RETURN r.w, target", + want, + ); + assert_both( + &db, + "MATCH ()-[r]->() CALL { WITH * MATCH (x)-[r]->(y) RETURN y.name AS target } \ + RETURN r.w, target", + want, + ); +} + +/// The edge stays bound across a subquery that returns other columns. +#[test] +fn a_bound_edge_after_a_subquery() { + let db = graph(); + assert_both( + &db, + "MATCH ()-[r]->() CALL { RETURN 1 AS one } MATCH (x)-[r]->(y) RETURN count(*)", + &[&["6"]], + ); +} + +/// OPTIONAL MATCH joins on the bound edge: the KNOWS edges find themselves, +/// the others none. +#[test] +fn an_optional_match_through_a_bound_edge() { + let db = graph(); + assert_both( + &db, + "MATCH ()-[r]->() OPTIONAL MATCH (x)-[r:KNOWS]->(y) RETURN r.w, y.name", + &[ + &["1", "Gus"], + &["2", "Vincent"], + &["3", "Alix"], + &["4", "Gus"], + &["5", "null"], + &["6", "null"], + ], + ); +} diff --git a/crates/grafeo-engine/tests/cdc_session_mutations.rs b/crates/grafeo-engine/tests/cdc_session_mutations.rs index 1f79e66a2..86de6fa60 100644 --- a/crates/grafeo-engine/tests/cdc_session_mutations.rs +++ b/crates/grafeo-engine/tests/cdc_session_mutations.rs @@ -17,7 +17,7 @@ use std::collections::HashMap; -use grafeo_common::types::Value; +use grafeo_common::types::{NodeId, Value}; use grafeo_engine::cdc::{ChangeKind, EntityId}; use grafeo_engine::{Config, GrafeoDB}; @@ -567,6 +567,102 @@ fn updates_to_a_node_created_in_the_same_transaction_fold_into_its_create() { ); } +/// Every event names the graph its entity is in, and `history` reads the +/// graph of the caller: the database the default graph, a session its current +/// graph, though the two nodes may share an id. +#[test] +fn events_name_their_graph() { + let db = db(); + db.create_graph("g").unwrap(); + let session = db.session(); + session.execute("INSERT (:InDefault {a: 1})").unwrap(); + session.use_graph("g"); + session.execute("INSERT (:InG {b: 1})").unwrap(); + + let id_of = |query: &str| -> u64 { + match session.execute(query).unwrap().rows()[0][0] { + Value::Int64(id) => u64::try_from(id).unwrap(), + ref other => panic!("expected an id, got {other:?}"), + } + }; + let in_g = id_of("MATCH (n:InG) RETURN id(n)"); + let in_g_history = session.history(NodeId::new(in_g)).unwrap(); + assert_eq!(in_g_history.len(), 1, "{in_g_history:?}"); + assert_eq!(in_g_history[0].graph.as_deref(), Some("g")); + assert_eq!(in_g_history[0].labels, Some(vec!["InG".to_string()])); + + session.use_graph("default"); + let in_default = id_of("MATCH (n:InDefault) RETURN id(n)"); + for history in [ + db.history(NodeId::new(in_default)).unwrap(), + session.history(NodeId::new(in_default)).unwrap(), + ] { + assert_eq!(history.len(), 1, "{history:?}"); + assert_eq!(history[0].graph, None); + assert_eq!(history[0].labels, Some(vec!["InDefault".to_string()])); + } + + let mut graphs: Vec> = db + .changes_between( + grafeo_common::types::EpochId::new(0), + grafeo_common::types::EpochId::new(u64::MAX), + ) + .unwrap() + .into_iter() + .map(|event| event.graph) + .collect(); + graphs.sort(); + assert_eq!(graphs, [None, Some("g".to_string())]); +} + +/// The fold stays within one graph. A transaction creates a node in the +/// default graph and one in a named graph, which number their nodes alike, +/// and changes the second: each node gets its own create event, and only the +/// second shows the change. +#[test] +fn creates_in_two_graphs_fold_separately() { + let db = db(); + db.create_graph("g").unwrap(); + let mut session = db.session(); + + session.begin_transaction().unwrap(); + session.execute("INSERT (:InDefault {a: 1})").unwrap(); + session.use_graph("g"); + session.execute("INSERT (:InG {b: 1})").unwrap(); + session.execute("MATCH (n:InG) SET n.c = 99").unwrap(); + session.commit().unwrap(); + + let changes = db + .changes_between( + grafeo_common::types::EpochId::new(0), + grafeo_common::types::EpochId::new(u64::MAX), + ) + .unwrap(); + let creates: Vec<_> = changes + .iter() + .filter(|e| e.kind == ChangeKind::Create) + .map(|e| (e.labels.clone().unwrap_or_default(), e.after.clone())) + .collect(); + assert_eq!(changes.len(), 2, "{changes:?}"); + assert!( + creates.contains(&( + vec!["InDefault".to_string()], + Some(HashMap::from([("a".to_string(), Value::Int64(1))])) + )), + "{changes:?}" + ); + assert!( + creates.contains(&( + vec!["InG".to_string()], + Some(HashMap::from([ + ("b".to_string(), Value::Int64(1)), + ("c".to_string(), Value::Int64(99)), + ])) + )), + "{changes:?}" + ); +} + /// Create and delete events say what was created or deleted: a node's labels, /// an edge's type and endpoints, and on a delete the last properties. #[test] diff --git a/crates/grafeo-engine/tests/compact_sessions.rs b/crates/grafeo-engine/tests/compact_sessions.rs index 839df2cca..ab1c6be31 100644 --- a/crates/grafeo-engine/tests/compact_sessions.rs +++ b/crates/grafeo-engine/tests/compact_sessions.rs @@ -148,6 +148,103 @@ fn a_wal_directory_keeps_writes_after_compact() { db.close().unwrap(); } +/// Direct calls from several threads on a compacted database: the WAL ends +/// with the state they left, so a reopen reads what memory held, never an +/// older value that one call logged after another call's newer one. +#[cfg(feature = "wal")] +#[test] +fn concurrent_writes_after_compact_replay_to_the_last_state() { + use grafeo_engine::config::StorageFormat; + + let dir = tempfile::tempdir().unwrap(); + let config = || { + Config::persistent(dir.path().join("db")).with_storage_format(StorageFormat::WalDirectory) + }; + let values = |db: &GrafeoDB| { + db.execute("MATCH (c:Counter) RETURN id(c), c.v ORDER BY id(c)") + .unwrap() + .rows() + .to_vec() + }; + let expected = { + let mut db = GrafeoDB::with_config(config()).unwrap(); + db.execute("INSERT (:Seed)").unwrap(); + db.compact().unwrap(); + let counters: Vec<_> = (0..4) + .map(|_| { + db.create_node_with_props(&["Counter"], [("v", Value::Int64(0))]) + .unwrap() + }) + .collect(); + std::thread::scope(|scope| { + for thread in 0..8_i64 { + let (db, counters) = (&db, &counters); + scope.spawn(move || { + for step in 0..200_i64 { + let counter = counters[usize::try_from(step + thread).unwrap() % 4]; + let value = Value::Int64(thread * 1_000 + step); + // Writes to one node conflict (each call is a + // transaction here): retry, a bounded number of times. + let mut attempts = 0; + while let Err(error) = db.set_node_property(counter, "v", value.clone()) { + attempts += 1; + assert!( + error.to_string().contains("conflict") && attempts < 10_000, + "{error}" + ); + std::thread::yield_now(); + } + } + }); + } + }); + let expected = values(&db); + db.close().unwrap(); + expected + }; + let db = GrafeoDB::with_config(config()).unwrap(); + assert_eq!(values(&db), expected); + db.close().unwrap(); +} + +/// Compacting again merges the overlay into a fresh base: inserts, updates +/// and deletes since the first `compact()` stay, and the database stays +/// writable. (The bindings have no `recompact()`; this is how they merge.) +#[test] +fn compacting_again_keeps_the_overlay_writes() { + let mut db = GrafeoDB::new_in_memory(); + db.execute("INSERT (:Person {name: 'Alix', city: 'Paris'})-[:KNOWS]->(:Person {name: 'Gus', city: 'Berlin'})") + .unwrap(); + db.compact().unwrap(); + db.execute("INSERT (:Person {name: 'Vincent', city: 'Prague'})") + .unwrap(); + db.execute("MATCH (p:Person {name: 'Alix'}) SET p.city = 'Amsterdam'") + .unwrap(); + db.execute("MATCH (p:Person {name: 'Gus'}) DETACH DELETE p") + .unwrap(); + let cities = |db: &GrafeoDB| { + db.execute("MATCH (p:Person) RETURN p.name, p.city ORDER BY p.name") + .unwrap() + .rows() + .to_vec() + }; + let before = cities(&db); + assert_eq!(before.len(), 2); + + db.compact().unwrap(); + assert_eq!(cities(&db), before); + db.execute("INSERT (:Person {name: 'Mia', city: 'Barcelona'})") + .unwrap(); + assert_eq!( + names(&db), + [ + Value::from("Alix"), + Value::from("Mia"), + Value::from("Vincent") + ] + ); +} + /// Direct calls and queries after `compact()` produce change events. #[cfg(feature = "cdc")] #[test] @@ -207,3 +304,25 @@ fn the_selected_graph_holds_after_compact() { db.set_current_graph(None).unwrap(); assert_eq!(names(&db), [Value::from("Alix")]); } + +/// `compact()` keeps a database opened read-only read-only: writes fail as +/// before, and `close()` has nothing to write back. +#[cfg(feature = "grafeo-file")] +#[test] +fn compact_keeps_a_read_only_database_read_only() { + let dir = tempfile::tempdir().unwrap(); + let path = dir.path().join("people.grafeo"); + { + let db = GrafeoDB::open(&path).unwrap(); + db.execute("INSERT (:Person {name: 'Alix'})").unwrap(); + db.close().unwrap(); + } + + let mut db = GrafeoDB::open_read_only(&path).unwrap(); + db.compact().unwrap(); + assert!(db.is_read_only()); + assert!(db.execute("INSERT (:Person {name: 'Gus'})").is_err()); + assert!(db.create_node(&["Person"]).is_err()); + assert_eq!(names(&db), [Value::from("Alix")]); + db.close().unwrap(); +} diff --git a/crates/grafeo-engine/tests/coverage_schema_ddl.rs b/crates/grafeo-engine/tests/coverage_schema_ddl.rs index c8dc90c7a..c3345f640 100644 --- a/crates/grafeo-engine/tests/coverage_schema_ddl.rs +++ b/crates/grafeo-engine/tests/coverage_schema_ddl.rs @@ -6,6 +6,7 @@ //! cargo test -p grafeo-engine --test coverage_schema_ddl //! ``` +use grafeo_common::types::Value; use grafeo_engine::GrafeoDB; // --------------------------------------------------------------------------- @@ -207,6 +208,41 @@ fn test_create_and_drop_procedure() { session.execute("DROP PROCEDURE get_adults").unwrap(); } +/// A pattern in a procedure body that comes back to a node it bound is a +/// cycle there too: the body skipped that check (it is planned without the +/// session's optimizer), so every path of three edges counted. +#[test] +fn test_procedure_body_closes_cycles() { + let db = GrafeoDB::new_in_memory(); + let session = db.session(); + session + .execute( + "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), \ + (vincent:Person {name: 'Vincent'}), (jules:Person {name: 'Jules'}), \ + (mia:Person {name: 'Mia'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(vincent), \ + (vincent)-[:KNOWS]->(alix), (jules)-[:KNOWS]->(mia), (alix)-[:KNOWS]->(jules)", + ) + .unwrap(); + session + .execute( + "CREATE PROCEDURE triangles() RETURNS (name STRING) AS { \ + MATCH (a:Person)-[:KNOWS]->(b)-[:KNOWS]->(c)-[:KNOWS]->(a) RETURN a.name AS name }", + ) + .unwrap(); + let result = session + .execute("CALL triangles() YIELD name RETURN name ORDER BY name") + .unwrap(); + let names: Vec = result.rows().iter().map(|row| row[0].clone()).collect(); + assert_eq!( + names, + [ + Value::from("Alix"), + Value::from("Gus"), + Value::from("Vincent") + ] + ); +} + #[test] fn test_create_or_replace_procedure() { let db = GrafeoDB::new_in_memory(); diff --git a/crates/grafeo-engine/tests/cyclic_patterns.rs b/crates/grafeo-engine/tests/cyclic_patterns.rs index c49364e1a..82dc1810a 100644 --- a/crates/grafeo-engine/tests/cyclic_patterns.rs +++ b/crates/grafeo-engine/tests/cyclic_patterns.rs @@ -78,3 +78,172 @@ fn a_triangle_of_comma_patterns_matches_each_triangle_once() { triangles() ); } + +/// A path that returns to an earlier variable closes the cycle: it matches +/// the rows of the same path ending in a fresh variable filtered to equal it. +#[test] +fn a_path_back_to_an_earlier_variable_closes_the_cycle() { + let db = cycles(); + + // The 2-cycle 1 <-> 2, from both ends. + assert_eq!( + rows(db.execute("MATCH (a)-[:K]->(b)-[:K]->(a) RETURN a.n, b.n")), + vec![vec![1, 2], vec![2, 1]] + ); + assert_eq!( + rows(db.execute("MATCH (a)-[:K]->(b)-[:K]->(c)-[:K]->(a) RETURN a.n, b.n, c.n")), + triangles() + ); + // The same cycle closed by a second MATCH. + assert_eq!( + rows(db.execute("MATCH (a)-[:K]->(b) MATCH (b)-[:K]->(a) RETURN a.n, b.n")), + vec![vec![1, 2], vec![2, 1]] + ); + // Each form agrees with the fresh variable and an equality filter. + assert_eq!( + rows( + db.execute("MATCH (a)-[:K]->(b)-[:K]->(c)-[:K]->(d) WHERE d = a RETURN a.n, b.n, c.n") + ), + triangles() + ); +} + +/// A variable-length hop back to an earlier variable closes the cycle too: +/// the same rows as ending in a fresh variable filtered to equal it. +#[test] +fn a_variable_length_path_back_to_an_earlier_variable_closes_the_cycle() { + let db = cycles(); + let closed = rows(db.execute("MATCH (a)-[:K]->(b)-[:K]->{1,2}(a) RETURN a.n, b.n")); + // One hop back over the 2-cycle, two hops back around each triangle. + assert_eq!( + closed, + vec![ + vec![1, 2], + vec![1, 2], + vec![2, 1], + vec![2, 3], + vec![2, 3], + vec![3, 1], + vec![3, 4], + vec![4, 2], + ] + ); + assert_eq!( + closed, + rows(db.execute("MATCH (a)-[:K]->(b)-[:K]->{1,2}(d) WHERE d = a RETURN a.n, b.n")) + ); +} + +#[cfg(feature = "cypher")] +#[test] +fn a_cypher_path_back_to_an_earlier_variable_closes_the_cycle() { + let db = cycles(); + assert_eq!( + rows(db.execute_cypher("MATCH (a)-[:K]->(b)-[:K]->(a) RETURN a.n, b.n")), + vec![vec![1, 2], vec![2, 1]] + ); + assert_eq!( + rows(db.execute_cypher("MATCH (a)-[:K]->(b)-[:K]->(c)-[:K]->(a) RETURN a.n, b.n, c.n")), + triangles() + ); +} + +#[cfg(feature = "sql-pgq")] +#[test] +fn a_sql_pgq_path_back_to_an_earlier_variable_closes_the_cycle() { + let db = cycles(); + assert_eq!( + rows(db.session().execute_sql( + "SELECT an, bn FROM GRAPH_TABLE (MATCH (a)-[:K]->(b)-[:K]->(a) COLUMNS (a.n AS an, b.n AS bn))" + )), + vec![vec![1, 2], vec![2, 1]] + ); +} + +/// A subquery that imports both ends, by name or with `WITH *`, closes the +/// cycle on them: only the 2-cycle 1 <-> 2 has an edge back. +#[test] +fn a_subquery_closes_the_cycle_on_the_nodes_it_imports() { + let db = cycles(); + for import in ["WITH a, b", "WITH *"] { + let query = format!( + "MATCH (a)-[:K]->(b) CALL {{ {import} MATCH (b)-[:K]->(a) RETURN count(*) AS back }} \ + RETURN a.n, b.n, back" + ); + let want = vec![ + vec![1, 2, 1], + vec![2, 1, 1], + vec![2, 3, 0], + vec![3, 1, 0], + vec![3, 4, 0], + vec![4, 2, 0], + ]; + assert_eq!(rows(db.execute(&query)), want, "GQL: {query}"); + #[cfg(feature = "cypher")] + assert_eq!(rows(db.execute_cypher(&query)), want, "Cypher: {query}"); + } +} + +/// After `WITH b` the earlier `a` is out of scope: the second `MATCH` binds a +/// new `a`, so every edge out of `b` counts, cycle or not. +#[cfg(feature = "cypher")] +#[test] +fn a_variable_out_of_scope_is_bound_again() { + let db = cycles(); + let query = "MATCH (a)-[:K]->(b) WITH b MATCH (b)-[:K]->(a) RETURN b.n, a.n"; + assert_eq!(rows(db.execute(query)), rows(db.execute_cypher(query))); + assert_eq!( + rows(db.execute(query)), + vec![ + vec![1, 2], + vec![1, 2], + vec![2, 1], + vec![2, 1], + vec![2, 3], + vec![2, 3], + vec![3, 1], + vec![3, 4], + vec![4, 2], + ] + ); +} + +/// A user variable spelled like the name the rewrite gives the closing node +/// still closes its cycle. +#[test] +fn a_cycle_on_a_variable_named_like_an_internal_one_closes() { + let db = cycles(); + assert_eq!( + rows(db.execute( + "MATCH (_cycle_end_0)-[:K]->(b)-[:K]->(_cycle_end_0) RETURN _cycle_end_0.n, b.n" + )), + vec![vec![1, 2], vec![2, 1]] + ); +} + +/// The closing node's name is not one that a later clause binds: that clause +/// scans its own nodes, so each 2-cycle row pairs with all four nodes. +#[test] +fn a_later_variable_named_like_an_internal_one_binds_on_its_own() { + let db = cycles(); + assert_eq!( + rows(db.execute("MATCH (a)-[:K]->(b)-[:K]->(a) MATCH (_cycle_end_0:P) RETURN count(*)")), + vec![vec![8]] + ); +} + +/// An anonymous node is a node of its own, also next to a user variable +/// spelled like the names anonymous nodes get: every edge out of `_v0` counts. +#[cfg(feature = "sql-pgq")] +#[test] +fn a_sql_pgq_anonymous_node_is_not_a_user_variable() { + let db = cycles(); + assert_eq!( + rows( + db.session().execute_sql( + "SELECT n FROM GRAPH_TABLE (MATCH (_v0)-[:K]->() COLUMNS (_v0.n AS n))" + ) + ), + vec![vec![1], vec![2], vec![2], vec![3], vec![3], vec![4]] + ); +} diff --git a/crates/grafeo-engine/tests/direct_api.rs b/crates/grafeo-engine/tests/direct_api.rs index e3dfce70a..9714ac68c 100644 --- a/crates/grafeo-engine/tests/direct_api.rs +++ b/crates/grafeo-engine/tests/direct_api.rs @@ -376,6 +376,18 @@ fn a_direct_write_conflicts_with_an_open_transaction() { assert_eq!(city(alix), Some(Value::from("Paris"))); } +/// Runs `attempt` until it succeeds, yielding between tries: a write +/// conflict that never clears fails the test instead of hanging it. +fn until_done(mut attempt: impl FnMut() -> bool) { + for _ in 0..100_000 { + if attempt() { + return; + } + std::thread::yield_now(); + } + panic!("a write conflict never cleared"); +} + /// Direct writes from several threads next to transactions, all writing one /// shared node too: nothing hangs, every committed write lands, nothing of a /// rolled-back transaction remains, and a direct write that meets an open @@ -392,7 +404,7 @@ fn direct_writes_and_transactions_run_side_by_side() { let db = &db; scope.spawn(move || { for round in 0..ROUNDS { - loop { + until_done(|| { let mut session = db.session(); session.begin_transaction().unwrap(); let kept = round % 3 != 0; @@ -403,22 +415,23 @@ fn direct_writes_and_transactions_run_side_by_side() { .is_ok(); if !kept { session.rollback().unwrap(); - break; + return true; } if wrote_hub && session.commit().is_ok() { - break; + return true; } let _ = session.rollback(); - } + false + }); } }); scope.spawn(move || { for _ in 0..ROUNDS { db.create_node(&["Mark"]).unwrap(); - while db - .set_node_property(hub, "by", Value::from(format!("direct {thread}"))) - .is_err() - {} + until_done(|| { + db.set_node_property(hub, "by", Value::from(format!("direct {thread}"))) + .is_ok() + }); } }); } diff --git a/crates/grafeo-engine/tests/direct_reads.rs b/crates/grafeo-engine/tests/direct_reads.rs index 67f127cdc..3195df255 100644 --- a/crates/grafeo-engine/tests/direct_reads.rs +++ b/crates/grafeo-engine/tests/direct_reads.rs @@ -66,6 +66,107 @@ fn direct_reads_see_the_compacted_data() { assert!(validation.errors.is_empty(), "{:?}", validation.errors); } +/// The schema views count the labels, edge types and property keys of the +/// compacted data next to those written since. +#[cfg(feature = "compact-store")] +#[test] +fn schema_views_see_the_compacted_data() { + use grafeo_engine::SchemaInfo; + + let mut db = GrafeoDB::new_in_memory(); + let alix = person(&db, "Alix"); + let amsterdam = db + .create_node_with_props(&["City"], [("population", Value::Int64(921_000))]) + .unwrap(); + db.create_edge(alix, amsterdam, "LIVES_IN").unwrap(); + db.compact().unwrap(); + let gus = person(&db, "Gus"); + db.create_edge_with_props(alix, gus, "KNOWS", [("since", Value::Int64(2020))]) + .unwrap(); + + // Person and City, LIVES_IN and KNOWS, name, population and since. + assert_eq!( + ( + db.label_count(), + db.edge_type_count(), + db.property_key_count() + ), + (2, 2, 3) + ); + let stats = db.detailed_stats(); + assert_eq!( + ( + stats.label_count, + stats.edge_type_count, + stats.property_key_count + ), + (2, 2, 3) + ); + + let SchemaInfo::Lpg(schema) = db.schema() else { + panic!("expected an LPG schema"); + }; + let mut labels: Vec<_> = schema + .labels + .iter() + .map(|label| (label.name.as_str(), label.count)) + .collect(); + labels.sort_unstable(); + assert_eq!(labels, [("City", 1), ("Person", 2)]); + let mut edge_types: Vec<_> = schema + .edge_types + .iter() + .map(|edge_type| (edge_type.name.as_str(), edge_type.count)) + .collect(); + edge_types.sort_unstable(); + assert_eq!(edge_types, [("KNOWS", 1), ("LIVES_IN", 1)]); + let mut keys = schema.property_keys; + keys.sort_unstable(); + assert_eq!(keys, ["name", "population", "since"]); +} + +/// The schema and the counts show committed data only: an open +/// transaction's node and edge appear once it commits, and never after it +/// rolls back. +#[test] +fn schema_counts_committed_data_only() { + use grafeo_engine::SchemaInfo; + + let db = GrafeoDB::new_in_memory(); + db.execute("INSERT (:Person {name: 'Alix'})-[:KNOWS]->(:Person {name: 'Gus'})") + .unwrap(); + let counts = |db: &GrafeoDB| { + let SchemaInfo::Lpg(schema) = db.schema() else { + panic!("expected an LPG schema"); + }; + let people = schema + .labels + .iter() + .find(|label| label.name == "Person") + .map_or(0, |label| label.count); + let knows = schema + .edge_types + .iter() + .find(|edge_type| edge_type.name == "KNOWS") + .map_or(0, |edge_type| edge_type.count); + let stats = db.detailed_stats(); + (people, knows, stats.node_count, stats.edge_count) + }; + let insert = "MATCH (a:Person {name: 'Alix'}) INSERT (a)-[:KNOWS]->(:Person {name: 'Django'})"; + + let mut session = db.session(); + session.begin_transaction().unwrap(); + session.execute(insert).unwrap(); + assert_eq!(counts(&db), (2, 1, 2, 1), "while the transaction is open"); + session.rollback().unwrap(); + assert_eq!(counts(&db), (2, 1, 2, 1), "after a rollback"); + + session.begin_transaction().unwrap(); + session.execute(insert).unwrap(); + session.commit().unwrap(); + assert_eq!(counts(&db), (3, 2, 3, 2), "after the commit"); +} + #[test] fn direct_reads_on_an_external_store() { let store = Arc::new(LpgStore::new().unwrap()); @@ -85,6 +186,14 @@ fn direct_reads_on_an_external_store() { assert_eq!(name(&db, alix), Some(Value::from("Alix"))); assert_eq!(db.get_edge(knows).map(|edge| edge.dst), Some(gus)); assert_eq!(db.get_node_labels(alix), Some(vec!["Person".to_string()])); + assert_eq!( + ( + db.label_count(), + db.edge_type_count(), + db.property_key_count() + ), + (1, 1, 1) + ); assert_eq!( db.graph("default") .unwrap() diff --git a/crates/grafeo-engine/tests/entity_lists.rs b/crates/grafeo-engine/tests/entity_lists.rs index db77a8850..0608ecbf4 100644 --- a/crates/grafeo-engine/tests/entity_lists.rs +++ b/crates/grafeo-engine/tests/entity_lists.rs @@ -7,6 +7,8 @@ //! cargo test -p grafeo-engine --all-features --test entity_lists //! ``` +#![cfg(all(feature = "lpg", feature = "gql"))] + use grafeo_common::types::{PropertyKey, Value}; use grafeo_engine::GrafeoDB; @@ -110,6 +112,47 @@ fn with_keeps_an_edge_list() { assert_eq!(relationships(&row[2]), vec![contains(1), contains(2)]); } +/// One item of an edge or node list stays an edge or a node through WITH: +/// its properties, type, id and labels can be read, and RETURN gives its map. +#[test] +fn with_keeps_an_item_of_an_edge_or_node_list() { + let db = chain(); + let row = row(db.execute( + "MATCH p = (d:Dir {name: 'a'})-[r]->{1,3}(x:File) \ + WITH head(r) AS first, r[1] AS second, last(nodes(p)) AS file, x \ + RETURN first, second.w, type(second), file, file.name, id(file) = id(x), labels(file)", + )); + + assert_eq!(field(&row[0], "w"), Value::Int64(1)); + assert_eq!(row[1], Value::Int64(2)); + assert_eq!(row[2], Value::from("CONTAINS")); + assert_eq!(field(&row[3], "name"), Value::from("f")); + assert_eq!(row[4], Value::from("f")); + assert_eq!(row[5], Value::Bool(true)); + assert_eq!(row[6], Value::List(vec![Value::from("File")].into())); +} + +/// A variable named with a leading underscore is the query's own: it binds +/// the list of relationships like any other name. An anonymous edge with a +/// property map still checks the map on every hop. +#[test] +fn an_underscore_edge_variable_binds_its_relationships() { + let db = chain(); + let gql = row(db.execute("MATCH (d:Dir {name: 'a'})-[_r]->{1,3}(x:File) RETURN _r")); + assert_eq!(relationships(&gql[0]), vec![contains(1), contains(2)]); + #[cfg(feature = "cypher")] + { + let cypher = + row(db.execute_cypher("MATCH (d:Dir {name: 'a'})-[_r*1..3]->(x:File) RETURN _r")); + assert_eq!(relationships(&cypher[0]), vec![contains(1), contains(2)]); + let first_hop = + row(db.execute_cypher("MATCH (d:Dir {name: 'a'})-[*1..3 {w: 1}]->(x) RETURN x.name")); + assert_eq!(first_hop, vec![Value::from("b")]); + } + let first_hop = row(db.execute("MATCH (d:Dir {name: 'a'})-[{w: 1}]->{1,3}(x) RETURN x.name")); + assert_eq!(first_hop, vec![Value::from("b")]); +} + /// A zero-length match binds the edge variable to an empty list. #[test] fn a_zero_length_match_binds_an_empty_list() { diff --git a/crates/grafeo-engine/tests/gremlin.rs b/crates/grafeo-engine/tests/gremlin.rs index 17c27702c..2f14dbca4 100644 --- a/crates/grafeo-engine/tests/gremlin.rs +++ b/crates/grafeo-engine/tests/gremlin.rs @@ -1019,6 +1019,24 @@ fn test_parameterized_query() { assert_eq!(result.row_count(), 1); } +/// Without a parameter map, a parameter the query reads is missing; it used +/// to reach the planner unset. +#[test] +fn test_unsupplied_parameter_is_missing() { + let db = create_social_network(); + let error = db + .execute_gremlin("g.V().has('name', $name).values('name')") + .unwrap_err() + .to_string(); + assert!(error.contains("Missing parameter: $name"), "{error}"); + let mut params = std::collections::HashMap::new(); + params.insert("name".to_string(), Value::String("Alix".into())); + let result = db + .execute_gremlin_with_params("g.V().has('name', $name).values('name')", params) + .unwrap(); + assert_eq!(result.rows(), [vec![Value::String("Alix".into())]]); +} + // ============================================================================ // Step-Level and() Filter // ============================================================================ diff --git a/crates/grafeo-engine/tests/index_persistence.rs b/crates/grafeo-engine/tests/index_persistence.rs index 7ea5f7dcf..c51516aa2 100644 --- a/crates/grafeo-engine/tests/index_persistence.rs +++ b/crates/grafeo-engine/tests/index_persistence.rs @@ -4,7 +4,7 @@ //! names that `DROP INDEX` and `DROP CONSTRAINT` use. //! //! ```bash -//! cargo test -p grafeo-engine --features full --test index_persistence +//! cargo test -p grafeo-engine --all-features --test index_persistence //! ``` #![cfg(all(feature = "lpg", feature = "gql"))] @@ -93,6 +93,45 @@ fn reopening_a_file_keeps_indexes_and_constraints() { db.close().unwrap(); } +/// 200 `Graph:File` nodes `n0`..`n199` in a chain of `T` edges, with a +/// property index on `id`, written to a `.grafeo` file and reopened. +#[cfg(feature = "grafeo-file")] +fn reopened_chain(dir: &tempfile::TempDir) -> GrafeoDB { + let path = dir.path().join("chain.grafeo"); + let db = GrafeoDB::open(&path).unwrap(); + db.execute("UNWIND range(0, 199) AS i INSERT (:Graph:File {id: 'n' + toString(i), i: i})") + .unwrap(); + db.execute("MATCH (a:File), (b:File) WHERE b.i = a.i + 1 INSERT (a)-[:T]->(b)") + .unwrap(); + db.create_property_index("id"); + db.close().unwrap(); + GrafeoDB::open(&path).unwrap() +} + +/// #459: on a reopened file a point lookup plus one hop seeks the property +/// index, as in memory, instead of scanning every node (0.5.43 lost the index +/// on reopen and scanned). +#[cfg(feature = "grafeo-file")] +#[test] +fn a_point_lookup_on_a_reopened_file_seeks_the_index() { + let dir = tempfile::tempdir().unwrap(); + let db = reopened_chain(&dir); + assert!(db.has_property_index("id")); + for query in [ + "MATCH (s {id: 'n10'})-[:T]->(d) RETURN d.id", + "MATCH (s:File {id: 'n10'})-[:T]->(d) RETURN d.id", + ] { + let profile = plan(&db.execute(&format!("PROFILE {query}")).unwrap()); + assert!(profile.contains("NodeList (s.id Eq"), "{query}: {profile}"); + assert_eq!( + db.execute(query).unwrap().rows(), + &[vec![Value::from("n11")]], + "{query}" + ); + } + db.close().unwrap(); +} + /// The first row's first column of a PROFILE: the plan. fn plan(result: &grafeo_engine::database::QueryResult) -> String { match &result.rows()[0][0] { diff --git a/crates/grafeo-engine/tests/node_seek.rs b/crates/grafeo-engine/tests/node_seek.rs index 516ae1718..8518fe00b 100644 --- a/crates/grafeo-engine/tests/node_seek.rs +++ b/crates/grafeo-engine/tests/node_seek.rs @@ -116,6 +116,91 @@ fn a_seek_returns_what_a_scan_returns() { ); } +/// A literal key is looked up in the index once; of what it finds, only the +/// visible nodes with the pattern's label count (`:Other {id: 'd1'}` shares the +/// key), also for labels set or removed in the transaction. +#[test] +fn a_literal_key_keeps_the_nodes_with_the_label() { + let (sought, scanned) = (docs(true), docs(false)); + for query in [ + "MATCH (n:Doc {id: 'd1'}) RETURN n.n", + "MATCH (n:Other {id: 'd1'}) RETURN n.id", + "MATCH (n:Doc) WHERE n.id IN ['d1', 'd2', 'x'] RETURN n.id", + "MATCH (n:Other) WHERE n.id IN ['d1', 'd2'] RETURN n.id", + ] { + assert_eq!( + rows(&sought, query, &[]), + rows(&scanned, query, &[]), + "{query}" + ); + } + assert_eq!( + rows(&sought, "MATCH (n:Doc {id: 'd1'}) RETURN n.n", &[]), + [vec![Value::Int64(1)]] + ); + + let mut session = sought.session(); + session.begin_transaction().unwrap(); + session + .execute("MATCH (n:Other {id: 'd1'}) SET n:Doc") + .unwrap(); + session + .execute("MATCH (n:Doc {id: 'd2'}) REMOVE n:Doc") + .unwrap(); + let count = |query: &str| session.execute(query).unwrap().rows().len(); + assert_eq!(count("MATCH (n:Doc {id: 'd1'}) RETURN n"), 2); + assert_eq!(count("MATCH (n:Doc {id: 'd2'}) RETURN n"), 0); + assert_eq!( + count("MATCH (n:Doc) WHERE n.id IN ['d1', 'd2'] RETURN n"), + 2 + ); + session.rollback().unwrap(); + assert_eq!( + rows(&sought, "MATCH (n:Doc {id: 'd2'}) RETURN n.n", &[]), + [vec![Value::Int64(2)]] + ); +} + +/// A labeled point lookup costs about what an unlabeled one does, however many +/// nodes have the label: the label is checked on the index's results, not by +/// collecting every node with it (which took 3 ms per lookup at 60,000 nodes). +#[cfg(not(debug_assertions))] +#[test] +fn a_labeled_point_lookup_does_not_grow_with_the_label() { + use std::time::{Duration, Instant}; + + let db = GrafeoDB::new_in_memory(); + // In batches: with the `tiered-storage` feature one transaction can create + // about 20,000 nodes until tiered storage is wired in (#433). + for start in (0..60_000).step_by(15_000) { + db.execute(&format!( + "UNWIND range({start}, {}) AS i INSERT (:Graph:File {{id: 'n' + toString(i)}})", + start + 14_999 + )) + .unwrap(); + } + db.create_property_index("id"); + // The fastest of five batches, to keep a busy machine out of the ratio. + let time = |query: &str| -> Duration { + (0..5) + .map(|_| { + let start = Instant::now(); + for _ in 0..100 { + db.execute(query).unwrap(); + } + start.elapsed() + }) + .min() + .unwrap() + }; + let unlabeled = time("MATCH (s {id: 'n10'}) RETURN s.id"); + let labeled = time("MATCH (s:File {id: 'n10'}) RETURN s.id"); + assert!( + labeled < unlabeled * 5, + "labeled {labeled:?} vs unlabeled {unlabeled:?} per 100 lookups" + ); +} + #[test] fn a_key_from_the_row_is_looked_up_in_the_index() { let db = docs(true); diff --git a/crates/grafeo-engine/tests/parameterized_queries.rs b/crates/grafeo-engine/tests/parameterized_queries.rs index e3974dac8..94fc3dfd7 100644 --- a/crates/grafeo-engine/tests/parameterized_queries.rs +++ b/crates/grafeo-engine/tests/parameterized_queries.rs @@ -24,6 +24,36 @@ fn count(db: &GrafeoDB, label: &str) -> Value { .clone() } +/// A write that uses a parameter nobody supplied fails before it writes: +/// it used to store the text "$e". Through `execute` (no parameter map at +/// all), an empty map, Cypher and GraphQL (a declared variable without a +/// default). +#[test] +fn an_unsupplied_parameter_fails_before_writing() { + let db = GrafeoDB::new_in_memory(); + for result in [ + db.execute("INSERT (:P {e: $e})"), + db.execute_with_params("INSERT (:P {e: $e})", HashMap::new()), + #[cfg(feature = "cypher")] + db.execute_cypher("CREATE (:P {e: $e})"), + #[cfg(feature = "graphql")] + db.execute_graphql("mutation ($e: String) { createP(e: $e) { e } }"), + ] { + let error = result.unwrap_err().to_string(); + assert!(error.contains("Missing parameter: $e"), "{error}"); + } + assert_eq!(count(&db, "P"), Value::Int64(0)); + + // EXPLAIN shows the plan without the values; PROFILE runs it, so it fails. + db.execute("EXPLAIN INSERT (:P {e: $e})").unwrap(); + let error = db + .execute("PROFILE INSERT (:P {e: $e})") + .unwrap_err() + .to_string(); + assert!(error.contains("Missing parameter: $e"), "{error}"); + assert_eq!(count(&db, "P"), Value::Int64(0)); +} + #[test] fn constraints_hold_for_parameterized_writes() { let db = GrafeoDB::new_in_memory(); @@ -151,6 +181,39 @@ fn a_cached_plan_does_not_keep_the_values() { } } +/// An empty map fills in nothing: a statement without parameters reuses its +/// optimized plan like the same call without a map, and one that names a +/// parameter still fails as missing it. +#[test] +fn an_empty_parameter_map_uses_the_cached_plan() { + let db = GrafeoDB::new_in_memory(); + db.execute("INSERT (:Person {name: 'Alix'})").unwrap(); + let query = "MATCH (p:Person) RETURN p.name"; + let hits = || db.query_cache().stats().optimized_hits; + let before = hits(); + for _ in 0..3 { + let rows = db + .execute_with_params(query, HashMap::new()) + .unwrap() + .rows() + .to_vec(); + assert_eq!(rows, [vec![Value::from("Alix")]]); + } + assert_eq!( + hits() - before, + 2, + "the second and third call reuse the plan" + ); + + let error = db + .execute_with_params( + "MATCH (p:Person) WHERE p.name = $name RETURN p.name", + HashMap::new(), + ) + .unwrap_err(); + assert!(error.to_string().contains("$name"), "{error}"); +} + #[test] fn explain_and_profile_take_parameters() { let db = GrafeoDB::new_in_memory(); @@ -229,5 +292,5 @@ fn dotted_access_on_a_node_expression_explains_itself() { .unwrap_err() .to_string(); assert!(err.contains("startNode(r) is not a map value"), "{err}"); - assert!(err.contains("read its property"), "{err}"); + assert!(err.contains("bound to a variable in the pattern"), "{err}"); } diff --git a/crates/grafeo-engine/tests/project_coverage.rs b/crates/grafeo-engine/tests/project_coverage.rs index 09454e2f3..e2dd426d0 100644 --- a/crates/grafeo-engine/tests/project_coverage.rs +++ b/crates/grafeo-engine/tests/project_coverage.rs @@ -318,31 +318,37 @@ fn order_by_nulls_last_puts_nulls_at_bottom() { #[test] fn order_by_desc_with_nulls_ordering() { - // DESC combined with an explicit NULLS clause exercises the Descending - // branch of the direction match plus the NullOrder pass-through. The - // underlying sort operator reverses the whole comparison (including null - // position) when direction=Descending, so DESC+NULLS LAST ends up placing - // nulls first. We pin that observed behavior so regressions in either - // plan_sort's mapping or the sort operator's semantics get caught. + // An explicit NULLS clause holds in either direction: DESC NULLS LAST puts + // the nulls after the values, DESC NULLS FIRST before them. (The sort used + // to reverse the null position along with the values for DESC, so both + // came out the other way around.) Pin the full row ordering, not just the + // non-null subsequence. let db = people_graph(); let session = db.session(); + let ages = |order: &str| -> Vec { + session + .execute(&format!( + "MATCH (n:Person) RETURN n.name AS name, n.age AS age ORDER BY age {order}" + )) + .unwrap() + .rows() + .iter() + .map(|row| row[1].clone()) + .collect() + }; - let r = session - .execute( - "MATCH (n:Person) RETURN n.name AS name, n.age AS age \ - ORDER BY age DESC NULLS LAST", - ) - .unwrap(); - - assert_eq!(r.rows().len(), 5); - // The underlying sort operator reverses the comparison including null - // position for DESC, so DESC NULLS LAST currently produces nulls first. - // Pin the full row ordering so any regression in either plan_sort's - // mapping or the sort operator's null handling is caught, not just - // the non-null subsequence. - let ages: Vec = r.rows().iter().map(|row| row[1].clone()).collect(); assert_eq!( - ages, + ages("DESC NULLS LAST"), + vec![ + Value::Int64(40), + Value::Int64(30), + Value::Int64(25), + Value::Null, + Value::Null, + ], + ); + assert_eq!( + ages("DESC NULLS FIRST"), vec![ Value::Null, Value::Null, @@ -350,9 +356,6 @@ fn order_by_desc_with_nulls_ordering() { Value::Int64(30), Value::Int64(25), ], - "DESC NULLS LAST currently places nulls first due to the operator \ - reversing null position along with value comparison; if this test \ - fails the sort semantics changed", ); } diff --git a/crates/grafeo-engine/tests/returned_entities.rs b/crates/grafeo-engine/tests/returned_entities.rs new file mode 100644 index 000000000..93438e5d5 --- /dev/null +++ b/crates/grafeo-engine/tests/returned_entities.rs @@ -0,0 +1,447 @@ +//! Nodes and edges a statement returns stay records through everything that +//! only orders, cuts, deduplicates or combines rows: `ORDER BY`, `LIMIT`, +//! `SKIP`, `DISTINCT`, `UNION`, `OTHERWISE`, `EXCEPT` and `INTERSECT`, in GQL +//! and Cypher. Edges used to come back as `0` after `ORDER BY`, `SKIP` or a +//! cut `LIMIT` (#482), and a later `UNION` or `OTHERWISE` branch returned raw +//! IDs, so `EXCEPT` and `INTERSECT` compared records with IDs. +//! +//! ```bash +//! cargo test -p grafeo-engine --all-features --test returned_entities +//! ``` + +#![cfg(all(feature = "lpg", feature = "gql"))] + +use grafeo_common::types::{PropertyKey, Value}; +use grafeo_engine::{Config, GrafeoDB}; + +/// `(:A {n: 1})-[:K {w: 1}]->(:B {n: 2})`, and the same with n 3, 4 and w 2, +/// and n 5, 6 and w 3. +fn three_edges(config: Config) -> GrafeoDB { + let db = GrafeoDB::with_config(config).unwrap(); + for (a, w, b) in [(1, 1, 2), (3, 2, 4), (5, 3, 6)] { + db.execute(&format!( + "INSERT (:A {{n: {a}}})-[:K {{w: {w}}}]->(:B {{n: {b}}})" + )) + .unwrap(); + } + db +} + +/// A short form of a returned value: `K w=1` for an edge record, `n=1` for +/// a node record, the debug form for anything else. +fn describe(value: &Value) -> String { + match value { + Value::Map(map) => { + let get = |key: &str| map.get(&PropertyKey::new(key)); + match (get("_type"), get("_labels")) { + (Some(Value::String(edge_type)), _) => { + format!( + "{edge_type} w={:?}", + get("w").cloned().unwrap_or(Value::Null) + ) + } + (None, Some(_)) => format!("n={:?}", get("n").cloned().unwrap_or(Value::Null)), + _ => format!("{value:?}"), + } + } + Value::List(items) => { + let items: Vec = items.iter().map(describe).collect(); + format!("[{}]", items.join(", ")) + } + other => format!("{other:?}"), + } +} + +/// The rows of `query` in `language`, each value described. +fn rows(db: &GrafeoDB, language: &str, query: &str) -> Vec> { + let result = match language { + "gql" => db.execute(query), + "cypher" => db.execute_cypher(query), + #[cfg(feature = "gremlin")] + "gremlin" => db.execute_gremlin(query), + other => panic!("unknown language {other}"), + } + .unwrap_or_else(|error| panic!("{language}: {query}: {error}")); + result + .rows() + .iter() + .map(|row| row.iter().map(describe).collect()) + .collect() +} + +fn edges(ws: &[i64]) -> Vec> { + ws.iter().map(|w| vec![format!("K w=Int64({w})")]).collect() +} + +fn nodes(ns: &[i64]) -> Vec> { + ns.iter().map(|n| vec![format!("n=Int64({n})")]).collect() +} + +fn sorted(mut rows: Vec>) -> Vec> { + rows.sort(); + rows +} + +const LANGUAGES: [&str; 2] = ["gql", "cypher"]; + +#[test] +fn edges_through_order_by() { + let db = three_edges(Config::in_memory()); + for language in LANGUAGES { + for (query, expected) in [ + ( + "MATCH (a)-[r]->(b) RETURN r ORDER BY r.w", + edges(&[1, 2, 3]), + ), + ( + "MATCH (a)-[r]->(b) RETURN r ORDER BY a.n DESC", + edges(&[3, 2, 1]), + ), + ( + "MATCH (a)-[r]->(b) RETURN r ORDER BY r.w DESC LIMIT 2", + edges(&[3, 2]), + ), + ( + "MATCH (a)-[r]->(b) RETURN r ORDER BY r.w SKIP 1", + edges(&[2, 3]), + ), + ( + "MATCH (a)-[r]->(b) RETURN DISTINCT r ORDER BY r.w", + edges(&[1, 2, 3]), + ), + ] { + assert_eq!(rows(&db, language, query), expected, "{language}: {query}"); + } + assert_eq!( + rows(&db, language, "MATCH (a)-[r]->(b) RETURN a, r ORDER BY a.n"), + [(1, 1), (3, 2), (5, 3)] + .map(|(n, w)| vec![format!("n=Int64({n})"), format!("K w=Int64({w})")]) + ); + } +} + +/// A chunk cut by `LIMIT` or `SKIP` keeps its records. +#[test] +fn edges_through_limit_and_skip() { + let db = three_edges(Config::in_memory()); + for language in LANGUAGES { + let limited = rows(&db, language, "MATCH (a)-[r]->(b) RETURN r LIMIT 2"); + assert_eq!(limited.len(), 2, "{language}"); + assert!( + limited.iter().all(|row| row[0].starts_with("K w=")), + "{language}: {limited:?}" + ); + let skipped = rows(&db, language, "MATCH (a)-[r]->(b) RETURN r SKIP 1"); + assert_eq!(skipped.len(), 2, "{language}"); + assert!( + skipped.iter().all(|row| row[0].starts_with("K w=")), + "{language}: {skipped:?}" + ); + } +} + +/// A sort key on a RETURN alias is not a result column. +#[test] +fn a_sort_key_on_an_alias_is_not_returned() { + let db = three_edges(Config::in_memory()); + for language in LANGUAGES { + let query = "MATCH (a)-[r]->(b) RETURN r AS e ORDER BY e.w"; + let result = match language { + "gql" => db.execute(query), + _ => db.execute_cypher(query), + } + .unwrap(); + assert_eq!(result.columns, ["e"], "{language}"); + assert_eq!(rows(&db, language, query), edges(&[1, 2, 3]), "{language}"); + } +} + +/// `RETURN *` with `ORDER BY` returns the pattern's variables as records, in +/// order, and no column for the sort key: a property of a returned variable, +/// the variable itself, and a cut or deduplicated result. +#[test] +fn return_star_with_order_by() { + let db = three_edges(Config::in_memory()); + let record = |a: i64, w: i64, b: i64| { + vec![ + format!("n=Int64({a})"), + format!("K w=Int64({w})"), + format!("n=Int64({b})"), + ] + }; + for language in LANGUAGES { + for (query, expected) in [ + ( + "MATCH (a)-[r]->(b) RETURN * ORDER BY r.w DESC", + vec![record(5, 3, 6), record(3, 2, 4), record(1, 1, 2)], + ), + ( + "MATCH (a)-[r]->(b) RETURN * ORDER BY a.n LIMIT 1", + vec![record(1, 1, 2)], + ), + ( + "MATCH (a)-[r]->(b) RETURN * ORDER BY b.n DESC SKIP 1", + vec![record(3, 2, 4), record(1, 1, 2)], + ), + ( + "MATCH (a)-[r]->(b) RETURN DISTINCT * ORDER BY r.w", + vec![record(1, 1, 2), record(3, 2, 4), record(5, 3, 6)], + ), + ( + "MATCH (a)-[r]->(b) RETURN * ORDER BY a DESC", + vec![record(5, 3, 6), record(3, 2, 4), record(1, 1, 2)], + ), + ] { + let result = match language { + "gql" => db.execute(query), + _ => db.execute_cypher(query), + } + .unwrap_or_else(|error| panic!("{language}: {query}: {error}")); + assert_eq!(result.columns, ["a", "r", "b"], "{language}: {query}"); + assert_eq!(rows(&db, language, query), expected, "{language}: {query}"); + } + } +} + +/// Every branch of a UNION returns records, so the branches' rows compare. +#[test] +fn every_union_branch_returns_records() { + let db = three_edges(Config::in_memory()); + for language in LANGUAGES { + for (query, expected) in [ + ( + "MATCH (a)-[r]->(b) WHERE a.n = 1 RETURN r \ + UNION MATCH (a)-[r]->(b) WHERE a.n = 3 RETURN r", + edges(&[1, 2]), + ), + ( + "MATCH (a)-[r]->(b) RETURN r \ + UNION MATCH (a)-[r]->(b) WHERE a.n = 3 RETURN r", + edges(&[1, 2, 3]), + ), + ( + "MATCH (a)-[r]->(b) WHERE a.n = 1 RETURN r \ + UNION ALL MATCH (a)-[r]->(b) WHERE a.n = 1 RETURN r", + edges(&[1, 1]), + ), + ( + "MATCH (a:A) WHERE a.n = 1 RETURN a UNION MATCH (a:A) WHERE a.n = 3 RETURN a", + nodes(&[1, 3]), + ), + ] { + assert_eq!( + sorted(rows(&db, language, query)), + expected, + "{language}: {query}" + ); + } + } +} + +/// GQL's other set operations compare and return records. +#[test] +fn gql_set_operations_on_records() { + let db = three_edges(Config::in_memory()); + for (query, expected) in [ + ( + "MATCH (a)-[r]->(b) RETURN r EXCEPT MATCH (a)-[r]->(b) WHERE a.n = 3 RETURN r", + edges(&[1, 3]), + ), + ( + "MATCH (a)-[r]->(b) RETURN r INTERSECT MATCH (a)-[r]->(b) WHERE a.n = 3 RETURN r", + edges(&[2]), + ), + ( + "MATCH (a)-[r]->(b) WHERE a.n = 99 RETURN r \ + OTHERWISE MATCH (a)-[r]->(b) WHERE a.n = 3 RETURN r", + edges(&[2]), + ), + ( + "MATCH (a:A) RETURN a EXCEPT MATCH (a:A) WHERE a.n = 3 RETURN a", + nodes(&[1, 5]), + ), + ( + "MATCH (a:A) RETURN a INTERSECT MATCH (a:A) WHERE a.n = 3 RETURN a", + nodes(&[3]), + ), + ] { + assert_eq!(sorted(rows(&db, "gql", query)), expected, "{query}"); + } +} + +/// The `shuffle_unordered` test option reorders records without changing them. +#[test] +fn shuffled_results_keep_their_records() { + let db = three_edges(Config::in_memory().with_shuffle_unordered(true)); + for language in LANGUAGES { + assert_eq!( + sorted(rows(&db, language, "MATCH (a)-[r]->(b) RETURN r")), + edges(&[1, 2, 3]), + "{language}" + ); + } +} + +/// A name bound to an edge in one branch and to a node in the other is each +/// in its own branch, in both orders. +#[test] +fn a_name_is_a_node_or_an_edge_per_branch() { + let db = three_edges(Config::in_memory()); + let expected = vec![ + vec!["K w=Int64(1)".to_string()], + vec!["n=Int64(3)".to_string()], + ]; + for language in LANGUAGES { + for query in [ + "MATCH ()-[x]->() WHERE x.w = 1 RETURN x UNION ALL MATCH (x:A) WHERE x.n = 3 RETURN x", + "MATCH (x:A) WHERE x.n = 3 RETURN x UNION ALL MATCH ()-[x]->() WHERE x.w = 1 RETURN x", + ] { + assert_eq!( + sorted(rows(&db, language, query)), + expected, + "{language}: {query}" + ); + } + } +} + +/// Edges ordered, cut or skipped before RETURN are still edges: their +/// properties and type read right after. +#[test] +fn edges_ordered_before_return_stay_edges() { + let db = three_edges(Config::in_memory()); + for query in [ + "MATCH (a)-[r]->(b) WITH r ORDER BY r.w DESC LIMIT 2 RETURN r.w AS w, type(r) AS t", + "MATCH (a)-[r]->(b) WITH r, a ORDER BY a.n SKIP 1 RETURN r.w AS w, type(r) AS t", + "MATCH (a)-[r]->(b) WITH DISTINCT r ORDER BY r.w LIMIT 2 RETURN r.w AS w, type(r) AS t", + ] { + let result = rows(&db, "cypher", query); + assert_eq!(result.len(), 2, "{query}"); + for row in &result { + assert!(row[0].starts_with("Int64("), "{query}: {result:?}"); + assert_eq!(row[1], "String(\"K\")", "{query}: {result:?}"); + } + } + let gql = "MATCH (a)-[r]->(b) LET x = r.w RETURN r.w AS w, type(r) AS t, x ORDER BY w"; + assert_eq!( + rows(&db, "gql", gql), + [1, 2, 3].map(|w| vec![ + format!("Int64({w})"), + "String(\"K\")".to_string(), + format!("Int64({w})") + ]) + ); +} + +/// An edge a subquery returns keeps its properties through the outer +/// query's ORDER BY and LIMIT. +#[cfg(feature = "cypher")] +#[test] +fn an_edge_from_a_subquery_keeps_its_properties() { + let db = three_edges(Config::in_memory()); + let query = "MATCH (a:A) CALL { WITH a MATCH (a)-[r]->(b) RETURN r } RETURN a.n AS n, r.w AS w ORDER BY n DESC LIMIT 2"; + assert_eq!( + rows(&db, "cypher", query), + [(5, 3), (3, 2)].map(|(n, w)| vec![format!("Int64({n})"), format!("Int64({w})")]) + ); +} + +/// Nodes and edges a `CALL` subquery returns are records when the outer +/// query returns them: the subquery passes them on as references (so a later +/// `MATCH` can start from them), whether it reads the outer row or comes +/// first. +#[test] +fn entities_from_a_call_subquery_are_records() { + let db = three_edges(Config::in_memory()); + let pairs = [(2, 1), (4, 2), (6, 3)] + .map(|(n, w)| vec![format!("n=Int64({n})"), format!("K w=Int64({w})")]); + for language in LANGUAGES { + for (query, expected) in [ + ( + "MATCH (a:A) CALL { WITH a MATCH (a)-[r]->(b) RETURN b, r } RETURN b, r", + pairs.to_vec(), + ), + ( + "MATCH (a:A) CALL { WITH a MATCH (a)-[r]->(b) RETURN DISTINCT b } RETURN b", + nodes(&[2, 4, 6]), + ), + ("CALL { MATCH (b:B) RETURN b } RETURN b", nodes(&[2, 4, 6])), + ( + "CALL { MATCH ()-[r]->() RETURN r } RETURN r", + edges(&[1, 2, 3]), + ), + ] { + assert_eq!( + sorted(rows(&db, language, query)), + sorted(expected), + "{language}: {query}" + ); + } + } + // A query does not end with such a subquery: a GQL query ends with + // RETURN, and Cypher rejects it as Neo4j does. + let error = db + .execute_cypher("MATCH (a:A) CALL { WITH a MATCH (a)-[r]->(b) RETURN b, r }") + .unwrap_err(); + assert!( + error + .to_string() + .contains("Query cannot conclude with CALL"), + "{error}" + ); +} + +/// Nodes and edges collected into a list, kept as a group key or unwound from +/// a collected list are returned as records: `collect` and grouping used to +/// return their raw IDs, which overlap (edge 0 and node 0 both exist here). +#[test] +fn collected_grouped_and_unwound_entities_are_records() { + let db = three_edges(Config::in_memory()); + for language in LANGUAGES { + let sorted_rows = |query: &str| sorted(rows(&db, language, query)); + assert_eq!( + sorted_rows("MATCH (a)-[r]->(b) WHERE r.w = 2 RETURN collect(r) AS rs"), + [["[K w=Int64(2)]"]], + "{language}" + ); + assert_eq!( + sorted_rows("MATCH (a:A) WHERE a.n = 3 RETURN collect(a) AS ns"), + [["[n=Int64(3)]"]], + "{language}" + ); + assert_eq!( + sorted_rows("MATCH (a)-[r]->(b) WITH r, count(*) AS c RETURN r"), + edges(&[1, 2, 3]), + "{language}" + ); + assert_eq!( + sorted_rows("MATCH (a:A)-[r]->(b) WITH a, count(r) AS c RETURN a"), + nodes(&[1, 3, 5]), + "{language}" + ); + assert_eq!( + sorted_rows("MATCH (a)-[r]->(b) WITH collect(r) AS rs UNWIND rs AS e RETURN e"), + edges(&[1, 2, 3]), + "{language}" + ); + } +} + +/// A step that combines an edge and a node under one name returns each as +/// what it is: the planner knows only that the name holds an edge in one +/// branch, so the result is resolved from each row's own column. +#[cfg(feature = "gremlin")] +#[test] +fn an_edge_and_a_node_from_one_union_step_are_both_records() { + let db = three_edges(Config::in_memory()); + for query in [ + "g.V().has('n', 1).union(outE('K'), out('K'))", + "g.V().has('n', 1).union(out('K'), outE('K'))", + ] { + assert_eq!( + sorted(rows(&db, "gremlin", query)), + [["K w=Int64(1)"], ["n=Int64(2)"]], + "{query}" + ); + } +} diff --git a/crates/grafeo-engine/tests/row_order.rs b/crates/grafeo-engine/tests/row_order.rs index 9a4133567..42b22b453 100644 --- a/crates/grafeo-engine/tests/row_order.rs +++ b/crates/grafeo-engine/tests/row_order.rs @@ -54,6 +54,55 @@ fn results_without_order_by_are_shuffled() { assert!(groups.len() > 1); } +/// A stream is shuffled one chunk at a time, so it keeps its bounded memory: +/// over several chunks, each chunk holds the same rows in every run (the rows +/// the scan put there), in an order that changes, and every row comes back. +/// A shuffle of the whole result would move rows between chunks. +#[test] +fn streamed_results_are_shuffled_per_chunk() { + let db = GrafeoDB::with_config(Config::in_memory().with_shuffle_unordered(true)).unwrap(); + db.execute("UNWIND range(0, 4999) AS v INSERT (:A {v: v})") + .unwrap(); + let chunks = || { + let mut stream = db.execute_streaming("MATCH (n:A) RETURN n.v").unwrap(); + let mut chunks = Vec::new(); + while let Some(chunk) = stream.next_chunk().unwrap() { + let column = chunk.column(0).unwrap(); + let values: Vec = chunk + .selected_indices() + .map(|row| column.get_value(row).and_then(|v| v.as_int64()).unwrap()) + .collect(); + chunks.push(values); + } + chunks + }; + let sorted = |chunks: &[Vec]| -> Vec> { + chunks + .iter() + .map(|chunk| { + let mut chunk = chunk.clone(); + chunk.sort_unstable(); + chunk + }) + .collect() + }; + + let first = chunks(); + assert!(first.len() > 1, "{} chunks", first.len()); + let mut all: Vec = first.iter().flatten().copied().collect(); + all.sort_unstable(); + assert_eq!(all, (0..5000).collect::>(), "every row once"); + + let runs: Vec>> = (0..4).map(|_| chunks()).collect(); + for run in &runs { + assert_eq!(sorted(run), sorted(&first), "each chunk keeps its rows"); + } + assert!( + runs.iter().any(|run| *run != first), + "five runs gave one order" + ); +} + #[test] fn ordered_results_keep_their_order() { let db = shuffled_database(); diff --git a/crates/grafeo-engine/tests/seam_dml_interactions.rs b/crates/grafeo-engine/tests/seam_dml_interactions.rs index 144a2c2b1..ef9adee3d 100644 --- a/crates/grafeo-engine/tests/seam_dml_interactions.rs +++ b/crates/grafeo-engine/tests/seam_dml_interactions.rs @@ -597,3 +597,74 @@ mod negative_literal_properties { assert_eq!(result.rows()[0][1], Value::Float64(106.845)); } } + +// ============================================================================ +// Reads after writes in one statement (#479) +// ============================================================================ + +mod reads_after_writes { + use super::*; + + fn ints(result: &grafeo_engine::database::QueryResult) -> Vec> { + result + .rows() + .iter() + .map(|row| row.iter().map(|v| v.as_int64().unwrap()).collect()) + .collect() + } + + /// Every row after the `WITH` sees the nodes that all rows inserted. + #[test] + fn every_row_sees_every_insert() { + let db = db(); + let result = db + .execute( + "UNWIND [1, 2] AS i INSERT (:N {i: i}) WITH i \ + MATCH (n:N) RETURN i, n.i ORDER BY i, n.i", + ) + .unwrap(); + assert_eq!( + ints(&result), + vec![vec![1, 1], vec![1, 2], vec![2, 1], vec![2, 2]] + ); + } + + /// The insert runs even when the `MATCH` after it finds nothing. + #[test] + fn an_insert_runs_when_the_match_after_it_finds_nothing() { + let db = db(); + let result = db + .execute("INSERT (:N {id: 'c'}) WITH 1 AS x MATCH (m:Missing) RETURN m") + .unwrap(); + assert_eq!(result.row_count(), 0); + + let result = db.execute("MATCH (n:N) RETURN n.id").unwrap(); + assert_eq!(result.rows(), &[vec![Value::from("c")]]); + } + + /// Both patterns of a comma-separated `MATCH` see the insert. + #[test] + fn every_pattern_of_a_match_sees_the_insert() { + let db = db(); + let result = db + .execute( + "INSERT (:N {id: 'c'})-[:K]->(:M {id: 'd'}) WITH 1 AS x \ + MATCH (a:N), (b:M) RETURN a.id, b.id", + ) + .unwrap(); + assert_eq!(result.rows(), &[vec![Value::from("c"), Value::from("d")]]); + } + + /// The `MATCH` sees the inserts of all input chunks, not only the first. + #[test] + fn a_match_sees_the_inserts_of_every_chunk() { + let db = db(); + let result = db + .execute( + "UNWIND range(1, 5000) AS i INSERT (:N {i: i}) WITH i WHERE i = 1 \ + MATCH (n:N) RETURN count(n)", + ) + .unwrap(); + assert_eq!(ints(&result), vec![vec![5000]]); + } +} diff --git a/crates/grafeo-engine/tests/seam_session_state.rs b/crates/grafeo-engine/tests/seam_session_state.rs index 248d6dce1..0ae08012f 100644 --- a/crates/grafeo-engine/tests/seam_session_state.rs +++ b/crates/grafeo-engine/tests/seam_session_state.rs @@ -774,6 +774,15 @@ mod introspection { assert!(error.contains("does not exist"), "{error}"); let error = session.execute("INSERT (:Person)").unwrap_err().to_string(); assert!(error.contains("does not exist"), "{error}"); + let error = session + .execute("EXPLAIN MATCH (n) RETURN n") + .unwrap_err() + .to_string(); + assert!(error.contains("does not exist"), "{error}"); + let Err(error) = session.execute_streaming("MATCH (n) RETURN n") else { + panic!("a stream of a graph that does not exist"); + }; + assert!(error.to_string().contains("does not exist"), "{error}"); session.execute("SESSION RESET SCHEMA").unwrap(); let result = session.execute("RETURN CURRENT_GRAPH AS g").unwrap(); diff --git a/crates/grafeo-engine/tests/sparql_aggregate_expressions.rs b/crates/grafeo-engine/tests/sparql_aggregate_expressions.rs index 413839ac3..d548c12d2 100644 --- a/crates/grafeo-engine/tests/sparql_aggregate_expressions.rs +++ b/crates/grafeo-engine/tests/sparql_aggregate_expressions.rs @@ -106,6 +106,26 @@ mod sparql_aggregate_expression_tests { ); } + /// A computed ORDER BY key is a column the sort adds to sort by: the + /// result keeps only the selected columns. + #[test] + fn sparql_order_by_expression_returns_only_the_selected_columns() { + let db = rdf_db(); + insert_sample_triples(&db); + + let qr = db + .execute_sparql("SELECT ?s WHERE { ?s ?o } ORDER BY DESC(STR(?s))") + .unwrap(); + assert_eq!(qr.columns, vec!["s"]); + assert_eq!(qr.column_types.len(), 1, "types: {:?}", qr.column_types); + assert_eq!(qr.row_count(), 2); + assert!( + qr.rows().iter().all(|row| row.len() == 1), + "rows: {:?}", + qr.rows() + ); + } + // --------------------------------------------------------------- // Area 1: SPARQL translator projection with function expressions // --------------------------------------------------------------- diff --git a/crates/grafeo-engine/tests/spec_compliance.rs b/crates/grafeo-engine/tests/spec_compliance.rs index 1d13f9f6b..577242b94 100644 --- a/crates/grafeo-engine/tests/spec_compliance.rs +++ b/crates/grafeo-engine/tests/spec_compliance.rs @@ -1511,6 +1511,57 @@ mod cypher_features { std::fs::remove_file(&csv_path).ok(); } + /// A loaded row is a value, not a node: returned whole or collected, it + /// comes back as the row's map. + #[test] + fn cypher_load_csv_rows_returned_whole_and_collected() { + use std::io::Write; + let dir = std::env::temp_dir(); + let csv_path = dir.join("grafeo_test_load_csv_collect.csv"); + { + let mut f = std::fs::File::create(&csv_path).unwrap(); + writeln!(f, "name,city").unwrap(); + writeln!(f, "Alix,Amsterdam").unwrap(); + writeln!(f, "Gus,Berlin").unwrap(); + } + let row = |name: &str, city: &str| { + Value::Map(std::sync::Arc::new( + [ + (PropertyKey::new("name"), Value::String(name.into())), + (PropertyKey::new("city"), Value::String(city.into())), + ] + .into_iter() + .collect(), + )) + }; + + let db = GrafeoDB::new_in_memory(); + let session = db.session(); + let path = csv_path.display(); + let whole = session + .execute_cypher(&format!( + "LOAD CSV WITH HEADERS FROM '{path}' AS row RETURN row ORDER BY row.name" + )) + .unwrap(); + assert_eq!( + whole.rows(), + [vec![row("Alix", "Amsterdam")], vec![row("Gus", "Berlin")]] + ); + let collected = session + .execute_cypher(&format!( + "LOAD CSV WITH HEADERS FROM '{path}' AS row RETURN collect(row) AS rows" + )) + .unwrap(); + let Value::List(rows) = &collected.rows()[0][0] else { + panic!("expected a list: {:?}", collected.rows()); + }; + let mut rows = rows.to_vec(); + rows.sort_by_key(|value| format!("{value:?}")); + assert_eq!(rows, [row("Alix", "Amsterdam"), row("Gus", "Berlin")]); + + std::fs::remove_file(&csv_path).ok(); + } + #[test] fn cypher_load_csv_without_headers() { use std::io::Write; diff --git a/crates/grafeo-engine/tests/subquery_integration.rs b/crates/grafeo-engine/tests/subquery_integration.rs index 077785fbb..522dbade0 100644 --- a/crates/grafeo-engine/tests/subquery_integration.rs +++ b/crates/grafeo-engine/tests/subquery_integration.rs @@ -237,3 +237,393 @@ mod cypher_subqueries { assert_eq!(harm_row[1], Value::Null); } } + +/// `EXISTS` and `COUNT` over a path: `top -> sub -> f` and `lone -> f`. +#[cfg(feature = "cypher")] +mod subqueries_over_paths { + use super::*; + + fn tree() -> GrafeoDB { + let db = GrafeoDB::new_in_memory(); + db.execute( + "INSERT (:Directory {id: 'top'})-[:CONTAINS]->(:Directory {id: 'sub'})\ + -[:CONTAINS]->(:File {id: 'f'})", + ) + .unwrap(); + db.execute("MATCH (f:File) INSERT (:Directory {id: 'lone'})-[:CONTAINS]->(f)") + .unwrap(); + db + } + + /// One edge from the outer node still decides a path with no condition on + /// its end, also in `RETURN`. + #[test] + fn exists_over_a_path_with_no_end_condition_in_return() { + let db = tree(); + let result = db + .execute_cypher( + "MATCH (n) RETURN n.id, EXISTS { MATCH (n)-[:CONTAINS*]->() } AS e ORDER BY n.id", + ) + .unwrap(); + let rows: Vec<(Value, Value)> = result + .rows() + .iter() + .map(|row| (row[0].clone(), row[1].clone())) + .collect(); + assert_eq!( + rows, + [("f", false), ("lone", true), ("sub", true), ("top", true)] + .map(|(id, e)| (Value::from(id), Value::Bool(e))) + ); + } + + /// In `RETURN`, a subquery that one edge cannot decide is answered for the + /// whole path, not the first hop: `top` reaches a file in two hops and has + /// two nodes below it, `sub` and `lone` one each. + #[test] + fn subqueries_one_edge_cannot_decide_in_return() { + let db = tree(); + for (query, expected) in [ + ( + "MATCH (n:Directory) RETURN n.id, EXISTS { MATCH (n)-[:CONTAINS*]->(:File) } AS e ORDER BY n.id", + [ + ("lone", Value::Bool(true)), + ("sub", Value::Bool(true)), + ("top", Value::Bool(true)), + ], + ), + ( + "MATCH (n:Directory) RETURN n.id, COUNT { MATCH (n)-[:CONTAINS*]->() } AS c ORDER BY n.id", + [ + ("lone", Value::Int64(1)), + ("sub", Value::Int64(1)), + ("top", Value::Int64(2)), + ], + ), + ( + "MATCH (n:Directory) RETURN n.id, EXISTS { MATCH (n)-[:CONTAINS*2..]->() } AS e ORDER BY n.id", + [ + ("lone", Value::Bool(false)), + ("sub", Value::Bool(false)), + ("top", Value::Bool(true)), + ], + ), + ] { + let result = db + .execute_cypher(query) + .unwrap_or_else(|e| panic!("{query}: {e}")); + let rows: Vec<(Value, Value)> = result + .rows() + .iter() + .map(|row| (row[0].clone(), row[1].clone())) + .collect(); + assert_eq!( + rows, + expected.map(|(id, value)| (Value::from(id), value)), + "{query}" + ); + } + } + + /// `COUNT` of single edges keeps the fast path: `f` has two incoming. + #[test] + fn count_of_single_edges_in_return() { + let db = tree(); + let result = db + .execute_cypher( + "MATCH (n) RETURN n.id, COUNT { MATCH (n)<-[:CONTAINS]-() } AS c ORDER BY n.id", + ) + .unwrap(); + let rows: Vec<(Value, Value)> = result + .rows() + .iter() + .map(|row| (row[0].clone(), row[1].clone())) + .collect(); + assert_eq!( + rows, + [("f", 2), ("lone", 0), ("sub", 1), ("top", 0)] + .map(|(id, c)| (Value::from(id), Value::Int64(c))) + ); + } +} + +/// `EXISTS` and `COUNT` through the edge of the row: Alix knows Gus, and Mia +/// likes herself. +mod subqueries_through_a_bound_edge { + use super::*; + + fn people() -> GrafeoDB { + let db = GrafeoDB::new_in_memory(); + db.execute( + "INSERT (:Person {name: 'Alix'})-[:KNOWS]->(:Person {name: 'Gus'}), \ + (mia:Person {name: 'Mia'})-[:LIKES]->(mia)", + ) + .unwrap(); + db + } + + /// The pattern reads the row's edge from the ends it names: forward from + /// the source, backward into the target, and both ways when undirected + /// with free ends. Mia's self-loop is both her source and her target. + #[test] + fn the_edge_is_read_from_the_ends_the_pattern_names() { + let db = people(); + let result = db + .execute( + "MATCH (a)-[r]->(b) RETURN a.name, \ + COUNT { MATCH (a)-[r]->(x) } AS from_a, \ + COUNT { MATCH (b)-[r]->(x) } AS from_b, \ + COUNT { MATCH (b)<-[r]-(x) } AS into_b, \ + COUNT { MATCH (x)-[r]-(y) } AS either_way, \ + EXISTS { MATCH (x)-[r]->(:Person) } AS to_a_person, \ + EXISTS { MATCH (x)-[r]->(:City) } AS to_a_city \ + ORDER BY a.name", + ) + .unwrap(); + let rows: Vec> = result.rows().to_vec(); + assert_eq!( + rows, + [("Alix", [1, 0, 1, 2]), ("Mia", [1, 1, 1, 2])] + .map(|(name, counts)| { + let mut row = vec![Value::from(name)]; + row.extend(counts.map(Value::Int64)); + row.extend([Value::Bool(true), Value::Bool(false)]); + row + }) + .to_vec() + ); + } + + /// A compared `COUNT` in `WHERE` is planned as a join; it counts the + /// undirected matches of the row's edge like the per-row check does. + #[test] + fn a_compared_count_counts_the_edge_both_ways() { + let db = people(); + let result = db + .execute( + "MATCH (a)-[r]->(b) WHERE COUNT { MATCH (x)-[r]-(y) } = 2 \ + RETURN a.name ORDER BY a.name", + ) + .unwrap(); + let names: Vec = result.rows().iter().map(|row| row[0].clone()).collect(); + assert_eq!(names, [Value::from("Alix"), Value::from("Mia")]); + } +} + +/// `EXISTS` and `COUNT` whose pattern holds more than its edge: another node +/// pattern, or a path mode on a variable-length edge. Alix knows Gus and +/// lives in Amsterdam, and Mia likes herself; there is no Robot. +mod subqueries_beyond_the_edge { + use super::*; + + fn people() -> GrafeoDB { + let db = GrafeoDB::new_in_memory(); + db.execute( + "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(:Person {name: 'Gus'}), \ + (alix)-[:LIVES_IN]->(:City {name: 'Amsterdam'}), \ + (mia:Person {name: 'Mia'})-[:LIKES]->(mia)", + ) + .unwrap(); + db + } + + /// In `RETURN`, such a subquery is answered for the whole pattern, not + /// its edge alone: there is no Robot, Alix knows Gus and there are three + /// people, Mia's only LIKES path returns to her (which ACYCLIC excludes), + /// and the one KNOWS edge is a trail from either end. + #[test] + fn subqueries_with_more_than_the_edge_in_return() { + let db = people(); + for (query, expected) in [ + ( + "MATCH (a:Person) RETURN a.name, EXISTS { MATCH (a)-[:KNOWS]->(b), (c:Robot) } AS e", + [Value::Bool(false), Value::Bool(false), Value::Bool(false)], + ), + ( + "MATCH (a:Person) RETURN a.name, EXISTS { MATCH (c:Robot), (a)-[:KNOWS]->(b) } AS e", + [Value::Bool(false), Value::Bool(false), Value::Bool(false)], + ), + ( + "MATCH (a:Person) RETURN a.name, COUNT { MATCH (a)-[:KNOWS]->(b), (c:Person) } AS n", + [Value::Int64(3), Value::Int64(0), Value::Int64(0)], + ), + ( + "MATCH (a:Person) RETURN a.name, EXISTS { MATCH ACYCLIC (a)-[:LIKES*1..2]->(x) } AS e", + [Value::Bool(false), Value::Bool(false), Value::Bool(false)], + ), + ( + "MATCH (a:Person) RETURN a.name, EXISTS { MATCH TRAIL (a)-[:KNOWS*1..2]-(x) } AS e", + [Value::Bool(true), Value::Bool(true), Value::Bool(false)], + ), + ] { + let query = format!("{query} ORDER BY a.name"); + let result = db + .execute(&query) + .unwrap_or_else(|e| panic!("{query}: {e}")); + let rows: Vec<(Value, Value)> = result + .rows() + .iter() + .map(|row| (row[0].clone(), row[1].clone())) + .collect(); + assert_eq!( + rows, + ["Alix", "Gus", "Mia"] + .into_iter() + .zip(expected) + .map(|(name, value)| (Value::from(name), value)) + .collect::>(), + "{query}" + ); + } + } + + /// A node pattern on the end of the edge is that end: its label joins the + /// edge check, also in `RETURN`. + #[cfg(feature = "cypher")] + #[test] + fn a_later_node_pattern_on_the_end_labels_it() { + let db = people(); + let result = db + .execute_cypher( + "MATCH (a:Person) RETURN a.name, \ + COUNT { MATCH (a)-[r]->(b) MATCH (b:Person) } AS people, \ + EXISTS { MATCH (a)-[r]->(b) MATCH (b:Robot) } AS robots \ + ORDER BY a.name", + ) + .unwrap(); + let rows: Vec> = result.rows().to_vec(); + assert_eq!( + rows, + [("Alix", 1), ("Gus", 0), ("Mia", 1)] + .map(|(name, people)| vec![ + Value::from(name), + Value::Int64(people), + Value::Bool(false) + ]) + .to_vec() + ); + } +} + +/// The `EXISTS` and `COUNT` checks see the graph the query sees: at an earlier +/// epoch an edge created later does not count, an edge another transaction +/// has not committed does not count, and neither does one this transaction +/// deleted. +mod subqueries_see_what_the_query_sees { + use super::*; + + /// One query per shape the check answers from edges: one edge to a bound + /// end, one edge counted, one edge in `WHERE`, a path to a bound end, a + /// path from a free end. + const SHAPES: [&str; 5] = [ + "MATCH (g:Person {name: 'Gus'}), (m:Person {name: 'Mia'}) \ + RETURN EXISTS { MATCH (g)-[:KNOWS]->(m) } AS e", + "MATCH (g:Person {name: 'Gus'}) RETURN COUNT { MATCH (g)-[:KNOWS]->() } AS c", + "MATCH (p:Person) WHERE EXISTS { MATCH (p)-[:KNOWS]->() } RETURN p.name AS n ORDER BY n", + "MATCH (a:Person {name: 'Alix'}), (m:Person {name: 'Mia'}) \ + RETURN EXISTS { MATCH (a)-[:KNOWS]->{1,3}(m) } AS e", + "MATCH (v:Person {name: 'Vincent'}) RETURN EXISTS { MATCH (v)-[:KNOWS]->{1,2}() } AS e", + ]; + + fn rows(result: &grafeo_engine::database::QueryResult) -> Vec> { + result.rows().to_vec() + } + + fn names(names: &[&str]) -> Vec> { + names + .iter() + .map(|name| vec![Value::String((*name).into())]) + .collect() + } + + /// The answers with Alix->Gus and Gus->Vincent only. + fn before() -> [Vec>; SHAPES.len()] { + [ + vec![vec![Value::Bool(false)]], + vec![vec![Value::Int64(1)]], + names(&["Alix", "Gus"]), + vec![vec![Value::Bool(false)]], + vec![vec![Value::Bool(false)]], + ] + } + + /// The answers once Gus->Mia and Vincent->Mia exist too. + fn after() -> [Vec>; SHAPES.len()] { + [ + vec![vec![Value::Bool(true)]], + vec![vec![Value::Int64(2)]], + names(&["Alix", "Gus", "Vincent"]), + vec![vec![Value::Bool(true)]], + vec![vec![Value::Bool(true)]], + ] + } + + fn people() -> GrafeoDB { + let db = GrafeoDB::new_in_memory(); + db.execute( + "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus'}), \ + (gus)-[:KNOWS]->(:Person {name: 'Vincent'}), (:Person {name: 'Mia'})", + ) + .unwrap(); + db + } + + const LATER_EDGES: &str = "MATCH (g:Person {name: 'Gus'}), (v:Person {name: 'Vincent'}), \ + (m:Person {name: 'Mia'}) INSERT (g)-[:KNOWS]->(m), (v)-[:KNOWS]->(m)"; + + #[test] + fn at_an_earlier_epoch_later_edges_do_not_count() { + let db = people(); + let epoch = db.current_epoch(); + db.execute(LATER_EDGES).unwrap(); + for ((query, then), now) in SHAPES.iter().zip(before()).zip(after()) { + assert_eq!( + rows(&db.execute_at_epoch(query, epoch).unwrap()), + then, + "at the earlier epoch: {query}" + ); + assert_eq!(rows(&db.execute(query).unwrap()), now, "now: {query}"); + } + } + + #[test] + fn edges_another_transaction_has_not_committed_do_not_count() { + let db = people(); + let mut writer = db.session(); + writer.begin_transaction().unwrap(); + writer.execute(LATER_EDGES).unwrap(); + for ((query, outside), inside) in SHAPES.iter().zip(before()).zip(after()) { + assert_eq!( + rows(&db.execute(query).unwrap()), + outside, + "another session: {query}" + ); + assert_eq!( + rows(&writer.execute(query).unwrap()), + inside, + "the writing transaction: {query}" + ); + } + writer.rollback().unwrap(); + } + + #[test] + fn edges_this_transaction_deleted_do_not_count() { + let db = people(); + db.execute(LATER_EDGES).unwrap(); + let mut session = db.session(); + session.begin_transaction().unwrap(); + session + .execute("MATCH (:Person {name: 'Gus'})-[r:KNOWS]->(:Person {name: 'Mia'}) DELETE r") + .unwrap(); + session + .execute( + "MATCH (:Person {name: 'Vincent'})-[r:KNOWS]->(:Person {name: 'Mia'}) DELETE r", + ) + .unwrap(); + for (query, expected) in SHAPES.iter().zip(before()) { + assert_eq!(rows(&session.execute(query).unwrap()), expected, "{query}"); + } + session.rollback().unwrap(); + } +} diff --git a/crates/grafeo-engine/tests/time_travel.rs b/crates/grafeo-engine/tests/time_travel.rs index 148679ff4..272f5c9d8 100644 --- a/crates/grafeo-engine/tests/time_travel.rs +++ b/crates/grafeo-engine/tests/time_travel.rs @@ -205,6 +205,63 @@ fn test_session_set_viewing_epoch() { assert_eq!(result.rows().len(), 2); } +/// A session that reads at an earlier epoch does not write: statements and +/// the session's direct writes fail until the viewing epoch is cleared, and +/// nothing reaches the store, so the past stays as it was. +#[test] +fn test_writes_fail_while_reading_an_earlier_epoch() { + let db = setup_db(); + let mut session = db.session(); + session.execute("INSERT (:Person {name: 'Alix'})").unwrap(); + bump_epoch(&mut session); + session.execute("INSERT (:Person {name: 'Gus'})").unwrap(); + let names = |session: &grafeo_engine::Session| -> Vec { + session + .execute("MATCH (p:Person) RETURN p.name ORDER BY p.name") + .unwrap() + .rows() + .iter() + .map(|row| row[0].clone()) + .collect() + }; + + session.set_viewing_epoch(EpochId::new(0)); + let error = session + .execute("INSERT (:Person {name: 'Vincent'})") + .unwrap_err(); + assert!(error.to_string().contains("earlier epoch"), "{error}"); + assert!( + session + .execute("MATCH (p:Person {name: 'Alix'}) SET p.age = 30") + .is_err() + ); + assert!(session.create_node(&["Person"]).is_err()); + // Reads still work at that epoch. + session.execute("MATCH (p:Person) RETURN p.name").unwrap(); + session.clear_viewing_epoch(); + + assert!( + session + .execute_at_epoch("INSERT (:Person {name: 'Mia'})", EpochId::new(0)) + .is_err() + ); + assert_eq!(names(&session), [Value::from("Alix"), Value::from("Gus")]); + assert!( + session + .execute("MATCH (p:Person) WHERE p.age IS NOT NULL RETURN p") + .unwrap() + .rows() + .is_empty() + ); + + // With the viewing epoch cleared, the session writes again. + session.execute("INSERT (:Person {name: 'Mia'})").unwrap(); + assert_eq!( + names(&session), + [Value::from("Alix"), Value::from("Gus"), Value::from("Mia")] + ); +} + #[test] fn test_session_reset_clears_viewing_epoch() { let db = setup_db(); diff --git a/crates/grafeo-engine/tests/transaction_edge_reads.rs b/crates/grafeo-engine/tests/transaction_edge_reads.rs new file mode 100644 index 000000000..f5abab4a1 --- /dev/null +++ b/crates/grafeo-engine/tests/transaction_edge_reads.rs @@ -0,0 +1,115 @@ +//! A transaction reads the edges it created, with their type, on every kind +//! of store: in memory, with CDC, in a `.grafeo` file, in a WAL directory and +//! after `compact()`. Persistent databases used to read the type of a new +//! edge at the committed state, so `-[:KNOWS]->` missed it and `type(r)` was +//! null until the commit. +//! +//! ```bash +//! cargo test -p grafeo-engine --all-features --test transaction_edge_reads +//! ``` + +#![cfg(all(feature = "lpg", feature = "gql"))] + +use grafeo_common::types::Value; +use grafeo_engine::{Config, GrafeoDB}; + +/// The rows of `query` run in `session`, as text, sorted. +fn rows(session: &grafeo_engine::Session, query: &str) -> Vec> { + let mut rows: Vec> = session + .execute(query) + .unwrap() + .rows() + .iter() + .map(|row| { + row.iter() + .map(|value| match value { + Value::String(text) => text.to_string(), + other => format!("{other:?}"), + }) + .collect() + }) + .collect(); + rows.sort(); + rows +} + +/// Inserts an edge in a transaction and reads it back by its type, before +/// and after the commit. +fn reads_its_new_edge(db: &GrafeoDB, store: &str) { + db.execute("INSERT (:P {name: 'a'}), (:P {name: 'b'})") + .unwrap(); + let mut session = db.session(); + session.begin_transaction().unwrap(); + session + .execute("MATCH (a:P {name: 'a'}), (b:P {name: 'b'}) INSERT (a)-[:KNOWS]->(b)") + .unwrap(); + let checks = [ + ("MATCH (a:P)-[:KNOWS]->(b) RETURN b.name", vec![vec!["b"]]), + ("MATCH (b:P)<-[:KNOWS]-(a) RETURN a.name", vec![vec!["a"]]), + ("MATCH ()-[r]->() RETURN type(r)", vec![vec!["KNOWS"]]), + ]; + for (query, expected) in &checks { + assert_eq!( + rows(&session, query), + *expected, + "{store}, in the transaction: {query}" + ); + } + session.commit().unwrap(); + for (query, expected) in &checks { + assert_eq!( + rows(&db.session(), query), + *expected, + "{store}, committed: {query}" + ); + } +} + +#[test] +fn in_memory() { + reads_its_new_edge(&GrafeoDB::new_in_memory(), "in memory"); +} + +#[cfg(feature = "cdc")] +#[test] +fn in_memory_with_cdc() { + let db = GrafeoDB::with_config(Config::in_memory().with_cdc()).unwrap(); + reads_its_new_edge(&db, "in memory with CDC"); +} + +#[cfg(feature = "grafeo-file")] +#[test] +fn in_a_grafeo_file() { + let dir = tempfile::tempdir().unwrap(); + let db = GrafeoDB::open(dir.path().join("edges.grafeo")).unwrap(); + reads_its_new_edge(&db, "a .grafeo file"); + db.close().unwrap(); +} + +#[cfg(all(feature = "grafeo-file", feature = "cdc"))] +#[test] +fn in_a_grafeo_file_with_cdc() { + let dir = tempfile::tempdir().unwrap(); + let db = GrafeoDB::with_config(Config::persistent(dir.path().join("edges.grafeo")).with_cdc()) + .unwrap(); + reads_its_new_edge(&db, "a .grafeo file with CDC"); + db.close().unwrap(); +} + +#[cfg(feature = "wal")] +#[test] +fn in_a_wal_directory() { + let dir = tempfile::tempdir().unwrap(); + let db = GrafeoDB::open(dir.path().join("edges")).unwrap(); + reads_its_new_edge(&db, "a WAL directory"); + db.close().unwrap(); +} + +#[cfg(feature = "compact-store")] +#[test] +fn after_compact() { + let mut db = GrafeoDB::new_in_memory(); + db.execute("INSERT (:Q {name: 'old'})").unwrap(); + db.compact().unwrap(); + reads_its_new_edge(&db, "after compact()"); +} diff --git a/crates/grafeo-engine/tests/upserts.rs b/crates/grafeo-engine/tests/upserts.rs index 07a6b1f82..eee6fb719 100644 --- a/crates/grafeo-engine/tests/upserts.rs +++ b/crates/grafeo-engine/tests/upserts.rs @@ -200,6 +200,147 @@ fn edges_are_created_then_updated_between_existing_nodes() { ); } +/// A row that names two edges with the same key between the same endpoints +/// (written without an upsert) updates both, as a MERGE binds every match; it +/// comes back once per edge of one pair of endpoints, which names no +/// ambiguous endpoint, so it is not skipped. +#[test] +fn duplicate_keyed_edges_are_all_updated() { + let db = GrafeoDB::new_in_memory(); + files(&db); + db.execute( + "MATCH (a:File {id: 'f1'}), (b:File {id: 'f2'}) \ + INSERT (a)-[:USES {id: 'u1', w: 1}]->(b), (a)-[:USES {id: 'u1', w: 1}]->(b)", + ) + .unwrap(); + let result = db + .upsert_edges( + "USES", + vec![row(&[ + ("src", Value::from("f1")), + ("dst", Value::from("f2")), + ("id", Value::from("u1")), + ("w", Value::Int64(2)), + ])], + &EdgeUpsertOptions::default(), + ) + .unwrap(); + assert_eq!(result, summary(0, 1, &[])); + assert_eq!( + rows(&db, "MATCH ()-[r:USES]->() RETURN r.w"), + [vec![Value::Int64(2)], vec![Value::Int64(2)]] + ); +} + +/// An endpoint key that more than one node has names no single endpoint: +/// the row is skipped and reported, and writes no edge at all. +#[test] +fn a_row_with_an_ambiguous_endpoint_is_skipped() { + let db = GrafeoDB::new_in_memory(); + db.create_property_index("id"); + files(&db); + db.execute("INSERT (:Other {id: 'f2'})").unwrap(); + let edge = |src: &str, dst: &str, id: &str| { + row(&[ + ("src", Value::from(src)), + ("dst", Value::from(dst)), + ("id", Value::from(id)), + ]) + }; + let result = db + .upsert_edges( + "USES", + vec![ + edge("f1", "f2", "u1"), + edge("f2", "f3", "u2"), + edge("f1", "f3", "u3"), + ], + &EdgeUpsertOptions::default(), + ) + .unwrap(); + assert_eq!(result, summary(1, 0, &[0, 1])); + assert_eq!( + rows(&db, "MATCH (s)-[r:USES]->(d) RETURN s.id, d.id, r.id"), + [vec![ + Value::from("f1"), + Value::from("f3"), + Value::from("u3") + ]] + ); + + // Inside a transaction the transaction's own writes stay. + let mut session = db.session(); + session.begin_transaction().unwrap(); + session.execute("INSERT (:Log {n: 1})").unwrap(); + let result = session + .upsert_edges( + "CALLS", + vec![edge("f1", "f2", "c1"), edge("f3", "f1", "c2")], + &EdgeUpsertOptions::default(), + ) + .unwrap(); + assert_eq!(result, summary(1, 0, &[0])); + session.commit().unwrap(); + assert_eq!( + rows(&db, "MATCH (l:Log) RETURN l.n"), + [vec![Value::Int64(1)]] + ); + assert_eq!( + rows(&db, "MATCH (s)-[r:CALLS]->(d) RETURN s.id, d.id"), + [vec![Value::from("f3"), Value::from("f1")]] + ); + + // Without auto-commit and outside a transaction the call is still one + // write: the undone attempt leaves nothing behind. + let mut manual = db.session(); + manual.set_auto_commit(false); + let result = manual + .upsert_edges( + "LINKS", + vec![edge("f1", "f2", "l1"), edge("f3", "f1", "l2")], + &EdgeUpsertOptions::default(), + ) + .unwrap(); + assert_eq!(result, summary(1, 0, &[0])); + assert_eq!( + rows(&db, "MATCH (s)-[r:LINKS]->(d) RETURN s.id, d.id"), + [vec![Value::from("f3"), Value::from("f1")]] + ); +} + +/// The edge key and the two endpoint fields name three different fields of +/// a row; otherwise one would consume another and every row would be skipped. +#[test] +fn clashing_field_names_are_rejected() { + let db = GrafeoDB::new_in_memory(); + files(&db); + let rows_of = || { + vec![row(&[ + ("src", Value::from("f1")), + ("dst", Value::from("f2")), + ("id", Value::from("u1")), + ])] + }; + for (key, src_field, dst_field) in [ + ("src", "src", "dst"), + ("dst", "src", "dst"), + ("id", "src", "src"), + ] { + let options = EdgeUpsertOptions { + key: key.to_string(), + src_field: src_field.to_string(), + dst_field: dst_field.to_string(), + ..EdgeUpsertOptions::default() + }; + let error = db.upsert_edges("USES", rows_of(), &options).unwrap_err(); + assert!( + error.to_string().contains("different fields"), + "{key} {src_field} {dst_field}: {error}" + ); + } + assert_eq!(db.edge_count(), 0); +} + #[test] fn endpoint_labels_and_field_names_are_configurable() { let db = GrafeoDB::new_in_memory(); diff --git a/crates/grafeo-engine/tests/write_checks.rs b/crates/grafeo-engine/tests/write_checks.rs index 9f473c41f..c8a134464 100644 --- a/crates/grafeo-engine/tests/write_checks.rs +++ b/crates/grafeo-engine/tests/write_checks.rs @@ -146,6 +146,31 @@ fn a_failed_direct_batch_in_a_transaction_leaves_nothing() { assert_eq!(doc_ids(&db), ids(&[])); } +/// A MERGE whose `ON CREATE SET` fails on the new edge does not leave the +/// edge it created. +#[test] +fn a_failed_edge_merge_in_a_transaction_leaves_nothing() { + let db = typed_docs(); + db.execute("CREATE EDGE TYPE CITES (since INTEGER, note STRING)") + .unwrap(); + db.execute("INSERT (:Doc {id: 1}), (:Doc {id: 2})").unwrap(); + let mut session = db.session(); + session.begin_transaction().unwrap(); + session + .execute( + "MATCH (a:Doc {id: 1}), (b:Doc {id: 2}) \ + MERGE (a)-[r:CITES {since: 2020}]->(b) ON CREATE SET r.note = r.since + 1", + ) + .unwrap_err(); + session.commit().unwrap(); + + let edges = db + .execute("MATCH ()-[r:CITES]->() RETURN count(r)") + .unwrap(); + assert_eq!(edges.rows()[0][0], grafeo_common::types::Value::Int64(0)); + assert_eq!(doc_ids(&db), ids(&[1, 2])); +} + /// The undone statement's WAL records are dropped too: a reopen replays only /// what committed. #[cfg(feature = "wal")] diff --git a/crates/grafeo-engine/tests/write_counters.rs b/crates/grafeo-engine/tests/write_counters.rs index 288175e8a..cff3be88a 100644 --- a/crates/grafeo-engine/tests/write_counters.rs +++ b/crates/grafeo-engine/tests/write_counters.rs @@ -105,6 +105,37 @@ fn merge_counts_only_what_it_creates() { ); } +/// A label written twice is one label, counted once. +#[test] +fn a_repeated_label_counts_once() { + let db = GrafeoDB::new_in_memory(); + let c = counters(&db, "INSERT (:City:City {name: 'Paris'})"); + assert_eq!((c.nodes_created, c.labels_added), (1, 1)); + assert_eq!( + db.execute("MATCH (n:City) RETURN labels(n)") + .unwrap() + .rows()[0][0], + Value::List(vec![Value::from("City")].into()) + ); +} + +/// Writes inside a stored procedure count for the statement that calls it. +#[cfg(feature = "algos")] +#[test] +fn a_procedure_counts_its_writes() { + let db = GrafeoDB::new_in_memory(); + db.execute( + "CREATE PROCEDURE add_city(name STRING) RETURNS (n INTEGER) AS { \ + INSERT (c:City {name: $name}) RETURN 1 AS n }", + ) + .unwrap(); + let c = counters(&db, "CALL add_city('Paris') YIELD n RETURN n"); + assert_eq!( + (c.nodes_created, c.labels_added, c.properties_set), + (1, 1, 1) + ); +} + #[test] fn reads_and_failed_checks_report_nothing() { let db = GrafeoDB::new_in_memory(); diff --git a/crates/grafeo-storage/src/file/manager.rs b/crates/grafeo-storage/src/file/manager.rs index 4d67dc74a..9fc76e537 100644 --- a/crates/grafeo-storage/src/file/manager.rs +++ b/crates/grafeo-storage/src/file/manager.rs @@ -120,15 +120,14 @@ impl GrafeoFileManager { })?; // Acquire an exclusive lock: prevents other processes from opening the same file - { - let _no_child_start = child_process::lock_acquisition(); - file.try_lock_exclusive().map_err(|_| { + child_process::take_lock(|| file.try_lock_exclusive(), is_lock_contended).map_err( + |_| { Error::Internal(format!( "database file is locked by another process: {}", path.display() )) - })?; - } + }, + )?; let file_header = FileHeader::new(); header::write_file_header(&mut file, &file_header)?; @@ -166,15 +165,14 @@ impl GrafeoFileManager { let mut file = OpenOptions::new().read(true).write(true).open(&path)?; // Acquire an exclusive lock: prevents other processes from opening the same file - { - let _no_child_start = child_process::lock_acquisition(); - file.try_lock_exclusive().map_err(|_| { + child_process::take_lock(|| file.try_lock_exclusive(), is_lock_contended).map_err( + |_| { Error::Internal(format!( "database file is locked by another process: {}", path.display() )) - })?; - } + }, + )?; finish_interrupted_checkpoint(&path, &mut file)?; @@ -224,15 +222,16 @@ impl GrafeoFileManager { // Acquire a shared lock: coexists with other shared locks but // blocks if an exclusive lock cannot be shared (platform-dependent). - { - let _no_child_start = child_process::lock_acquisition(); - database_file.try_lock_shared().map_err(|_| { - Error::Internal(format!( - "database file cannot be locked for reading: {}", - path.display() - )) - })?; - } + child_process::take_lock( + || database_file.try_lock_shared(), + |e| matches!(e, std::fs::TryLockError::WouldBlock), + ) + .map_err(|_| { + Error::Internal(format!( + "database file cannot be locked for reading: {}", + path.display() + )) + })?; let pending_image = checkpoint_image_path(&path); let (mut file, lock_holder) = if pending_image.exists() { @@ -1081,6 +1080,12 @@ fn install_image(file: &mut File, image: &Path) -> Result<()> { Ok(()) } +/// Whether a failed `fs2` lock attempt failed because another handle holds +/// the lock (as opposed to an I/O error). +fn is_lock_contended(error: &std::io::Error) -> bool { + error.raw_os_error() == fs2::lock_contended_error().raw_os_error() +} + /// Opening a database file for writing: an image that was still being /// written is discarded (the database file was not touched yet), and a /// complete image whose install was cut off is installed. @@ -2188,6 +2193,7 @@ mod tests { (manager, completed) } + #[cfg(feature = "testing-crash-injection")] fn try_lpg_payload(manager: &GrafeoFileManager) -> Result> { use grafeo_common::storage::SectionType; diff --git a/crates/grafeo-storage/src/lock.rs b/crates/grafeo-storage/src/lock.rs index d8ce07a5d..b98dbed4b 100644 --- a/crates/grafeo-storage/src/lock.rs +++ b/crates/grafeo-storage/src/lock.rs @@ -41,16 +41,17 @@ impl DirectoryLock { .truncate(false) .open(&path)?; - { - let _no_child_start = child_process::lock_acquisition(); - file.try_lock().map_err(|e| match e { - std::fs::TryLockError::WouldBlock => Error::Internal(format!( - "database is locked by another process: {}", - dir.display() - )), - std::fs::TryLockError::Error(e) => Error::Io(e), - })?; - } + child_process::take_lock( + || file.try_lock(), + |e| matches!(e, std::fs::TryLockError::WouldBlock), + ) + .map_err(|e| match e { + std::fs::TryLockError::WouldBlock => Error::Internal(format!( + "database is locked by another process: {}", + dir.display() + )), + std::fs::TryLockError::Error(e) => Error::Io(e), + })?; Ok(Self { _file: file, path }) } diff --git a/crates/grafeo/Cargo.toml b/crates/grafeo/Cargo.toml index 29f136fa0..7ba3d1ec4 100644 --- a/crates/grafeo/Cargo.toml +++ b/crates/grafeo/Cargo.toml @@ -57,7 +57,7 @@ parquet-import = ["grafeo-engine/parquet-import"] # Apache Parquet bulk import # Persona-based profiles lpg = ["grafeo-engine/lpg", "gql", "cypher", "gremlin", "sql-pgq", "storage", "regex"] # Graph App Developer: LPG model + all LPG query languages -rdf = ["triple-store", "gql", "sparql", "graphql", "storage", "regex", "shacl"] # Knowledge Engineer: RDF model + SPARQL/GraphQL + SHACL validation +rdf = ["triple-store", "grafeo-engine/lpg", "gql", "sparql", "graphql", "storage", "regex", "shacl"] # Knowledge Engineer: RDF model + SPARQL/GraphQL + SHACL validation; the LPG store is needed for persistence until #544 analytics = ["algos", "vector-index", "text-index", "hybrid-search", "jsonl-import", "parquet-import"] # Data Scientist: algorithms + search + bulk import ai = ["vector-index", "text-index", "hybrid-search", "cdc"] # AI/Agent Developer: search + change tracking edge = ["grafeo-engine/lpg", "gql", "regex-lite"] # Frontend/Edge: minimal WASM-friendly diff --git a/crates/grafeo/tests/rdf_profile.rs b/crates/grafeo/tests/rdf_profile.rs new file mode 100644 index 000000000..b331bf17a --- /dev/null +++ b/crates/grafeo/tests/rdf_profile.rs @@ -0,0 +1,44 @@ +//! The `rdf` profile on its own keeps its data across a reopen, in both +//! storage formats. Without the LPG store it kept nothing (#544): loading +//! the file, replaying the WAL and checkpoints need it, so the profile +//! includes it. +//! +//! ```bash +//! cargo test -p grafeo --no-default-features --features rdf --test rdf_profile +//! ``` + +#![cfg(feature = "rdf")] + +use grafeo::{Config, GrafeoDB}; + +fn triples(db: &GrafeoDB) -> usize { + db.execute_sparql("SELECT ?s ?p ?o WHERE { ?s ?p ?o }") + .unwrap() + .rows() + .len() +} + +#[test] +fn triples_survive_a_reopen() { + let dir = std::env::temp_dir().join(format!("grafeo-rdf-profile-{}", std::process::id())); + // A run that failed leaves its files behind. + let _ = std::fs::remove_dir_all(&dir); + std::fs::create_dir_all(&dir).unwrap(); + // A `.grafeo` path is a single file, any other path a WAL directory. + for name in ["single.grafeo", "wal-directory"] { + let path = dir.join(name); + { + let db = GrafeoDB::with_config(Config::persistent(&path)).unwrap(); + db.execute_sparql( + "INSERT DATA { }", + ) + .unwrap(); + assert_eq!(triples(&db), 1, "{name}"); + db.close().unwrap(); + } + let db = GrafeoDB::with_config(Config::persistent(&path)).unwrap(); + assert_eq!(triples(&db), 1, "{name} after reopen"); + db.close().unwrap(); + } + std::fs::remove_dir_all(&dir).unwrap(); +} diff --git a/docs/api/python/database.md b/docs/api/python/database.md index 0ac6bd53d..c2925ba3b 100644 --- a/docs/api/python/database.md +++ b/docs/api/python/database.md @@ -360,8 +360,10 @@ db.upsert_nodes(["Graph", "File"], [{"id": "f1", "size": 3}, {"id": "f2", "size" One edge of `edge_type` per row, between the nodes whose `endpoint_key` is the row's `src_field` and `dst_field` value (restricted to `endpoint_labels` when given). The edge is identified by its endpoints, -type and `key`; every other field of the row is an edge property. A row whose endpoint does not exist, or -without the key, is skipped, never created. A property index on `endpoint_key` makes the lookups fast. +type and `key`; every other field of the row is an edge property. A row is skipped when it lacks the key, +`src_field` or `dst_field`, or when no node or more than one node has its endpoint key; endpoints are never +created. `key`, `src_field` and `dst_field` must be different fields. A property index on `endpoint_key` +makes the lookups fast. ```python def upsert_edges( @@ -379,6 +381,7 @@ def upsert_edges( ```python db.create_property_index("id") +db.upsert_nodes(["File"], [{"id": "f1"}, {"id": "f2"}]) db.upsert_edges("USES", [{"src": "f1", "dst": "f2", "id": "u1", "weight": 1}]) # {'created': 1, 'updated': 0, 'skipped': 0, 'skipped_rows': []} ``` @@ -1061,7 +1064,7 @@ test_db = file_db.to_memory() # safe copy for experiments, indexes included Converts the database to a layered [CompactStore](../../user-guide/compact-store.md) for faster queries: a columnar base with CSR adjacency, built from a snapshot of all nodes and edges, plus a mutable overlay. The original store is dropped to free memory. -The database stays writable: new writes land in the overlay, and `recompact()` merges the overlay into a fresh base. Gives ~60x memory reduction and 100x+ traversal speedup for read-mostly workloads. +The database stays writable: new writes land in the overlay, and calling `compact()` again merges them into a fresh base. Gives ~60x memory reduction and 100x+ traversal speedup for read-mostly workloads. ```python def compact(self) -> None diff --git a/docs/architecture/feature-profiles.md b/docs/architecture/feature-profiles.md index 9a1ccac62..eb35792e0 100644 --- a/docs/architecture/feature-profiles.md +++ b/docs/architecture/feature-profiles.md @@ -16,7 +16,7 @@ Since 0.5.35, profiles are named after *what you are building*, not *where it ru | Profile | Persona | What it enables | | --- | --- | --- | | `lpg` | Graph App Developer | Labeled property graph model, GQL, Cypher, Gremlin, SQL/PGQ, storage, regex | -| `rdf` | Knowledge Engineer | RDF triple store, GQL, SPARQL, GraphQL, SHACL validation, storage, regex | +| `rdf` | Knowledge Engineer | RDF triple store, GQL, SPARQL, GraphQL, SHACL validation, storage (with the LPG store it needs), regex | | `analytics` | Data Scientist | Graph algorithms, vector, text and hybrid search, JSON Lines and Parquet import | | `ai` | AI Memory / Agent Developer | Vector, text and hybrid search, change data capture | | `edge` | Frontend / Edge Developer | LPG model, GQL, lightweight regex (minimal, WASM-friendly) | @@ -62,10 +62,10 @@ All labeled property graph query languages plus persistence. The default choice ### RDF ```toml -rdf = ["triple-store", "gql", "sparql", "graphql", "storage", "regex", "shacl"] +rdf = ["triple-store", "grafeo-engine/lpg", "gql", "sparql", "graphql", "storage", "regex", "shacl"] ``` -RDF triple store with SPARQL, GraphQL, SHACL validation and persistence, for knowledge engineers working with ontologies and linked data. Add the `ring-index` atom for compact RDF indexing (it pulls in `succinct-indexes`). +RDF triple store with SPARQL, GraphQL, SHACL validation and persistence, for knowledge engineers working with ontologies and linked data. Persistence needs the LPG store for now, so the profile includes it ([#544](https://github.com/GrafeoDB/grafeo/issues/544)): without it a database lost its triples on reopen. Add the `ring-index` atom for compact RDF indexing (it pulls in `succinct-indexes`). > **Note:** in the lower-level crates (`grafeo-core`, `grafeo-adapters`, `grafeo-engine`), `rdf` is a deprecated alias for the `triple-store` atom only. The profile above applies to the facade and binding crates. diff --git a/docs/getting-started/cli.md b/docs/getting-started/cli.md index 7fb550f61..826cb2553 100644 --- a/docs/getting-started/cli.md +++ b/docs/getting-started/cli.md @@ -97,7 +97,7 @@ grafeo shell ./mydb ``` ``` -Grafeo 0.5.43 - Lpg mode, 42 nodes, 87 edges +Grafeo 0.5.44 - Lpg mode, 42 nodes, 87 edges Type :help for commands, :quit to exit. grafeo> MATCH (n:Person) RETURN n.name, n.age @@ -231,7 +231,7 @@ grafeo completions powershell >> $PROFILE ```bash $ grafeo version -grafeo 0.5.43 +grafeo 0.5.44 Build: rustc: 1.91.1 diff --git a/docs/getting-started/installation.md b/docs/getting-started/installation.md index 28de5a038..75b3c8720 100644 --- a/docs/getting-started/installation.md +++ b/docs/getting-started/installation.md @@ -125,7 +125,7 @@ Add to `pubspec.yaml`: ```yaml dependencies: - grafeo: ^0.5.43 + grafeo: ^0.5.44 ``` ### Verify Installation diff --git a/docs/index.md b/docs/index.md index 101a2e67c..befc85670 100644 --- a/docs/index.md +++ b/docs/index.md @@ -292,7 +292,7 @@ Choose the query language that fits the project: ```yaml # pubspec.yaml dependencies: - grafeo: ^0.5.43 + grafeo: ^0.5.44 ``` === "WASM" diff --git a/docs/overrides/main.html b/docs/overrides/main.html index 47d304677..bf114bd1c 100644 --- a/docs/overrides/main.html +++ b/docs/overrides/main.html @@ -2,6 +2,6 @@ {% block announce %} - Grafeo v0.5.43 is now available! + Grafeo v0.5.44 is now available! {% endblock %} diff --git a/docs/user-guide/cdc.md b/docs/user-guide/cdc.md index a0ee85d53..ec01b7b7c 100644 --- a/docs/user-guide/cdc.md +++ b/docs/user-guide/cdc.md @@ -74,6 +74,7 @@ Once CDC is enabled, every mutation records a `ChangeEvent` with: | Field | Description | |-----------------|-----------------------------------------------------------------------------| | `entity_id` | Node or edge ID | +| `graph` | The graph the entity is in: its name (`schema/name` inside a schema, and `schema/__default__` for a schema's default graph); `None` for the default graph | | `kind` | `Create`, `Update`, or `Delete` | | `epoch` | Commit epoch (monotonically increasing) | | `timestamp` | HLC timestamp (hybrid logical clock) | @@ -92,6 +93,11 @@ A node or edge created in a transaction has one `Create` event that shows it as the transaction left it: property and label changes made later in the same transaction are part of that event rather than events of their own. +Node and edge IDs repeat across named graphs, so an ID names an entity only +together with its graph. `history` on the database reads the default graph, and +on a session the session's current graph; `changes_between` returns the events +of every graph, each with its `graph`. + ### Per-entity history ```rust diff --git a/docs/user-guide/compact-store.md b/docs/user-guide/compact-store.md index 12fe833b3..33a78e518 100644 --- a/docs/user-guide/compact-store.md +++ b/docs/user-guide/compact-store.md @@ -15,7 +15,8 @@ memory and query wins. After ingesting data, call `compact()` to switch the data to a columnar layout with CSR adjacency. From 0.5.39, `compact()` is **non-destructive and writable**: it produces a layered store with an immutable columnar base plus a mutable overlay. Inserts and property updates after `compact()` land in the overlay; -`recompact()` merges the overlay back into a fresh base. +compacting again merges the overlay back into a fresh base (`recompact()` in Rust, +`compact()` again in Python and Node.js). Queries keep working across all supported languages, indexes (vector, text, hybrid) can be created and searched post-compact, and named graphs are preserved across @@ -144,7 +145,7 @@ memory reads. 3. **Builds** forward and backward CSR adjacency for each edge type 4. **Swaps** the database to a layered store: the new columnar tables become the immutable base and a mutable overlay is attached on top to absorb subsequent - writes. `recompact()` later folds the overlay back into a fresh base. + writes. Compacting again later folds the overlay back into a fresh base. The result is a `CompactStore` backed by: @@ -180,11 +181,11 @@ Since 0.5.39, `compact()` returns a layered store: an immutable columnar base pl mutable overlay. New inserts and property updates land in the overlay and are visible to subsequent queries (`get_node`, property reads, pattern matching, `list_graphs`). -Call `recompact()` to merge the overlay back into a fresh base: +Compact again to merge the overlay back into a fresh base (`recompact()` in Rust): db.compact() db.execute("INSERT (:Person {name: 'Mia'})") # lands in overlay - db.recompact() # merges overlay into new base + db.compact() # merges overlay into new base Indexes (`create_vector_index`, `create_text_index`, hybrid search) work on layered stores: vector/text scan and search now fall through both layers. @@ -193,7 +194,7 @@ stores: vector/text scan and search now fall through both layers. - **Overlay write path**: writes go through the overlay, which is less optimized than `LpgStore`'s full MVCC path. Sustained write-heavy workloads should stay on `LpgStore` - or call `recompact()` periodically. + or compact again periodically. - **Multi-label nodes**: nodes with multiple labels are stored under a compound key (e.g., `"Actor|Person"`, sorted alphabetically). A query like `MATCH (n:Person)` will not match nodes stored under `"Actor|Person"`. Workarounds: @@ -202,6 +203,11 @@ stores: vector/text scan and search now fall through both layers. - **Alternative:** assign a canonical "primary" label and store additional labels as a list property instead. - **No disk serialization**: `compact()` operates in memory. To persist a compacted database, use snapshot export (WASM) or save before compacting. +- **Missing properties read as empty values**: after `compact()`, when other entities with the + same label or type have a property that a node or edge lacks, reading it on that node or edge + gives the column's empty value (`''`, `0`, `0.0` or `false`) instead of null, `keys()` lists + it and `IS NULL` does not match it. Give every entity the property before compacting, or avoid relying on its absence + ([#542](https://github.com/GrafeoDB/grafeo/issues/542), planned for 0.5.45). ## Feature Flag diff --git a/docs/user-guide/cypher/aggregations.md b/docs/user-guide/cypher/aggregations.md index 0159743a3..a6f8d91ef 100644 --- a/docs/user-guide/cypher/aggregations.md +++ b/docs/user-guide/cypher/aggregations.md @@ -102,6 +102,9 @@ WHERE friend_count > 5 RETURN p.name, friend_count ``` +As in openCypher, an expression in `WITH` needs an alias (`WITH p.name AS name`), and a variable a `WITH` leaves +out is not visible after it. + ## UNWIND The `UNWIND` clause expands a list into rows: diff --git a/docs/user-guide/cypher/basic-queries.md b/docs/user-guide/cypher/basic-queries.md index d41e269e8..909bc8443 100644 --- a/docs/user-guide/cypher/basic-queries.md +++ b/docs/user-guide/cypher/basic-queries.md @@ -82,6 +82,10 @@ RETURN p.name, p.age ORDER BY p.age DESC, p.name ASC ``` +Nulls sort last in ascending order and first in descending order. + +Values of different types in one sort key, such as a property that holds a number on some nodes and a string on others, follow the openCypher order: maps, lists, paths, temporal values (zoned datetimes, datetimes, dates, zoned times, times, durations), strings, booleans, numbers, then null. Integers and floats compare as numbers, with NaN after infinity. Lists compare element by element with a prefix first, and maps by size, then keys, then values. Grafeo's own types fit in as follows: vectors after paths, bytes before strings and counters before numbers. + ## Limiting Results ```cypher diff --git a/docs/user-guide/cypher/mutations.md b/docs/user-guide/cypher/mutations.md index 62161b852..9a78b0ff5 100644 --- a/docs/user-guide/cypher/mutations.md +++ b/docs/user-guide/cypher/mutations.md @@ -143,6 +143,9 @@ MERGE (a:Person {id: 1})-[r:KNOWS]->(b:Person {id: 2}) SET r.weight = 0.5 ``` +When the pattern matches more than one existing node or relationship, `MERGE` binds each of them, one row per +match, and `ON MATCH` and a later `SET` apply to all of them. + ## FOREACH Iterate over a list and execute mutations for each element: @@ -163,7 +166,11 @@ FOREACH (person IN people | ## CALL Subqueries -Run a subquery for each input row. Variables from the outer query are visible inside the block. +Run a subquery for each input row. The subquery sees the outer variables its importing `WITH` lists, or the ones +a variable scope clause names (`CALL (p) { ... }`, `CALL (*) { ... }` for all of them, `CALL () { ... }` for +none). An importing `WITH` only lists variables: a `WHERE`, `DISTINCT`, alias or expression in it, or an +`ORDER BY`, `SKIP` or `LIMIT` right after it, is an error (a second `WITH` can do those). A subquery returns new +names only, so rename an imported variable to return it (`RETURN p AS person`). ```cypher -- Per-person friend count via subquery @@ -185,6 +192,23 @@ CALL { RETURN count(*) AS deleted } RETURN p.name, deleted + +-- Each person's oldest friend, with a scope clause (a person who knows nobody is left out) +MATCH (p:Person) +CALL (p) { + MATCH (p)-[:KNOWS]->(friend) + RETURN friend.name AS oldest_friend ORDER BY friend.age DESC LIMIT 1 +} +RETURN p.name, oldest_friend + +-- Parts joined by UNION, each with its own importing WITH +MATCH (p:Person) +CALL { + WITH p MATCH (p)-[:KNOWS]->(other) RETURN other.name AS contact + UNION + WITH p MATCH (other)-[:KNOWS]->(p) RETURN other.name AS contact +} +RETURN p.name, contact ``` ## UNION diff --git a/docs/user-guide/cypher/patterns.md b/docs/user-guide/cypher/patterns.md index b65a1136f..ffabbf5d4 100644 --- a/docs/user-guide/cypher/patterns.md +++ b/docs/user-guide/cypher/patterns.md @@ -84,3 +84,7 @@ MATCH (p:Person) OPTIONAL MATCH (p)-[:HAS_PET]->(pet) RETURN p.name, pet.name ``` + +A `WHERE` right after an `OPTIONAL MATCH` decides which matches count, also when it reads a variable bound +before: a row none of whose matches pass it keeps `null`. An `OPTIONAL MATCH` can also start a query (no match +gives one row of `null`s), and `EXISTS { }` and `COUNT { }` take `OPTIONAL MATCH` clauses. diff --git a/docs/user-guide/gql/basic-queries.md b/docs/user-guide/gql/basic-queries.md index 3b0f5beda..11f569b17 100644 --- a/docs/user-guide/gql/basic-queries.md +++ b/docs/user-guide/gql/basic-queries.md @@ -65,7 +65,7 @@ RETURN friend.name ## Ordering Results -Without `ORDER BY`, rows come in no particular order. The order can change between runs, builds and versions (parallel execution, compaction and planner changes all affect it), and so can which rows `LIMIT` keeps. When the order matters, say so with `ORDER BY`. To find code that relies on the order anyway, open the database with the `shuffle_unordered` option in tests (Python: `GrafeoDB(shuffle_unordered=True)`, Node.js: `GrafeoDB.create(path, { shuffleUnordered: true })`, Rust: `Config::with_shuffle_unordered(true)`): every result without `ORDER BY` then comes back in random order. +Without `ORDER BY`, rows come in no particular order. The order can change between runs, builds and versions (parallel execution, compaction and planner changes all affect it), and so can which rows `LIMIT` keeps. When the order matters, say so with `ORDER BY`. To find code that relies on the order anyway, open the database with the `shuffle_unordered` option in tests (Python: `GrafeoDB(shuffle_unordered=True)`, Node.js: `GrafeoDB.create(path, { shuffleUnordered: true })`, Rust: `Config::with_shuffle_unordered(true)`): every result without `ORDER BY` then comes back in random order (a streamed result within each chunk, so the stream keeps its bounded memory). ```sql -- Order by property @@ -93,6 +93,10 @@ RETURN p.name, p.age ORDER BY p.age DESC NULLS LAST ``` +Nulls sort last in ascending order and first in descending order, unless `NULLS FIRST` or `NULLS LAST` says otherwise, which holds in either direction. + +Values of different types in one sort key, such as a property that holds a number on some nodes and a string on others, follow one fixed order, the one openCypher defines: maps, lists, paths, temporal values (zoned datetimes, datetimes, dates, zoned times, times, durations), strings, booleans, numbers, then null. Integers and floats compare as numbers, with NaN after infinity. Lists compare element by element with a prefix first, and maps by size, then keys, then values. Grafeo's own types fit in as follows: vectors after paths, bytes before strings and counters before numbers. + ## Limiting Results ```sql @@ -134,6 +138,16 @@ OPTIONAL MATCH (c)-[:LOCATED_IN]->(city:City) RETURN p.name, c.name, city.name ``` +A condition on the optional part (a `WHERE` after it, or one inside its pattern) decides which matches count, +also when it reads a variable bound before: a row none of whose matches pass it keeps `null`. + +```sql +-- Friends of Alix's friends who are older than Alix; a friend without one keeps null +MATCH (a:Person {name: 'Alix'})-[:KNOWS]->(b) +OPTIONAL MATCH (b)-[:KNOWS]->(c WHERE c.age > a.age) +RETURN b.name, c.name +``` + ## SELECT (ISO Alternative to RETURN) The ISO GQL standard uses `SELECT` as an alternative to `RETURN`. The semantics are identical. @@ -160,7 +174,9 @@ FINISH ## Query Composition with NEXT -`NEXT` chains queries together: the output of the left query feeds into the right query as input. This enables multi-step transformations. +`NEXT` chains queries together: the rows the left query returns are the input of the right query, as the rows of +a `WITH` are for the clauses after it. Only the last query's `RETURN` is the result, and what a `RETURN` before +`NEXT` leaves out is not visible after it. ```sql -- Find friends, then filter by age @@ -189,6 +205,9 @@ WHERE friend.age > 25 RETURN p.name, friend.name ``` +An expression in `WITH` needs a name (`WITH p.name AS name`), and a variable a `WITH` leaves out is not visible +after it. + ## LET (Variable Binding) `LET` assigns computed values to variables for use in subsequent clauses. @@ -239,6 +258,30 @@ CALL { RETURN p.name, friend_count ``` +A variable scope clause limits what the subquery sees: `CALL (p) { ... }` sees only `p`, and `CALL () { ... }` +sees no outer variable. A subquery returns new names only: returning an outer variable is an error, so rename it +(`RETURN p AS person`). The body can order and cut its rows, for the top rows per input row, and combine queries +with `UNION`, `EXCEPT`, `INTERSECT` or `OTHERWISE`: + +```sql +-- Each person's oldest friend (a person who knows nobody is left out; OPTIONAL CALL keeps them) +MATCH (p:Person) +CALL (p) { + MATCH (p)-[:KNOWS]->(friend) + RETURN friend.name AS oldest_friend ORDER BY friend.age DESC LIMIT 1 +} +RETURN p.name, oldest_friend + +-- Whom each person knows or is known by +MATCH (p:Person) +CALL (p) { + MATCH (p)-[:KNOWS]->(other) RETURN other.name AS contact + UNION + MATCH (other)-[:KNOWS]->(p) RETURN other.name AS contact +} +RETURN p.name, contact +``` + ### OPTIONAL CALL `OPTIONAL CALL` returns `null` for output variables when the subquery produces no results, instead of filtering the row. diff --git a/docs/user-guide/gql/mutations.md b/docs/user-guide/gql/mutations.md index c3dbb3c85..ad2d66487 100644 --- a/docs/user-guide/gql/mutations.md +++ b/docs/user-guide/gql/mutations.md @@ -224,6 +224,9 @@ ON CREATE SET p.name = person.name, p.created = timestamp() ON MATCH SET p.lastSeen = timestamp() ``` +When the pattern matches more than one existing node or relationship, `MERGE` binds each of them, one row per +match, and `ON MATCH` and a later `SET` apply to all of them. + ## LOAD DATA (Multi-Format Import) Import data from external files directly in GQL: diff --git a/docs/user-guide/gql/schema.md b/docs/user-guide/gql/schema.md index 2d2c356d0..40ca9fb84 100644 --- a/docs/user-guide/gql/schema.md +++ b/docs/user-guide/gql/schema.md @@ -214,7 +214,7 @@ DROP INDEX index_name DROP INDEX IF EXISTS index_name ``` -When the key comes from the rows of the query (`UNWIND`, an earlier `MATCH`), an index lookup can miss values that `=` treats as equal without one: a number stored as a string in another spelling (`'42.0'` or `'042'` for `42`), and floats that differ only in the last digit (`0.1 + 0.2` for `0.3`). On properties that mix strings and numbers, or that hold computed floats, such a query can return fewer rows once an index exists. Keys written in the query itself or passed as parameters compare exactly either way. Keep a property's values of one type to get the same rows either way. +When the pattern follows an earlier clause (`UNWIND`, another `MATCH`), an index lookup can miss values that `=` treats as equal without one: a number stored as a string in another spelling (`'42.0'` or `'042'` for `42`), and floats that differ only in the last digit (`0.1 + 0.2` for `0.3`). This holds whether the key comes from the rows, is written in the query or is passed as a parameter. On properties that mix strings and numbers, or that hold computed floats, such a query can return fewer rows once an index exists. Keep a property's values of one type to get the same rows either way. ### Text Indexes diff --git a/packages/grafeo-cli-npm/package.json b/packages/grafeo-cli-npm/package.json index a2cb49d91..3cd724199 100644 --- a/packages/grafeo-cli-npm/package.json +++ b/packages/grafeo-cli-npm/package.json @@ -1,6 +1,6 @@ { "name": "@grafeo-db/cli", - "version": "0.5.43", + "version": "0.5.44", "description": "Command-line interface for Grafeo graph database", "license": "Apache-2.0", "repository": { @@ -25,11 +25,11 @@ "LICENSE" ], "optionalDependencies": { - "@grafeo-db/cli-linux-x64": "0.5.43", - "@grafeo-db/cli-linux-arm64": "0.5.43", - "@grafeo-db/cli-darwin-x64": "0.5.43", - "@grafeo-db/cli-darwin-arm64": "0.5.43", - "@grafeo-db/cli-win32-x64": "0.5.43" + "@grafeo-db/cli-linux-x64": "0.5.44", + "@grafeo-db/cli-linux-arm64": "0.5.44", + "@grafeo-db/cli-darwin-x64": "0.5.44", + "@grafeo-db/cli-darwin-arm64": "0.5.44", + "@grafeo-db/cli-win32-x64": "0.5.44" }, "engines": { "node": ">=18" diff --git a/packages/grafeo-cli-python/grafeo_cli/__init__.py b/packages/grafeo-cli-python/grafeo_cli/__init__.py index a2d25878f..dc4dc5fdf 100644 --- a/packages/grafeo-cli-python/grafeo_cli/__init__.py +++ b/packages/grafeo-cli-python/grafeo_cli/__init__.py @@ -12,7 +12,7 @@ import sys from pathlib import Path -__version__ = "0.5.43" +__version__ = "0.5.44" # GitHub release download URL template _GITHUB_RELEASE = "https://github.com/GrafeoDB/grafeo/releases/download" diff --git a/packages/grafeo-cli-python/pyproject.toml b/packages/grafeo-cli-python/pyproject.toml index bc40868ac..b353e7d25 100644 --- a/packages/grafeo-cli-python/pyproject.toml +++ b/packages/grafeo-cli-python/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "grafeo-cli" -version = "0.5.43" +version = "0.5.44" description = "Command-line interface for Grafeo graph database" readme = "README.md" license = { text = "Apache-2.0"} diff --git a/scripts/build-wasm.sh b/scripts/build-wasm.sh index f81bf0ddc..6fe9fd6e8 100644 --- a/scripts/build-wasm.sh +++ b/scripts/build-wasm.sh @@ -143,8 +143,8 @@ if [[ "$FEATURES" == *"full"* ]]; then FAIL_THRESHOLD=1468006 # 1.4 MB LABEL="full profile" else - WARN_THRESHOLD=696320 # 680 KB - FAIL_THRESHOLD=737280 # 720 KB + WARN_THRESHOLD=757760 # 740 KB + FAIL_THRESHOLD=778240 # 760 KB LABEL="browser profile" fi diff --git a/scripts/difftest/corpus.py b/scripts/difftest/corpus.py new file mode 100644 index 000000000..7a624c905 --- /dev/null +++ b/scripts/difftest/corpus.py @@ -0,0 +1,600 @@ +"""The difftest corpus: fixtures and the queries that run on them. + +The fixtures are built so that mistakes show. Nodes carry a property `w` of 100 and +up, edges one below 100, so an edge read as a node (or the reverse) gives a value from +the wrong range. `chain` has 5,000 nodes, so sorts, cuts and skips cross row batches. + +Each case has an id, the query, the languages it runs in, its fixture, and whether its +rows are ordered (unordered rows compare as a multiset). Queries only read: each fixture +is built once and shared by its cases. Results are matched by id and +language, so never change or reuse an id: add new cases at the end of a section, or +start a new section. +""" + +from __future__ import annotations + +from dataclasses import dataclass + +GQL = ("gql",) +CYPHER = ("cypher",) +BOTH = ("gql", "cypher") +LANGUAGES = frozenset(BOTH) + + +@dataclass(frozen=True) +class Case: + id: str + query: str + languages: tuple[str, ...] = BOTH + fixture: str = "social" + ordered: bool = False + + +def social(grafeo): + """Five people and three cities. KNOWS: a triangle Alix, Gus, Vincent, plus + Alix to Jules and Jules to Mia. LIVES_IN: Alix in Amsterdam, Gus in Berlin, Mia in + Paris. Jules, Mia and Berlin have no `w`.""" + db = grafeo.GrafeoDB() + people = [ + ("Alix", 30, 100), + ("Gus", 25, 101), + ("Vincent", 40, 103), + ("Jules", 35, None), + ("Mia", 28, None), + ] + for name, age, w in people: + props = f"name: '{name}', age: {age}" + (f", w: {w}" if w is not None else "") + db.execute(f"INSERT (:Person {{{props}}})") + for name, w in [("Amsterdam", 104), ("Berlin", None), ("Paris", 105)]: + props = f"name: '{name}'" + (f", w: {w}" if w is not None else "") + db.execute(f"INSERT (:City {{{props}}})") + knows = [ + ("Alix", "Gus", 2010, 1), + ("Gus", "Vincent", 2012, 2), + ("Vincent", "Alix", 2015, 3), + ("Jules", "Mia", 2020, 4), + ("Alix", "Jules", 2018, 5), + ] + for a, b, since, w in knows: + db.execute( + f"MATCH (a:Person {{name: '{a}'}}), (b:Person {{name: '{b}'}}) " + f"INSERT (a)-[:KNOWS {{since: {since}, w: {w}}}]->(b)" + ) + lives = [ + ("Alix", "Amsterdam", 5, 6), + ("Gus", "Berlin", 3, 7), + ("Mia", "Paris", 1, 8), + ] + for a, c, years, w in lives: + db.execute( + f"MATCH (a:Person {{name: '{a}'}}), (c:City {{name: '{c}'}}) " + f"INSERT (a)-[:LIVES_IN {{years: {years}, w: {w}}}]->(c)" + ) + return db + + +def chain(grafeo): + """5,000 `N` nodes (`i` 0 to 4999, `m` = i % 7) linked by NEXT edges (`k` = i).""" + db = grafeo.GrafeoDB() + db.execute("UNWIND range(0, 4999) AS i INSERT (:N {i: i, m: i % 7})") + db.execute( + "MATCH (a:N), (b:N) WHERE b.i = a.i + 1 INSERT (a)-[:NEXT {k: a.i}]->(b)" + ) + return db + + +def empty(grafeo): + return grafeo.GrafeoDB() + + +FIXTURES = {"social": social, "chain": chain, "empty": empty} + +CASES: list[Case] = [] + + +def case( + case_id: str, + query: str, + languages: tuple[str, ...] = BOTH, + fixture: str = "social", + ordered: bool = False, +) -> None: + CASES.append(Case(case_id, query, languages, fixture, ordered)) + + +def ordered( + case_id: str, query: str, languages: tuple[str, ...] = BOTH, fixture: str = "social" +): + case(case_id, query, languages, fixture, ordered=True) + + +# The cases, one per line. +# fmt: off + +# A: nodes and edges returned through ORDER BY, LIMIT, SKIP and DISTINCT +ordered("A1", "MATCH (a:Person)-[r:KNOWS]->(b) RETURN r ORDER BY r.since") +ordered("A2", "MATCH (a:Person)-[r:KNOWS]->(b) RETURN r ORDER BY r.since DESC LIMIT 2") +ordered("A3", "MATCH (a:Person)-[r:KNOWS]->(b) RETURN r ORDER BY r.since SKIP 2") +ordered("A4", "MATCH (a:Person)-[r:KNOWS]->(b) RETURN r ORDER BY r.since SKIP 1 LIMIT 2") +case("A5", "MATCH (a:Person)-[r:KNOWS]->(b) RETURN r SKIP 1 LIMIT 2") +ordered("A6", "MATCH (a:Person)-[r]->(b) RETURN DISTINCT type(r) AS t ORDER BY t") +ordered("A7", "MATCH (a:Person)-[r]->(b) RETURN DISTINCT r ORDER BY r.w") +ordered("A8", "MATCH (a:Person)-[r]->(b) RETURN a, r, b ORDER BY r.w LIMIT 3") +ordered("A9", "MATCH p = (a:Person)-[:KNOWS]->(b) RETURN p ORDER BY a.name LIMIT 2") +ordered("A10", "MATCH (a:Person) RETURN a ORDER BY a.age DESC SKIP 1 LIMIT 2") +ordered("A11", "MATCH (a:Person)-[r]->(b) RETURN r ORDER BY type(r), r.w") +ordered("A12", "MATCH (a:Person)-[r]->(b) RETURN r, b ORDER BY b.name, r.w") +ordered("A13", "MATCH (a:Person)-[r*1..2]->(b) RETURN r ORDER BY size(r), b.name, a.name LIMIT 3") +ordered("A14", "MATCH (a:Person)-[r*1..2]->(b) RETURN DISTINCT b ORDER BY b.name") +ordered("A15", "MATCH (a:Person) RETURN a.name AS n, a ORDER BY n LIMIT 2") +ordered("A16", "MATCH (a:Person)-[r]->(b) RETURN r ORDER BY r.w DESC SKIP 7") +case("A17", "MATCH (a:Person)-[r]->(b) RETURN DISTINCT r SKIP 0 LIMIT 100") +case("A18", "MATCH (a:Person)-[r]->(b) RETURN r LIMIT 0") +case("A19", "MATCH (a:Person)-[r]->(b) RETURN r SKIP 100") + +# B: nodes and edges through ORDER BY, LIMIT, SKIP and DISTINCT before RETURN, then a +# property read or a later pattern +case("B1", "MATCH (a:Person)-[r:KNOWS]->(b) WITH r ORDER BY r.since DESC LIMIT 2 RETURN r.w AS w, r.since AS s, type(r) AS t", CYPHER) +case("B2", "MATCH (a:Person) WITH a ORDER BY a.age SKIP 1 LIMIT 2 MATCH (a)-[:KNOWS]->(b) RETURN a.name, b.name", CYPHER) +case("B3", "MATCH (a:Person)-[r]->(b) WITH DISTINCT r RETURN r.w AS w", CYPHER) +case("B4", "MATCH (a:Person)-[r]->(b) WITH r, b ORDER BY r.w SKIP 2 RETURN r.w, b.name", CYPHER) +ordered("B5", "MATCH (a:Person)-[r]->(b) LET x = r RETURN x.w AS w ORDER BY w", GQL) +ordered("B6", "MATCH (a:Person)-[r]->(b) LET x = r RETURN x ORDER BY x.w", GQL) +case("B7", "MATCH (a:Person)-[r]->(b) WITH r AS e ORDER BY e.w LIMIT 3 RETURN e.w, type(e)", CYPHER) +case("B8", "MATCH (a:Person)-[r]->(b) WITH r ORDER BY r.w LIMIT 3 MATCH ()-[r]->(c) RETURN r.w, c.name", CYPHER) +case("B9", "MATCH (a:Person) WITH a ORDER BY a.name LIMIT 1 MATCH (a)-[r]->(x) RETURN type(r) AS t, x.name AS n", CYPHER) +case("B10", "MATCH (a:Person)-[r]->(b) WITH r, a ORDER BY a.name RETURN startNode(r).name AS s, endNode(r).name AS e", CYPHER) +ordered("B11", "MATCH (a:Person)-[r]->(b) FILTER r.w > 2 RETURN r.w ORDER BY r.w", GQL) +case("B12", "MATCH (a:Person)-[r]->(b) WITH DISTINCT a RETURN a.w AS w, a.name AS n", CYPHER) +case("B13", "MATCH (a:Person)-[r]->(b) WITH r SKIP 1 RETURN r.w AS w, type(r) AS t", CYPHER) +case("B14", "MATCH (a:Person)-[r]->(b) WITH r ORDER BY r.w DESC LIMIT 2 RETURN r", CYPHER) +case("B15", "MATCH (a:Person)-[r]->(b) WITH a, r ORDER BY r.w LIMIT 4 RETURN a.name, r.w, type(r)", CYPHER) +case("B16", "MATCH (a:Person)-[r]->(b) WITH collect(r) AS rs UNWIND rs AS x RETURN x.w AS w", CYPHER) +ordered("B17", "MATCH (a:Person)-[r]->(b) LET x = r LET y = x.w RETURN y ORDER BY y", GQL) + +# C: set operations +case("C1", "MATCH (a:Person) RETURN a.name AS n UNION MATCH (c:City) RETURN c.name AS n") +case("C2", "MATCH (a:Person) RETURN a AS x UNION MATCH (c:City) RETURN c AS x") +case("C3", "MATCH ()-[r:KNOWS]->() RETURN r AS x UNION ALL MATCH ()-[r:LIVES_IN]->() RETURN r AS x") +case("C4", "MATCH (a:Person) RETURN a AS x UNION ALL MATCH ()-[r:LIVES_IN]->() RETURN r AS x") +case("C5", "MATCH ()-[x:KNOWS]->() RETURN x UNION ALL MATCH (x:City) RETURN x") +case("C6", "MATCH (a:Person) WHERE a.age > 30 RETURN a AS x UNION MATCH (c:City) RETURN c AS x UNION MATCH ()-[r:LIVES_IN]->() RETURN r AS x") +case("C7", "MATCH (a:Person)-[r*1..2]->(b) WHERE a.name = 'Jules' RETURN r UNION ALL MATCH ()-[r:LIVES_IN]->() RETURN r") +case("C8", "MATCH (a:Person)-[]->(b) RETURN a EXCEPT ALL MATCH (a:Person {name: 'Alix'}) RETURN a", GQL) +case("C9", "MATCH (a:Person)-[]->(b) RETURN a INTERSECT ALL MATCH (a:Person)-[:KNOWS]->(b) RETURN a", GQL) +case("C10", "UNWIND [1, 2, 2, 3, null] AS x RETURN x EXCEPT UNWIND [2, null] AS x RETURN x", GQL, "empty") +case("C11", "UNWIND [1, 2, 2, 3] AS x RETURN x INTERSECT ALL UNWIND [2, 2, 2, 4] AS x RETURN x", GQL, "empty") +case("C12", "MATCH (a:Person {name: 'Mia'}) RETURN a OTHERWISE MATCH (c:City) RETURN c AS a", GQL) +case("C13", "CALL { MATCH (a:Person) RETURN a AS x UNION MATCH (c:City) RETURN c AS x } RETURN x.name AS n", CYPHER) +case("C14", "MATCH (a:Person) WHERE a.age > 30 RETURN a.name AS n, a AS x UNION MATCH (a:Person) WHERE a.age < 30 RETURN a.name AS n, a AS x") +case("C16", "MATCH (a:Person)-[r]->(b) RETURN r, a UNION ALL MATCH (a:Person)-[r]->(b) RETURN r, b AS a") +case("C17", "MATCH (a:Person)-[:KNOWS]->(b) | (a:Person)-[:LIVES_IN]->(b) RETURN a.name, b.name", GQL) +case("C18", "MATCH (a:Person) WHERE a.name = 'Alix' RETURN a UNION ALL MATCH (a:Person) WHERE a.name = 'Alix' RETURN a") +case("C19", "MATCH (a:Person) WHERE a.name = 'Alix' RETURN a UNION MATCH (a:Person) WHERE a.name = 'Alix' RETURN a") +case("C20", "MATCH (a:Person)-[r:KNOWS]->(b) RETURN a, r EXCEPT MATCH (a:Person {name: 'Alix'})-[r:KNOWS]->(b) RETURN a, r", GQL) +case("C21", "MATCH (a:Person) RETURN a.name AS n EXCEPT MATCH (a:Person) WHERE a.age > 29 RETURN a.name AS n", GQL) +case("C22", "MATCH (a:Person) WHERE a.age > 100 RETURN a.name AS n OTHERWISE MATCH (c:City) RETURN c.name AS n", GQL) +case("C23", "MATCH (a:Person)-[r]->(b) WHERE r.w < 3 RETURN r UNION MATCH (a:Person)-[r]->(b) WHERE r.w > 1 RETURN r") +case("C24", "MATCH (x:Person {name: 'Gus'}) RETURN x UNION ALL MATCH ()-[x:KNOWS]->() WHERE x.w = 1 RETURN x UNION ALL MATCH (x:City {name: 'Paris'}) RETURN x") +case("C25", "MATCH (a:Person) RETURN a.w AS w UNION ALL MATCH ()-[r:KNOWS]->() RETURN r.w AS w") + +# D: sort keys that are not returned, and sort keys on aliases +ordered("D1", "MATCH (a:Person) RETURN a AS x ORDER BY x.age") +ordered("D2", "MATCH (a:Person) RETURN a.name AS n ORDER BY n") +ordered("D3", "MATCH (a:Person) RETURN a.name AS n ORDER BY a.age DESC") +ordered("D4", "MATCH (a:Person)-[r]->(b) RETURN a AS x, b AS y ORDER BY x.age, y.name") +ordered("D5", "MATCH (a:Person) RETURN DISTINCT a AS x ORDER BY x.name") +ordered("D6", "MATCH (a:Person) RETURN a AS x ORDER BY x.age SKIP 1 LIMIT 2") +ordered("D7", "MATCH (a:Person) RETURN a.age AS g, count(*) AS c ORDER BY g") +ordered("D8", "MATCH (a:Person)-[r]->(b) RETURN r AS e, b.name AS n ORDER BY e.w DESC, n") +ordered("D9", "MATCH (a:Person) RETURN a.name AS n ORDER BY toLower(n)") +ordered("D10", "MATCH (a:Person) RETURN a AS x ORDER BY labels(x)[0], x.name") +ordered("D11", "MATCH (a:Person)-[r]->(b) RETURN type(r) AS t ORDER BY r.w") +ordered("D12", "MATCH (a:Person) RETURN a.name ORDER BY a.age") +ordered("D13", "MATCH (a:Person) RETURN a ORDER BY a.age DESC LIMIT 1") +case("D14", "MATCH (a:Person) WITH a AS x ORDER BY x.age RETURN x.name AS n", CYPHER) +ordered("D15", "MATCH (a:Person) RETURN a.name AS n, a.age AS g ORDER BY g DESC, n") +ordered("D16", "MATCH (a:Person) RETURN a AS x ORDER BY x.age DESC, x.name LIMIT 3") +ordered("D17", "MATCH (a:Person)-[r]->(b) RETURN r AS e ORDER BY e.w, e.since LIMIT 4") +ordered("D18", "MATCH (a:Person) RETURN a AS x, a.name AS n ORDER BY x.age") +ordered("D19", "MATCH (a:Person) RETURN a AS x ORDER BY x.w") +ordered("D20", "MATCH (a:Person)-[r]->(b) RETURN DISTINCT r AS e ORDER BY e.w LIMIT 2") +ordered("D21", "MATCH (a:Person) RETURN a AS x ORDER BY x.age + 1") +ordered("D22", "MATCH (a:Person) RETURN a.name AS n, a AS x ORDER BY x.age SKIP 2") + +# E: properties and ids read after a node or edge went through ORDER BY, LIMIT, SKIP +# or DISTINCT +for case_id, query in [ + ("E1", "MATCH (a:Person)-[r]->(b) WITH r ORDER BY type(r), r.w LIMIT 3 RETURN r.w AS w"), + ("E2", "MATCH (a:Person)-[r]->(b) WITH r ORDER BY toString(r.w) LIMIT 3 RETURN r.w AS w"), + ("E3", "MATCH (a:Person) WITH a ORDER BY a.age DESC LIMIT 2 RETURN a.w AS w, a.name AS n"), + ("E4", "MATCH (a:Person)-[r]->(b) WITH a, r ORDER BY toString(r.w) RETURN r.w AS w, a.name AS n"), + ("E5", "MATCH (a:Person) WITH a ORDER BY a.name SKIP 1 RETURN a.w AS w, a.name AS n"), + ("E6", "MATCH p = (a:Person)-[:KNOWS]->(b) WITH p, a ORDER BY a.name LIMIT 2 RETURN [n IN nodes(p) | n.w] AS ws"), + ("E7", "MATCH (a:Person)-[r]->(b) WITH r ORDER BY r.w LIMIT 3 RETURN id(r) AS i, r.w AS w"), + ("E8", "MATCH (a:Person)-[r]->(b) WITH r ORDER BY r.w LIMIT 2 RETURN properties(r) AS p"), + ("E9", "MATCH (a:Person)-[r:KNOWS]->(b) WITH r ORDER BY r.w DESC RETURN r.since AS s, r.w AS w"), + ("E10", "MATCH (a:Person)-[r]->(b) WITH DISTINCT r, b ORDER BY b.name LIMIT 4 RETURN r.w AS w, b.w AS bw"), + ("E11", "MATCH (a:Person)-[r]->(b) WITH r, a ORDER BY a.age, r.w SKIP 1 LIMIT 3 RETURN r.w AS w, a.w AS aw"), + ("E12", "MATCH (a:Person)-[r]->(b) WITH r ORDER BY r.w LIMIT 3 MATCH (x)-[r]->(y) RETURN x.w AS xw, y.name AS y"), + ("E13", "MATCH (a:Person)-[r]->(b) WITH r ORDER BY labels(startNode(r))[0], r.w LIMIT 2 RETURN r.w AS w"), + ("E14", "MATCH (a:Person)-[r]->(b) WITH b ORDER BY b.name LIMIT 3 RETURN b.w AS w, b.name AS n"), + ("E15", "MATCH (a:Person)-[r]->(b) WITH r ORDER BY r.w DESC SKIP 5 RETURN keys(r) AS k, r.w AS w"), +]: + case(case_id, query, CYPHER) +ordered("E16", "MATCH (a:Person)-[r]->(b) RETURN r.w AS w, a.w AS aw ORDER BY r.w SKIP 2 LIMIT 3") +ordered("E17", "MATCH (a:Person)-[r]->(b) RETURN DISTINCT a.w AS aw, b.w AS bw ORDER BY aw, bw") + +# F: cuts across row batches +ordered("F1", "MATCH (n:N) RETURN n.i ORDER BY n.i SKIP 2046 LIMIT 5", fixture="chain") +ordered("F2", "MATCH (n:N) RETURN n.i ORDER BY n.i DESC LIMIT 3", fixture="chain") +ordered("F3", "MATCH (n:N) RETURN n ORDER BY n.i SKIP 4998", fixture="chain") +ordered("F4", "MATCH (n:N)-[r]->(m) RETURN r ORDER BY r.k SKIP 2047 LIMIT 3", fixture="chain") +ordered("F5", "MATCH (n:N) RETURN DISTINCT n.m AS m ORDER BY m", fixture="chain") +ordered("F6", "MATCH (n:N)-[r]->(m) WHERE r.k % 1000 = 0 RETURN r ORDER BY r.k", fixture="chain") +case("F7", "MATCH (n:N) RETURN count(*) AS c", fixture="chain") +case("F8", "MATCH (n:N) WITH n ORDER BY n.i DESC LIMIT 3 MATCH (n)<-[r]-(p) RETURN n.i, r.k, p.i", CYPHER, "chain") +case("F9", "UNWIND range(1, 5000) AS i RETURN i SKIP 2047 LIMIT 3", fixture="empty") +ordered("F10", "MATCH (n:N) RETURN n.i AS i ORDER BY i LIMIT 2050", fixture="chain") +ordered("F11", "MATCH (n:N)-[r]->(m) RETURN DISTINCT r ORDER BY r.k LIMIT 2", fixture="chain") +case("F12", "MATCH (n:N) RETURN n SKIP 2047 LIMIT 2", fixture="chain") +case("F13", "MATCH (n:N)-[r]->(m) RETURN r SKIP 4997", fixture="chain") +case("F14", "MATCH (n:N)-[r]->(m) WITH r SKIP 4990 RETURN r.k AS k", CYPHER, "chain") +case("F15", "MATCH (n:N) WITH n ORDER BY n.i SKIP 2045 LIMIT 5 RETURN n.i AS i, n.m AS m", CYPHER, "chain") +ordered("F16", "MATCH (n:N) RETURN n.m AS m, count(*) AS c ORDER BY m", fixture="chain") + +# G: values of different types in one column +ordered("G1", "UNWIND [3, 'a', 2.5, null, true, [1, 2], {k: 1}] AS x RETURN x ORDER BY x", fixture="empty") +ordered("G2", "UNWIND [3, 'a', 2.5, null, true, [1, 2], {k: 1}] AS x RETURN x ORDER BY x DESC LIMIT 3", fixture="empty") +case("G3", "UNWIND [1, 1, 1.0, '1', null, null, true] AS x RETURN DISTINCT x", fixture="empty") +case("G4", "UNWIND [1, 'a', 2.5, null] AS x RETURN x SKIP 1 LIMIT 2", fixture="empty") +case("G5", "UNWIND [1, 'a', null] AS x RETURN x EXCEPT UNWIND ['a'] AS x RETURN x", GQL, "empty") +ordered("G6", "UNWIND range(1, 3000) AS i RETURN CASE WHEN i % 2 = 0 THEN i ELSE toString(i) END AS v ORDER BY i LIMIT 3", fixture="empty") +case("G7", "UNWIND range(1, 3000) AS i WITH CASE WHEN i < 2500 THEN i ELSE 'x' END AS v RETURN v SKIP 2498 LIMIT 3", CYPHER, "empty") +ordered("G8", "UNWIND range(1, 3000) AS i RETURN CASE WHEN i > 2048 THEN 'late' ELSE i END AS v ORDER BY i DESC LIMIT 2", fixture="empty") +ordered("G9", "UNWIND [date('2024-01-02'), date('2023-05-06')] AS d RETURN d ORDER BY d", fixture="empty") +ordered("G10", "UNWIND [1, 2, 3] AS x RETURN x, CASE WHEN x = 2 THEN 'two' ELSE x END AS y ORDER BY x DESC SKIP 1", fixture="empty") +case("G11", "UNWIND range(1, 3000) AS i RETURN DISTINCT CASE WHEN i > 2048 THEN 'late' ELSE i % 3 END AS v", fixture="empty") +ordered("G12", "UNWIND range(1, 5) AS i RETURN i, i * 1.5 AS f ORDER BY f DESC LIMIT 2", fixture="empty") + +# H: GQL FOR and LET with ordering +ordered("H1", "FOR x IN [3, 1, 2] RETURN x ORDER BY x LIMIT 2", GQL, "empty") +ordered("H2", "MATCH (a:Person) LET n = a.name RETURN n ORDER BY n SKIP 1 LIMIT 2", GQL) +ordered("H3", "MATCH (a:Person) LET b = a RETURN b ORDER BY b.age LIMIT 2", GQL) + +# I: RETURN * with ORDER BY +ordered("I1", "MATCH (a:Person)-[r:LIVES_IN]->(c) RETURN * ORDER BY r.years") +ordered("I2", "MATCH (a:Person)-[r:LIVES_IN]->(c) RETURN * ORDER BY a.name LIMIT 1") + +# J: aggregation over nodes and edges +ordered("J1", "MATCH (a:Person)-[r]->(b) RETURN b.name AS n, count(r) AS c ORDER BY c DESC, n") +case("J2", "MATCH (a:Person)-[r]->(b) RETURN r, count(*) AS c") +case("J3", "MATCH (a:Person)-[r]->(b) RETURN collect(r.w) AS ws") + +# K: variables bound before a later pattern, and values through a later MATCH +for case_id, query, languages in [ + ("K1", "MATCH ()-[r]->() MATCH (x)-[r]->(y) RETURN r.w AS w, x.name AS x, y.name AS y", BOTH), + ("K2", "MATCH (a)-[r]->(b) MATCH (a)-[r]->(c) RETURN r.w AS w, c.name AS c", BOTH), + ("K3", "MATCH (a)-[r]->(b) MATCH (x)<-[r]-(y) RETURN r.w AS w, x.name AS x, y.name AS y", BOTH), + ("K4", "MATCH (a)-[r]->(b) MATCH (x)-[r]-(y) RETURN r.w AS w, x.name AS x", BOTH), + ("K5", "MATCH ()-[r:KNOWS]->() MATCH ()-[r:LIVES_IN]->() RETURN count(*) AS n", BOTH), + ("K6", "MATCH ()-[r]->() MATCH (x)-[r {w: 3}]->(y) RETURN x.name AS x", BOTH), + ("K7", "MATCH ()-[r]->() CALL { WITH r MATCH (x)-[r]->(y) RETURN y.name AS t } RETURN r.w AS w, t", BOTH), + ("K8", "MATCH ()-[r]->() CALL { WITH * MATCH (x)-[r]->(y) RETURN y.name AS t } RETURN r.w AS w, t", BOTH), + ("K9", "MATCH (a)-->(b) CALL { WITH a, b MATCH (b)-->(a) RETURN count(*) AS c } RETURN a.name AS a, b.name AS b, c", BOTH), + ("K10", "MATCH (a:Person) MATCH (c:City) RETURN a.name AS n, a.w AS w, c.name AS c", BOTH), + ("K11", "MATCH ()-[r:KNOWS]->() MATCH (c:City) RETURN r.w AS w, c.name AS c", BOTH), + ("K12", "MATCH ()-[r:KNOWS]->() MATCH (c:City) WHERE r.w > 2 RETURN r.w AS w, c.name AS c", BOTH), + ("K13", "MATCH (a:Person) MATCH (a)-[:LIVES_IN]->(c) RETURN a.w AS w, c.w AS cw", BOTH), + ("K14", "MATCH (a:Person) WITH a MATCH (b:Person) WHERE a.age < b.age RETURN a.name AS a, b.name AS b", BOTH), + ("K15", "MATCH (a:Person)-[r:KNOWS]->(b) MATCH (c:City) RETURN a.w AS aw, r.w AS rw, b.w AS bw, c.w AS cw", BOTH), + ("K16", "MATCH (a:Person) MATCH (b:Person) RETURN count(*) AS n", BOTH), + ("K17", "MATCH (a:Person) MATCH (c:City) RETURN a, c ORDER BY a.name, c.name LIMIT 4", BOTH), + ("K18", "MATCH ()-[r:KNOWS]->() MATCH (c:City) RETURN r ORDER BY r.w, c.name LIMIT 3", BOTH), + ("K19", "MATCH (a:Person) MATCH p = (a)-[:KNOWS*1..2]->(b) RETURN a.name AS a, length(p) AS l, b.w AS bw", BOTH), + ("K20", "MATCH ()-[r:KNOWS]->() MATCH p = (x:Person {name: 'Mia'})-->(y) RETURN r.w AS w, y.name AS y", BOTH), + ("K21", "MATCH (a:Person) MATCH (c:City) RETURN sum(a.w) AS s, count(a.w) AS n", BOTH), + ("K22", "UNWIND [1, 2] AS k MATCH (a:Person {name: 'Jules'}) RETURN k, a.w AS w", BOTH), + ("K23", "MATCH (a:Person {name: 'Jules'}) WITH a MATCH (c:City {name: 'Paris'}) RETURN a.w AS w, c.w AS cw", BOTH), + ("K24", "MATCH (a:Person)-[r]->(b) MATCH (a)-[s]->(c) WHERE r <> s RETURN r.w AS rw, s.w AS sw", BOTH), + ("K25", "MATCH (a)-[r*1..2]->(b) MATCH (x)-[r*1..2]->(y) RETURN count(*) AS n", BOTH), + ("K26", "MATCH (a)-[r]->(b)-[r]->(c) RETURN count(*) AS n", BOTH), + ("K27", "MATCH ()-[r]->() OPTIONAL MATCH (x)-[r:KNOWS]->(y) RETURN r.w AS w, y.name AS y", BOTH), + ("K28", "MATCH (a:Person) OPTIONAL MATCH (a)-[:LIVES_IN]->(c) RETURN a.name AS n, c.name AS c", BOTH), + ("K29", "MATCH (a:Person) MATCH (b:Person) WHERE id(a) < id(b) RETURN count(*) AS n", BOTH), + ("K30", "MATCH (a:Person)-[r:KNOWS]->(b) MATCH (b)-[s:LIVES_IN]->(c) RETURN a.name AS a, r.w AS rw, s.w AS sw, c.w AS cw", BOTH), + ("K31", "MATCH (a:Person) MATCH (c:City) WITH a, c ORDER BY a.name, c.name LIMIT 3 RETURN a.w AS w, c.w AS cw", CYPHER), + ("K32", "MATCH (a:Person) MATCH (c:City) RETURN DISTINCT a.w AS w", BOTH), +]: + case(case_id, query, languages) +case("K40", "MATCH (a:N) WHERE a.i < 3 MATCH (b:N) WHERE b.i >= 4990 RETURN a.i AS ai, b.i AS bi, b.k AS bk", fixture="chain") +case("K41", "MATCH (a:N) WHERE a.i < 1 MATCH (b:N) RETURN count(*) AS n, sum(b.i) AS s", fixture="chain") +case("K42", "MATCH ()-[r:NEXT]->() WHERE r.k < 3 MATCH (b:N) WHERE b.i < 2 RETURN r.k AS k, r.i AS ri, b.i AS i", fixture="chain") +case("K43", "MATCH ()-[r:NEXT]->() WHERE r.k < 2100 MATCH (b:N {i: 7}) RETURN count(r.k) AS n, sum(r.k) AS s, count(r.i) AS ri", fixture="chain") + +# L: ORDER BY across types, NaN and nulls, top-K and percentiles +for case_id, query, languages, fixture in [ + ("L1", "UNWIND [3, 'a', 2.5, true, null, [1], {k: 1}] AS x RETURN x ORDER BY x", BOTH, "social"), + ("L2", "UNWIND [3, 'a', 2.5, true, null, [1], {k: 1}] AS x RETURN x ORDER BY x DESC", BOTH, "social"), + ("L3", "MATCH (n) RETURN n.w AS w ORDER BY w", BOTH, "social"), + ("L4", "MATCH (n) RETURN n.w AS w ORDER BY w DESC", BOTH, "social"), + ("L5", "MATCH (n) RETURN n.w AS w ORDER BY w DESC NULLS LAST", GQL, "social"), + ("L6", "MATCH (n) RETURN n.w AS w ORDER BY w ASC NULLS FIRST", GQL, "social"), + ("L7", "MATCH (n) RETURN n.name AS n2, n.w AS w ORDER BY w DESC NULLS LAST, n2 LIMIT 3", GQL, "social"), + ("L8", "MATCH (a:Person) RETURN a.name AS n ORDER BY a.age DESC", BOTH, "social"), + ("L9", "MATCH (a:Person)-[r]->(b) RETURN type(r) AS t, r.w AS w ORDER BY t DESC, w", BOTH, "social"), + ("L10", "UNWIND [1, 0.0 / 0.0, -1, 1.0 / 0.0] AS x RETURN x ORDER BY x", BOTH, "social"), + ("L11", "MATCH (n) RETURN labels(n) AS l, n.name AS n2 ORDER BY l, n2", BOTH, "social"), + ("L12", "MATCH (n) WITH n ORDER BY n.w DESC LIMIT 3 RETURN n.name AS n2", CYPHER, "social"), + ("L13", "MATCH (n) RETURN n.w AS w ORDER BY w DESC LIMIT 2", BOTH, "social"), + ("L14", "MATCH (n:N) RETURN n.i AS i ORDER BY i DESC LIMIT 3", BOTH, "chain"), + ("L15", "MATCH (n:N) RETURN n.i AS i, n.m AS m ORDER BY m DESC, i LIMIT 5", BOTH, "chain"), + ("L16", "MATCH (n:N) WHERE n.i < 10 RETURN percentileCont(n.i, 0.5) AS p, percentileDisc(n.i, 0.5) AS d", BOTH, "chain"), + ("L17", "MATCH ()-[r:NEXT]->() RETURN r.k AS k ORDER BY k DESC SKIP 10 LIMIT 3", BOTH, "chain"), + ("L18", "UNWIND range(1, 3000) AS i RETURN CASE WHEN i % 3 = 0 THEN toString(i) ELSE i END AS v ORDER BY v LIMIT 3", BOTH, "social"), + ("L19", "MATCH (n) RETURN n.name AS n2 ORDER BY n.w DESC, n2 LIMIT 4", BOTH, "social"), + ("L20", "MATCH (a:Person) OPTIONAL MATCH (a)-[:LIVES_IN]->(c) RETURN a.name AS n, c.name AS c ORDER BY c DESC, n", BOTH, "social"), +]: + ordered(case_id, query, languages, fixture) + +# M: EXISTS and COUNT subqueries: nodes and edges shared with the row, nulls, paths, a +# second pattern and path modes +for case_id, query, languages, fixture in [ + ("M1", "MATCH (a:Person)-[:KNOWS]->(b:Person) WHERE EXISTS { MATCH (b)-[:KNOWS]->(a) } RETURN a.name AS a, b.name AS b", BOTH, "social"), + ("M2", "MATCH (a:Person)-[:KNOWS]->(b:Person) WHERE NOT EXISTS { MATCH (b)-[:KNOWS]->(a) } RETURN a.name AS a, b.name AS b", BOTH, "social"), + ("M3", "MATCH (a:Person)-[:KNOWS]->(b:Person) WHERE EXISTS { MATCH (b)-[:KNOWS]->(a) } OR a.name = 'Jules' RETURN a.name AS a, b.name AS b", BOTH, "social"), + ("M4", "MATCH (a:Person)-[:KNOWS]->(b)-[:KNOWS]->(c) WHERE EXISTS { MATCH (c)-[:KNOWS]->(a) } RETURN a.name AS a, b.name AS b, c.name AS c", BOTH, "social"), + ("M5", "MATCH (a:Person)-[:KNOWS]->(b)-[:KNOWS]->(c) RETURN a.name AS a, c.name AS c, EXISTS { MATCH (c)-[:KNOWS]->(a) } AS closes, COUNT { MATCH (c)-[:KNOWS]->(a) } AS n", BOTH, "social"), + ("M6", "MATCH (a)-[r]->(b) WHERE EXISTS { MATCH (x)-[r]->(:City) } RETURN a.name AS a, b.name AS b", BOTH, "social"), + ("M7", "MATCH (a)-[r]->(b) RETURN a.name AS a, b.name AS b, EXISTS { MATCH (x)-[r]->(:City) } AS home, COUNT { MATCH ()-[r]->() } AS one", BOTH, "social"), + ("M8", "MATCH (c:City) WHERE EXISTS { MATCH (x)-[:WORKS_AT]->() } RETURN c.name AS c", BOTH, "social"), + ("M9", "MATCH (c:City) WHERE NOT EXISTS { MATCH (x)-[:WORKS_AT]->() } RETURN c.name AS c", BOTH, "social"), + ("M10", "MATCH (c:City) RETURN c.name AS c, COUNT { MATCH (x)-[:KNOWS]->(y) } AS knows, EXISTS { MATCH (x)-[:LIVES_IN]->(y) } AS lives", BOTH, "social"), + ("M11", "MATCH (c:City) RETURN c.name AS c, EXISTS { MATCH (p)-[:LIVES_IN]->(c) } AS e, COUNT { MATCH (p)-[:LIVES_IN]->(c) } AS n", BOTH, "social"), + ("M12", "MATCH (c:City) WHERE EXISTS { MATCH (p)-[:LIVES_IN]->(c) } RETURN c.name AS c", BOTH, "social"), + ("M13", "MATCH (a:Person) OPTIONAL MATCH (a)-[:LIVES_IN]->(c) RETURN a.name AS a, EXISTS { MATCH (c)<-[:LIVES_IN]-() } AS e, COUNT { MATCH (c)<-[:LIVES_IN]-() } AS n", BOTH, "social"), + ("M14", "MATCH (a:Person) OPTIONAL MATCH (a)-[:LIVES_IN]->(c) WITH a, c WHERE NOT EXISTS { MATCH (c)<-[:LIVES_IN]-() } RETURN a.name AS a", BOTH, "social"), + ("M15", "MATCH (a:Person) OPTIONAL MATCH (a)-[:LIVES_IN]->(c) WITH a, c WHERE EXISTS { MATCH (c)<-[:LIVES_IN]-() } RETURN a.name AS a", BOTH, "social"), + ("M16", "MATCH (a:Person {name: 'Alix'}), (b:Person) RETURN b.name AS b, EXISTS { MATCH (a)-[:KNOWS*1..2]->(b) } AS near", BOTH, "social"), + ("M17", "MATCH (a:Person {name: 'Alix'}), (b:Person) WHERE EXISTS { MATCH (a)-[:KNOWS*1..2]->(b) } RETURN b.name AS b", BOTH, "social"), + ("M18", "MATCH (a:Person {name: 'Alix'}), (b:Person) RETURN b.name AS b, EXISTS { MATCH (a)-[:KNOWS*]->(b) } AS reach", BOTH, "social"), + ("M19", "MATCH (a:Person) WITH a, a AS b RETURN a.name AS a, EXISTS { MATCH (a)-[:KNOWS*1..2]-(b) } AS back, EXISTS { MATCH (a)-[:KNOWS*1..3]->(b) } AS closed", BOTH, "social"), + ("M20", "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS*1..3]->(b) RETURN b.name AS b, size(rs) AS hops, EXISTS { MATCH (x)<-[rs:KNOWS*]-(y) } AS rev, EXISTS { MATCH (x)-[rs:KNOWS*1..2]->(y) } AS short, EXISTS { MATCH (b)-[rs:KNOWS*]->(y) } AS from_b", BOTH, "social"), + ("M21", "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b), (c:Robot) } RETURN a.name AS a", BOTH, "social"), + ("M22", "MATCH (a:Person) RETURN a.name AS a, EXISTS { MATCH (a)-[:KNOWS]->(b), (c:Robot) } AS e", BOTH, "social"), + ("M23", "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b) MATCH (b:City) } RETURN a.name AS a", CYPHER, "social"), + ("M24", "MATCH (a:Person) RETURN a.name AS a, COUNT { MATCH (a)-[r]->(b) MATCH (b:City) } AS cities", CYPHER, "social"), + ("M25", "MATCH (a:Person) WHERE EXISTS { MATCH ACYCLIC (a)-[:KNOWS*1..3]->(a) } RETURN a.name AS a", GQL, "social"), + ("M26", "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS*1..3]->(a) } RETURN a.name AS a", GQL, "social"), + ("M27", "MATCH (a:Person) RETURN a.name AS a, EXISTS { MATCH TRAIL (a)-[:KNOWS*1..2]->(x) } AS e", GQL, "social"), + ("M28", "MATCH (a:Person) RETURN a.name AS a, CASE WHEN EXISTS { MATCH (a)-[:LIVES_IN]->() } THEN 'housed' ELSE 'not' END AS h", BOTH, "social"), + ("M29", "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:LIVES_IN]->(:City) } OR a.age > 35 RETURN a.name AS a", BOTH, "social"), + ("M30", "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:LIVES_IN]->() } AND EXISTS { MATCH (a)-[:KNOWS]->()-[:KNOWS]->() } RETURN a.name AS a", BOTH, "social"), + ("M31", "MATCH (a:Person), (b:Person) WHERE (a)-[:KNOWS]->(b) RETURN a.name AS a, b.name AS b", CYPHER, "social"), + ("M32", "MATCH (a:Person), (b:Person) WHERE NOT (a)-[:KNOWS]->(b) AND a <> b RETURN a.name AS a, b.name AS b", CYPHER, "social"), + ("M33", "MATCH (a:Person), (c:City) WHERE EXISTS { MATCH (a)-[:LIVES_IN]->(c:City) } RETURN a.name AS a, c.name AS c", BOTH, "social"), + ("M34", "MATCH (a:Person) RETURN a.name AS a, COUNT { MATCH (a)-[:KNOWS]->(f) } AS out, COUNT { MATCH (a)<-[:KNOWS]-(f) } AS inn, COUNT { MATCH (a)-[:KNOWS]-(f) } AS both", BOTH, "social"), + ("M35", "MATCH (a:Person) WHERE COUNT { MATCH (a)-[:KNOWS]->(f) } = 2 RETURN a.name AS a", BOTH, "social"), + ("M36", "MATCH (a:Person)-[:KNOWS]->(b) WITH a, b WHERE COUNT { MATCH (b)-[:KNOWS]->(a) } = 0 RETURN a.name AS a, b.name AS b", BOTH, "social"), + ("M37", "MATCH (a:N)-[:NEXT]->(b:N) WHERE EXISTS { MATCH (b)-[:NEXT]->(a) } RETURN count(*) AS n", BOTH, "chain"), + ("M38", "MATCH (a:N)-[:NEXT]->(b:N) WHERE NOT EXISTS { MATCH (b)-[:NEXT]->(a) } RETURN count(*) AS n", BOTH, "chain"), + ("M39", "MATCH (n:N) WHERE n.i < 4 RETURN n.i AS i, EXISTS { MATCH (n)-[:NEXT]->(m) } AS out, COUNT { MATCH (m)-[:NEXT]->(n) } AS inn ORDER BY i", BOTH, "chain"), + ("M40", "MATCH (a:N {i: 10})-[:NEXT]->(b) RETURN b.i AS i, EXISTS { MATCH (a)-[:NEXT]->(b) } AS e, COUNT { MATCH (b)<-[:NEXT]-(a) } AS c", BOTH, "chain"), +]: + case(case_id, query, languages, fixture) + +# N: nodes and edges through joins, EXISTS, CALL, UNWIND and group keys +for case_id, query in [ + ("N1", "MATCH (a:Person) OPTIONAL MATCH (a)-[:LIVES_IN]->(c) RETURN a.name AS a, a.w AS w, c.w AS cw"), + ("N2", "MATCH (a:Person)-[:KNOWS]->(b), (b)-[r:LIVES_IN]->(c) RETURN a.name AS a, b.name AS b, b.w AS bw, r.w AS rw"), + ("N3", "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(m)-[:LIVES_IN]->() } RETURN a.name AS a, a.w AS w"), + ("N4", "MATCH (a:Person) WHERE NOT EXISTS { MATCH (a)-[:KNOWS]->()-[:KNOWS]->() } RETURN a.name AS a, a.w AS w"), + ("N5", "MATCH (a:Person) CALL { WITH a RETURN 1 AS one } RETURN a.name AS a, a.w AS w"), + ("N6", "MATCH ()-[r:KNOWS]->() CALL { RETURN 1 AS one } RETURN r.w AS w"), + ("N7", "MATCH (a:Person) CALL { WITH a MATCH (a)-[r:KNOWS]->(b) RETURN r, b } RETURN a.name AS a, r.w AS rw, b.w AS bw"), + ("N8", "MATCH (a:Person) UNWIND [1, 2] AS k RETURN a.name AS a, a.w AS w, k"), + ("N9", "MATCH (a:Person)-[r]->() WITH a, count(r) AS n RETURN a.name AS a, a.w AS w, n"), + ("N10", "MATCH ()-[r:KNOWS]->() WITH r, count(*) AS n RETURN r.w AS w, n"), + ("N11", "MATCH (c:City) OPTIONAL MATCH (p:Person)-[:LIVES_IN]->(c) RETURN c.name AS c, c.w AS w, p.name AS p"), +]: + case(case_id, query) + +# O: lists of nodes and edges (collect, path functions) and UNWIND of lists computed per row +for case_id, query in [ + ("O1", "MATCH (a:Person) WITH collect(a) AS people UNWIND people AS p RETURN p.name AS n, p.w AS w"), + ("O2", "MATCH ()-[r:KNOWS]->() WITH collect(r) AS rs UNWIND rs AS e RETURN e.w AS w"), + ("O3", "MATCH ()-[r:LIVES_IN]->() WITH collect(r) AS rs UNWIND rs AS e RETURN type(e) AS t, e.w AS w"), + ("O4", "MATCH ()-[r:KNOWS]->() WITH collect(r) AS rs RETURN size([x IN rs WHERE x.w < 10]) AS n"), + ("O5", "MATCH p = (:Person {name: 'Alix'})-[:KNOWS*2]->() UNWIND relationships(p) AS e RETURN e.w AS w"), + ("O6", "MATCH p = (:Person {name: 'Jules'})-[:KNOWS]->() UNWIND nodes(p) AS n RETURN n.name AS n, n.w AS w"), + ("O7", "MATCH (a:Person) UNWIND range(1, a.w - 99) AS i RETURN a.name AS a, i"), + ("O8", "MATCH ()-[r:LIVES_IN {w: 6}]->() RETURN collect(r) AS rs"), + ("O9", "MATCH (a:Person)-[r:KNOWS]->() WITH a, count(r) AS n RETURN a, n"), + ("O10", "MATCH ()-[r:LIVES_IN]->() WITH collect(r) AS rs UNWIND rs AS e MATCH (x)-[e]->(y) RETURN x.name AS x, y.name AS y"), + ("O11", "MATCH ()-[r:KNOWS {w: 5}]->() WITH collect(r) AS rs RETURN rs[0].w AS w, rs[-1].since AS since"), +]: + case(case_id, query) + +# P: keys() of nodes and edges, and a variable that is a node, an edge or a value +for case_id, query, languages in [ + ("P1", "MATCH ()-[r:KNOWS {w: 1}]->() RETURN size(keys(r)) AS n, 'since' IN keys(r) AS s", BOTH), + ("P2", "MATCH ()-[r]->() WHERE 'since' IN keys(r) RETURN count(*) AS n", BOTH), + ("P3", "MATCH (r) MATCH ()-[r]->() RETURN count(*) AS c", BOTH), + ("P4", "MATCH (a)-[a]->(b) RETURN count(*) AS c", BOTH), + ("P5", "MATCH (a)-[r]->(r) RETURN count(*) AS c", BOTH), + ("P6", "WITH 1 AS r MATCH ()-[r]->() RETURN count(*) AS c", CYPHER), + ("P7", "MATCH (a:Person) CALL { WITH a MATCH ()-[a]->(b) RETURN b } RETURN count(*) AS c", BOTH), + ("P8", "MATCH ()-[r:KNOWS]->() CALL { WITH r MATCH (x)-[r]->(y) RETURN x.name AS xn } RETURN count(*) AS c", BOTH), +]: + case(case_id, query, languages) + +# Q: EXISTS and COUNT subqueries tied to the outer row by a value, or that one edge does not decide +for case_id, query in [ + ("Q1", "MATCH (a:Person) RETURN a.name AS n, EXISTS { MATCH (c:City) WHERE c.w = a.w + 4 } AS e"), + ("Q2", "MATCH (a:Person) RETURN a.name AS n, COUNT { MATCH (b:Person) WHERE b.age < a.age } AS younger"), + ("Q3", "MATCH (a:Person) WHERE COUNT { MATCH (b:Person) WHERE b.age < a.age } = 2 RETURN a.name AS n"), + ("Q4", "UNWIND [25, 30] AS k RETURN k, COUNT { MATCH (p:Person {age: k}) } AS c"), + ("Q5", "MATCH (a:Person) RETURN a.name AS n, COUNT { MATCH (a)-[:KNOWS*1..2]->() } AS c"), + ("Q6", "MATCH (a:Person) WHERE COUNT { MATCH (a)-[:KNOWS]->(b), (c:Robot) } = 0 RETURN a.name AS n"), + ("Q7", "MATCH (a:Person)-[:KNOWS]->(b) WITH a WHERE COUNT { MATCH (a)-[:KNOWS]->(x), (y:City) } = 6 RETURN a.name AS n"), + ("Q8", "MATCH (a:Person) WITH a, COUNT { MATCH (b:Person) WHERE b.age < a.age } AS y RETURN a.name AS n, y"), +]: + case(case_id, query) + +# R: EXISTS in WHERE tied to the outer row by a value, inside OR, and through the row's edge; +# R6: a CALL subquery over two edges in a row, which runs again for each outer row like those +for case_id, query in [ + ("R1", "MATCH (a:Person) WHERE EXISTS { MATCH (c:City) WHERE c.w = a.w + 4 } RETURN a.name AS n"), + ("R2", "MATCH (a:Person) WHERE NOT EXISTS { MATCH (c:City) WHERE c.w = a.w + 4 } RETURN a.name AS n"), + ("R3", "MATCH (a:Person)-[:KNOWS]->(b) WITH a WHERE a.name = 'Jules' OR EXISTS { MATCH (a)-[:KNOWS]->(x), (c:City) WHERE c.w = a.w + 4 } RETURN a.name AS n"), + ("R4", "MATCH (a:Person)-[r:KNOWS]->(b) WHERE EXISTS { MATCH (a)-[s:KNOWS]->(c) WHERE s <> r } RETURN a.name AS a, b.name AS b"), + ("R5", "UNWIND [25, 31] AS k WITH k WHERE EXISTS { MATCH (p:Person {age: k}) } RETURN k"), +]: + case(case_id, query) +case("R6", "MATCH (a:Person) CALL { WITH a MATCH (a)-[:KNOWS]->(m)-[:KNOWS]->(x) RETURN x.name AS x } RETURN a.name AS a, x", CYPHER) + +# S: GQL VALUE subqueries returning count(x) skip nulls and count DISTINCT values once +for case_id, query in [ + ("S1", "MATCH (a:Person) RETURN a.name AS n, VALUE { MATCH (a)-[:KNOWS]->(b) RETURN count(b.w) } AS c"), + ("S2", "MATCH (a:Person) RETURN a.name AS n, VALUE { MATCH (a)-[:KNOWS]->(b), (c:City) RETURN count(DISTINCT b) } AS c"), + ("S3", "MATCH (a:Person) WHERE VALUE { MATCH (a)-[:KNOWS]->(b) RETURN count(b.w) } = 1 RETURN a.name AS n"), +]: + case(case_id, query, GQL) + +# T: the MATCH clauses of a subquery go on from each other +for case_id, query, languages in [ + ("T1", "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b) MATCH (b)-[:LIVES_IN]->(c:City) } RETURN a.name AS n", BOTH), + ("T2", "MATCH (a:Person) RETURN a.name AS n, COUNT { MATCH (a)-[:KNOWS]->(b) MATCH (b:City) } AS c", BOTH), + ("T3", "MATCH (a:Person) RETURN a.name AS n, COUNT { MATCH (a)-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:LIVES_IN]->(c) } AS c", GQL), +]: + case(case_id, query, languages) + +# U: a Cypher importing WITH only lists outer variables, as in Neo4j +for case_id, query in [ + ("U1", "MATCH (a:Person) CALL { WITH a WHERE a.age > 30 RETURN a.name AS m } RETURN m"), + ("U2", "MATCH (a:Person) CALL { WITH a AS b RETURN b.name AS m } RETURN m"), + ("U3", "MATCH (a:Person) CALL { WITH a WITH a WHERE a.age > 30 RETURN a.name AS m } RETURN m"), + ("U4", "MATCH (a:Person) CALL { WITH a WITH a AS b RETURN b.name AS m } RETURN a.name AS a, m"), +]: + case(case_id, query, CYPHER) + +# V: the nodes and edges a CALL subquery returns are the nodes and edges themselves +V_NODE = "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b }" +V_EDGE = "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[r:KNOWS]->() RETURN r }" +for case_id, query in [ + ("V1", f"{V_NODE} MATCH (b)-[:KNOWS]->(y) RETURN b.name AS b, y.name AS y"), + ("V2", f"{V_NODE} MATCH (x:Person)-[:KNOWS]->(b) RETURN b.name AS b, x.name AS x"), + ("V3", f"{V_NODE} MATCH (y:Person) WHERE y = b RETURN y.name AS y"), + ("V4", "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b, id(b) AS inner } RETURN b.name AS b, id(b) = inner AS same"), + ("V5", f"{V_EDGE} RETURN type(r) AS t, r.w AS w"), + ("V6", f"{V_EDGE} MATCH (x)-[r]->(z) RETURN x.name AS x, z.name AS z"), + ("V7", "CALL { MATCH (c:City) RETURN c } MATCH (p:Person)-[:LIVES_IN]->(c) RETURN p.name AS p, c.name AS c"), +]: + case(case_id, query) +case("V8", V_NODE, CYPHER) + +# W: a GQL CALL subquery sees the outer row's variables, or those its scope clause names +for case_id, query in [ + ("W1", "MATCH (a:Person {name: 'Alix'}) CALL { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN bn"), + ("W2", "MATCH ()-[r:KNOWS]->() CALL { MATCH (x)-[r]->(y) RETURN x.name AS xn } RETURN count(*) AS c"), + ("W3", "MATCH (a:Person) CALL { WITH a WHERE a.age > 30 RETURN a.name AS m } RETURN m"), + ("W4", "MATCH (a:Person {name: 'Alix'}) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN bn"), + ("W5", "MATCH (a:Person {name: 'Alix'}) CALL () { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN count(*) AS c"), + ("W6", "MATCH (a:Person {name: 'Alix'}), (c:City {name: 'Paris'}) CALL (a) { MATCH (c:City) RETURN count(c) AS n } RETURN n"), + ("W7", "MATCH (a:Person) OPTIONAL CALL (a) { MATCH (a)-[:LIVES_IN]->(c) RETURN c.name AS c } RETURN a.name AS a, c"), +]: + case(case_id, query, GQL) + +# X: a Cypher CALL subquery with a variable scope clause +for case_id, query in [ + ("X1", "MATCH (a:Person {name: 'Alix'}) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN bn"), + ("X2", "MATCH (a:Person {name: 'Alix'}) CALL (*) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN bn"), + ("X3", "MATCH (a:Person {name: 'Alix'}) CALL () { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN count(*) AS c"), + ("X4", "MATCH (a:Person {name: 'Alix'}), (c:City {name: 'Paris'}) CALL (a) { MATCH (c:City) RETURN count(c) AS n } RETURN n"), + ("X5", "MATCH (a:Person) CALL (a) { WITH a WHERE a.age > 30 RETURN a.name AS m } RETURN m"), +]: + case(case_id, query, CYPHER) + +# Y: RETURN * in a CALL subquery returns the variables the subquery binds itself +Y_STAR = "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN * }" +for case_id, query, languages in [ + ("Y1", f"{Y_STAR} RETURN a.name AS a, b.name AS b", BOTH), + ("Y2", f"{Y_STAR} MATCH (x:Person)-[:KNOWS]->(b) RETURN b.name AS b, x.name AS x", BOTH), + ("Y3", "CALL { MATCH (c:City) RETURN * } RETURN c.name AS c", BOTH), + ("Y4", "MATCH (a:Person {name: 'Alix'}) CALL { MATCH (a)-[r:KNOWS]->(b) RETURN * } RETURN type(r) AS t, b.name AS b", GQL), +]: + case(case_id, query, languages) + +# Z: a CALL subquery's RETURN under ORDER BY and LIMIT passes on its nodes too +for case_id, query in [ + ("Z1", "MATCH (a:Person {name: 'Alix'}) CALL { MATCH (a)-[:KNOWS]->(b) RETURN b ORDER BY b.name DESC LIMIT 1 } MATCH (b)-[:KNOWS]->(y) RETURN b.name AS b, y.name AS y"), + ("Z2", "MATCH (a:Person {name: 'Alix'}) CALL { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS n ORDER BY b.age LIMIT 1 } RETURN n"), +]: + case(case_id, query, GQL) + +# AA: one-hop quantifiers bind a list, collected list items stay edges, pattern comprehensions in aggregates +for case_id, query, languages in [ + ("AA1", "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS*1..1]->(b) RETURN b.name AS b, size(rs) AS n", CYPHER), + ("AA2", "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS]->{1,1}(b) RETURN b.name AS b, size(rs) AS n", GQL), + ("AA3", "MATCH p = (a:Person {name: 'Alix'})-[:KNOWS]->(b) WITH collect(last(relationships(p))) AS es UNWIND es AS e RETURN e.w AS w", BOTH), + ("AA4", "MATCH (a:Person) RETURN sum(size([(a)-[:KNOWS]->(b) | b])) AS n", CYPHER), + ("AA5", "MATCH (a:Person) RETURN a.name AS a, size([(a)-[:KNOWS]->(b) | b]) AS n", CYPHER), + ("AA6", "MATCH (a:Person) RETURN a.name AS a, size(a{.name, knows: [(a)-[:KNOWS]->(b) | b]}.knows) AS n", CYPHER), +]: + case(case_id, query, languages) + +# AB: a CALL subquery returns new names only; a WITH ends the scope of what it leaves out +for case_id, query in [ + ("AB1", "MATCH (a:Person {name: 'Alix'}) CALL { RETURN 1 AS a } RETURN a"), + ("AB2", "MATCH (a:Person {name: 'Alix'}) CALL (a) { RETURN a } RETURN a.name AS n"), + ("AB3", "MATCH (a:Person {name: 'Alix'}) WITH 1 AS x RETURN a.name AS n"), + ("AB4", "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) WITH a, count(b) AS k RETURN a.name AS n, k } RETURN n, k"), + ("AB5", "MATCH (a:Person {name: 'Alix'}) WITH a.name RETURN a.name"), +]: + case(case_id, query, BOTH) + +# AC: subquery bodies: ORDER BY, SKIP, LIMIT and UNION in CALL; OPTIONAL MATCH in and before subqueries +for case_id, query, languages in [ + ("AC1", "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS f ORDER BY b.age DESC LIMIT 1 } RETURN a.name AS a, f", BOTH), + ("AC2", "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS f ORDER BY f SKIP 1 } RETURN a.name AS a, f", BOTH), + ("AC3", "MATCH (a:Person {name: 'Alix'}) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS x UNION MATCH (b)-[:KNOWS]->(a) RETURN b.name AS x } RETURN x", BOTH), + ("AC4", "MATCH (a:Person {name: 'Alix'}) CALL (a) { RETURN a.name AS x UNION ALL RETURN a.name AS x } RETURN x", BOTH), + ("AC5", "MATCH (a:Person) CALL { WITH a ORDER BY a.age MATCH (a)-[:KNOWS]->(b) RETURN b.name AS f } RETURN f", CYPHER), + ("AC6", "RETURN 1 AS x UNION ALL RETURN 1 AS x UNION RETURN 1 AS x", CYPHER), + ("AC7", "MATCH (a:Person) RETURN a.name AS a, COUNT { MATCH (a)-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:LIVES_IN]->(c) } AS n", BOTH), + ("AC8", "OPTIONAL MATCH (x:Robot) RETURN x.name AS x", BOTH), + ("AC9", "MATCH (p:Person) RETURN p.name AS p, VALUE { OPTIONAL MATCH (p)-[:KNOWS]->(f) RETURN count(f) } AS friends", GQL), + ("AC10", "MATCH (p:Person) RETURN p.name AS p, COUNT { OPTIONAL MATCH (p)-[:KNOWS]->(f) } AS n", BOTH), + ("AC11", "MATCH (p:Person) WHERE EXISTS { OPTIONAL MATCH (p)-[:LIVES_IN]->(c) } RETURN p.name AS p", BOTH), + ("AC12", "MATCH (a:Person {name: 'Alix'})-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c) WHERE c.age > a.age RETURN b.name AS b, c.name AS c", BOTH), + ("AC13", "MATCH (a:Person {name: 'Alix'})-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c WHERE c.age > a.age) RETURN b.name AS b, c.name AS c", GQL), +]: + case(case_id, query, languages) + +# AD: a subquery that imports nothing sees no outer variable +for case_id, query, languages in [ + ("AD1", "MATCH (a:Person) CALL () { MATCH (b:Person) WHERE b.age > a.age RETURN count(b) AS c } RETURN a.name AS a, c", BOTH), + ("AD2", "MATCH (a:Person) CALL { MATCH (b:Person) WHERE b.age > a.age RETURN count(b) AS c } RETURN a.name AS a, c", CYPHER), + ("AD3", "UNWIND [1] AS a CALL () { MATCH (a) RETURN a AS x } RETURN count(x) AS n", BOTH), +]: + case(case_id, query, languages) + +# AE: a later pattern back to a variable bound before a shortest path, and subqueries that share nothing +for case_id, query, languages in [ + ("AE1", "MATCH p = shortestPath((a:Person {name: 'Gus'})-[:KNOWS*]->(b:Person {name: 'Alix'})) MATCH (b)-[:KNOWS]->(c)-[:KNOWS]->(a) RETURN c.name AS c", CYPHER), + ("AE2", "MATCH p = ANY SHORTEST (a:Person {name: 'Gus'})-[:KNOWS]->+(b:Person {name: 'Alix'}) MATCH (b)-[:KNOWS]->(c)-[:KNOWS]->(a) RETURN c.name AS c", GQL), + ("AE3", "MATCH (c:City) RETURN c.name AS c, COUNT { MATCH (x)-[:KNOWS]->(y) } AS n", BOTH), + ("AE4", "MATCH (c:City) RETURN c.name AS c, EXISTS { MATCH (x)-[:LIVES_IN]->(y:City {name: 'Paris'}) } AS e", BOTH), +]: + case(case_id, query, languages) + +# AF: GQL NEXT passes rows on; a VALUE subquery reads the outer row (read only: the fixtures are shared) +for case_id, query in [ + ("AF1", "MATCH (a:Person {name: 'Alix'}) RETURN a NEXT MATCH (a)-[:KNOWS]->(b) RETURN b.name AS b"), + ("AF2", "MATCH (a:Person {name: 'Alix'}) RETURN a.age AS x NEXT RETURN x + 1 AS y"), + ("AF3", "MATCH (p:Person) RETURN p.name AS p, VALUE { MATCH (p)-[:KNOWS]->(f) RETURN f.name ORDER BY f.name LIMIT 1 } AS first"), + ("AF4", "MATCH (p:Person) WHERE VALUE { MATCH (p)-[:KNOWS]->(f) RETURN f.name ORDER BY f.name LIMIT 1 } IS NOT NULL RETURN p.name AS p"), +]: + case(case_id, query, GQL) + +# fmt: on diff --git a/scripts/difftest/difftest.py b/scripts/difftest/difftest.py new file mode 100644 index 000000000..2d4668a0a --- /dev/null +++ b/scripts/difftest/difftest.py @@ -0,0 +1,413 @@ +"""Differential test: run one query corpus on two builds of Grafeo and compare them. + +Every difference between two builds is an intended change or a regression. Before a +release, run the corpus on the previous release (from PyPI) and on the release branch, +and review every difference. The reviewed differences of a release are listed in +`scripts/difftest/reviewed/.txt`, one per line: the case key, the fingerprint of +the reviewed result and why it changed. The gate then fails on a difference nobody +reviewed, on a reviewed result that changed since, and on a reviewed difference that is +gone (a fix that was undone). The corpus and its fixtures are in `corpus.py`. + +Run from the repository root (needs uv; `candidate` also needs maturin and Rust): + + # The grafeo this interpreter imports, for example a `maturin develop` build + python scripts/difftest/difftest.py run target/difftest/dev.json + + # A published release, installed from PyPI into its own environment + python scripts/difftest/difftest.py baseline 0.5.43 target/difftest/0.5.43.json + + # A commit, built as a release wheel in a clean worktree + python scripts/difftest/difftest.py candidate HEAD target/difftest/head.json + + # The cases whose results differ (with --reviewed: fail on anything not reviewed) + python scripts/difftest/difftest.py compare OLD.json NEW.json --reviewed FILE + + # The cases whose GQL and Cypher results differ within one run + python scripts/difftest/difftest.py parity RESULTS.json + + # The release gate: baseline, candidate and compare + python scripts/difftest/difftest.py gate 0.5.43 --reviewed scripts/difftest/reviewed/0.5.44.txt + +Work files (environments, the worktree, wheels, results) go to `target/difftest/`. +""" + +from __future__ import annotations + +import argparse +import hashlib +import json +import math +import os +import re +import shutil +import subprocess +import sys +from pathlib import Path + +HERE = Path(__file__).resolve().parent +ROOT = HERE.parents[1] +WORK = ROOT / "target" / "difftest" +# The Python wheels are abi3 wheels for Python 3.12 and later. +PYTHON = "3.12" +ERROR_LENGTH = 200 + +if str(HERE) not in sys.path: + sys.path.insert(0, str(HERE)) + + +# --------------------------------------------------------------------------- +# Running the corpus +# --------------------------------------------------------------------------- + + +def normalize(value): + """A JSON form of a returned value that compares equal across builds.""" + if isinstance(value, dict): + items = sorted(value.items(), key=lambda item: str(item[0])) + return {str(key): normalize(item) for key, item in items} + if isinstance(value, (list, tuple)): + return [normalize(item) for item in value] + if isinstance(value, float): + if math.isnan(value): + return {"float": "NaN"} + if math.isinf(value): + return {"float": "Infinity" if value > 0 else "-Infinity"} + return round(value, 9) + if isinstance(value, (str, int, bool)) or value is None: + return value + for attribute in ("to_dict", "as_dict"): + if hasattr(value, attribute): + return { + "type": type(value).__name__, + **normalize(getattr(value, attribute)()), + } + return {"repr": repr(value)} + + +def error_text(error: Exception) -> str: + lines = str(error).splitlines() + return lines[0][:ERROR_LENGTH] if lines else type(error).__name__ + + +def run_case(db, language: str, query: str) -> dict: + """The result of one query: its columns and rows, or the first line of its error.""" + execute = {"gql": db.execute, "cypher": db.execute_cypher}[language] + try: + result = execute(query) + records = [dict(row) for row in result] + except Exception as error: # noqa: BLE001 (a failing query is a result too) + return {"error": error_text(error)} + columns = list(getattr(result, "columns", None) or (records[0] if records else [])) + rows = [[normalize(value) for value in record.values()] for record in records] + return {"columns": columns, "rows": rows} + + +def run_corpus(out: Path) -> None: + # Imported here, so that comparing results works without grafeo installed. + import grafeo + from corpus import CASES, FIXTURES + + databases = {} + results = {} + for case in CASES: + if case.fixture not in databases: + databases[case.fixture] = FIXTURES[case.fixture](grafeo) + for language in case.languages: + result = run_case(databases[case.fixture], language, case.query) + results[f"{case.id}|{language}"] = { + "query": case.query, + "ordered": case.ordered, + **result, + } + build_info = getattr(grafeo, "build_info", None) + meta = { + "version": grafeo.__version__, + "commit": build_info()["commit"] if build_info else None, + } + out.parent.mkdir(parents=True, exist_ok=True) + text = json.dumps({"meta": meta, "results": results}, indent=1, sort_keys=True) + out.write_text(text + "\n", encoding="utf-8") + print(f"ran {len(results)} cases on grafeo {describe(meta)}, results in {out}") + + +# --------------------------------------------------------------------------- +# Comparing results +# --------------------------------------------------------------------------- + + +def describe(meta: dict) -> str: + commit = meta.get("commit") + return ( + f"{meta.get('version')} ({commit[:8]})" if commit else f"{meta.get('version')}" + ) + + +def load(path: Path) -> dict: + return json.loads(path.read_text(encoding="utf-8")) + + +def canonical(result: dict): + """What two results must share to count as equal: rows in order when the case is + ordered, otherwise as a multiset; for a failed query, its error.""" + if "error" in result: + return ["error", result["error"]] + rows = result["rows"] + if not result.get("ordered"): + rows = sorted(rows, key=lambda row: json.dumps(row, sort_keys=True)) + return ["rows", result.get("columns"), rows] + + +def canonical_text(result: dict) -> str: + """The canonical form as JSON text: unlike Python values, it tells 1, 1.0 + and true apart, so a value that changes type counts as a difference.""" + return json.dumps(canonical(result), sort_keys=True) + + +def fingerprint(result: dict) -> str: + return hashlib.sha256(canonical_text(result).encode("utf-8")).hexdigest()[:10] + + +def case_order(key: str): + case_id, _, language = key.partition("|") + match = re.fullmatch(r"([A-Za-z]+)(\d+)", case_id) + if match: + return (match.group(1), int(match.group(2)), language) + return (case_id, 0, language) + + +def differing(old: dict, new: dict) -> list[str]: + """The case keys of both runs whose results differ, in corpus order.""" + common = old.keys() & new.keys() + keys = [ + key for key in common if canonical_text(old[key]) != canonical_text(new[key]) + ] + return sorted(keys, key=case_order) + + +def read_reviewed(path: Path) -> dict[str, tuple[str, str]]: + """The reviewed differences: case key -> (fingerprint, reason).""" + reviewed = {} + lines = path.read_text(encoding="utf-8").splitlines() + for number, line in enumerate(lines, 1): + line = line.strip() + if not line or line.startswith("#"): + continue + parts = line.split(maxsplit=2) + if len(parts) < 3 or "|" not in parts[0]: + raise SystemExit( + f"{path}:{number}: expected 'ID|language fingerprint reason'" + ) + key, digest, reason = parts + if key in reviewed: + raise SystemExit(f"{path}:{number}: {key} is listed twice") + reviewed[key] = (digest, reason) + return reviewed + + +def summary(result: dict, limit: int = 6) -> str: + if "error" in result: + return "ERROR " + result["error"] + rows = result["rows"] + shown = json.dumps(rows[:limit], sort_keys=True) + more = f" ... ({len(rows)} rows)" if len(rows) > limit else f" ({len(rows)} rows)" + return f"columns {result.get('columns')} {shown}{more}" + + +def compare(old_path: Path, new_path: Path, reviewed_path: Path | None = None) -> int: + old, new = load(old_path), load(new_path) + old_results, new_results = old["results"], new["results"] + print(f"old: grafeo {describe(old['meta'])}, new: grafeo {describe(new['meta'])}") + reviewed = read_reviewed(reviewed_path) if reviewed_path else {} + keys = differing(old_results, new_results) + failures = 0 + for key in keys: + digest = fingerprint(new_results[key]) + if not reviewed_path: + state = "changed" + elif key not in reviewed: + state = "NOT REVIEWED" + elif reviewed[key][0] != digest: + state = "CHANGED SINCE THE REVIEW" + else: + state = "reviewed" + failures += state not in ("changed", "reviewed") + print(f"=== {key} [{state}]: {new_results[key]['query']}") + print(" old:", summary(old_results[key])) + print(" new:", summary(new_results[key])) + if key in reviewed: + print(" why:", reviewed[key][1]) + elif reviewed_path: + print(f" to review: {key} {digest} ") + only = sorted(old_results.keys() ^ new_results.keys(), key=case_order) + if only: + print(f"{len(only)} cases ran in one build only: {', '.join(only)}") + print( + f"{len(keys)} differences in {len(old_results.keys() & new_results.keys())} cases" + ) + if not reviewed_path: + return 0 + gone = sorted(set(reviewed) - set(keys), key=case_order) + for key in gone: + print(f"=== {key} [NO LONGER DIFFERENT]: reviewed as '{reviewed[key][1]}'") + failures += len(gone) + print(f"{failures} of them need attention" if failures else "all reviewed") + return 1 if failures else 0 + + +def parity(path: Path) -> int: + """The cases whose GQL and Cypher runs disagree: one fails and the other does not, + or both succeed with different rows. Both failing counts as agreeing (the error + texts differ by language).""" + results = load(path)["results"] + mismatches = [] + for key in sorted(results, key=case_order): + case_id, _, language = key.partition("|") + other = f"{case_id}|cypher" + if language != "gql" or other not in results: + continue + gql, cypher = results[key], results[other] + if "error" in gql and "error" in cypher: + continue + # Rows only: the column names may differ by language. + gql_rows = json.dumps(canonical(gql)[2:], sort_keys=True) + cypher_rows = json.dumps(canonical(cypher)[2:], sort_keys=True) + if gql_rows != cypher_rows: + mismatches.append(case_id) + print(f"=== {case_id}: {gql['query']}") + print(" gql: ", summary(gql)) + print(" cypher:", summary(cypher)) + print(f"{len(mismatches)} cases where GQL and Cypher disagree") + return 1 if mismatches else 0 + + +# --------------------------------------------------------------------------- +# Builds to compare +# --------------------------------------------------------------------------- + + +def tool(*command: str, cwd: Path = ROOT, env: dict | None = None) -> str: + done = subprocess.run( + command, cwd=cwd, env=env, check=True, capture_output=True, text=True + ) + return done.stdout.strip() + + +def venv_python(venv: Path) -> Path: + windows = venv / "Scripts" / "python.exe" + return windows if windows.exists() else venv / "bin" / "python" + + +def environment(name: str, requirement: str, reinstall: bool) -> Path: + """A uv environment in target/difftest with `requirement` installed.""" + venv = WORK / f"venv-{name}" + if not venv_python(venv).exists(): + tool("uv", "venv", str(venv), "--python", PYTHON, "--quiet") + python = venv_python(venv) + command = ["uv", "pip", "install", "--quiet", "--python", str(python), requirement] + tool(*command, *(["--reinstall"] if reinstall else [])) + return python + + +def run_in(python: Path, out: Path) -> None: + subprocess.run( + [str(python), str(Path(__file__).resolve()), "run", str(out)], check=True + ) + + +def baseline(version: str, out: Path) -> None: + run_in(environment(version, f"grafeo=={version}", reinstall=False), out) + + +def build_wheel(ref: str) -> Path: + """A release wheel of `ref`, built in a clean worktree of this repository (the main + tree may hold a module from `maturin develop` that the wheel build trips over).""" + commit = tool("git", "rev-parse", "--verify", f"{ref}^{{commit}}") + worktree = WORK / "worktree" + if (worktree / ".git").exists(): + tool("git", "-C", str(worktree), "checkout", "--detach", "--force", commit) + else: + tool("git", "worktree", "add", "--detach", str(worktree), commit) + wheels = WORK / "wheels" + shutil.rmtree(wheels, ignore_errors=True) + maturin = shutil.which("maturin") + command = [maturin] if maturin else ["uvx", "maturin"] + env = {**os.environ, "CARGO_TARGET_DIR": str(WORK / "cargo")} + print(f"building a release wheel of {ref} ({commit[:8]})") + subprocess.run( + [*command, "build", "--release", "--out", str(wheels)], + cwd=worktree / "crates" / "bindings" / "python", + env=env, + check=True, + ) + built = list(wheels.glob("*.whl")) + if len(built) != 1: + raise SystemExit(f"expected one wheel in {wheels}, found {len(built)}") + return built[0] + + +def candidate(ref: str, out: Path) -> None: + run_in(environment("candidate", str(build_wheel(ref)), reinstall=True), out) + + +def gate(version: str, ref: str, reviewed: Path | None) -> int: + old = WORK / f"results-{version}.json" + new = WORK / "results-candidate.json" + baseline(version, old) + candidate(ref, new) + return compare(old, new, reviewed) + + +def main(argv: list[str] | None = None) -> int: + # Line by line, so that this script's lines stay in order with its subprocesses'. + sys.stdout.reconfigure(line_buffering=True) + parser = argparse.ArgumentParser(description=__doc__.splitlines()[0]) + commands = parser.add_subparsers(dest="command", required=True) + command = commands.add_parser( + "run", help="run the corpus with the grafeo installed here" + ) + command.add_argument("out", type=Path) + command = commands.add_parser( + "baseline", help="run the corpus on a release from PyPI" + ) + command.add_argument("version") + command.add_argument("out", type=Path) + command = commands.add_parser( + "candidate", help="run the corpus on a release build of a commit" + ) + command.add_argument("ref") + command.add_argument("out", type=Path) + command = commands.add_parser("compare", help="list the cases whose results differ") + command.add_argument("old", type=Path) + command.add_argument("new", type=Path) + command.add_argument("--reviewed", type=Path) + command = commands.add_parser( + "parity", help="list cases where GQL and Cypher disagree" + ) + command.add_argument("results", type=Path) + command = commands.add_parser( + "gate", help="baseline, candidate and compare in one go" + ) + command.add_argument( + "version", help="the release to compare against, for example 0.5.43" + ) + command.add_argument("--ref", default="HEAD") + command.add_argument("--reviewed", type=Path) + args = parser.parse_args(argv) + + if args.command == "run": + run_corpus(args.out) + elif args.command == "baseline": + baseline(args.version, args.out) + elif args.command == "candidate": + candidate(args.ref, args.out) + elif args.command == "compare": + return compare(args.old, args.new, args.reviewed) + elif args.command == "parity": + return parity(args.results) + elif args.command == "gate": + return gate(args.version, args.ref, args.reviewed) + return 0 + + +if __name__ == "__main__": + sys.exit(main()) diff --git a/scripts/difftest/reviewed/0.5.44.txt b/scripts/difftest/reviewed/0.5.44.txt new file mode 100644 index 000000000..61294159f --- /dev/null +++ b/scripts/difftest/reviewed/0.5.44.txt @@ -0,0 +1,368 @@ +# Reviewed differences between 0.5.43 and 0.5.44: case key, fingerprint of the reviewed result, why it changed. +A11|cypher b9ec8c9563 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A11|gql b9ec8c9563 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A12|cypher 84f42d9510 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A12|gql 84f42d9510 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A13|cypher bb20e37cb5 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A13|gql bb20e37cb5 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A16|cypher c614bee8b6 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A16|gql c614bee8b6 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A1|cypher ccffc715d4 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A1|gql ccffc715d4 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A2|cypher 8b1f46d48f Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A2|gql 8b1f46d48f Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A3|cypher 48a7411c57 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A3|gql 48a7411c57 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A4|cypher 58d0a75d29 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A4|gql 58d0a75d29 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A5|cypher 699776ae0e Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A5|gql 699776ae0e Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A7|cypher b9ec8c9563 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A7|gql b9ec8c9563 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A8|cypher 6b024f37c1 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +A8|gql 6b024f37c1 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). +AA1|cypher 685a587bd0 A one-hop quantifier binds a list, collected list items stay edges, pattern comprehensions run in aggregates. +AA2|gql 685a587bd0 A one-hop quantifier binds a list, collected list items stay edges, pattern comprehensions run in aggregates. +AA3|cypher b541c5a750 A one-hop quantifier binds a list, collected list items stay edges, pattern comprehensions run in aggregates. +AA3|gql b541c5a750 A one-hop quantifier binds a list, collected list items stay edges, pattern comprehensions run in aggregates. +AA4|cypher 01a4b3d404 A one-hop quantifier binds a list, collected list items stay edges, pattern comprehensions run in aggregates. +AA5|cypher 50005e4e24 A one-hop quantifier binds a list, collected list items stay edges, pattern comprehensions run in aggregates. +AA6|cypher 50005e4e24 A one-hop quantifier binds a list, collected list items stay edges, pattern comprehensions run in aggregates. +AB1|cypher 9b36882444 A CALL subquery returns new names only; a variable a WITH left out is undefined, not an internal error. +AB1|gql 9b36882444 A CALL subquery returns new names only; a variable a WITH left out is undefined, not an internal error. +AB2|cypher 9b36882444 A CALL subquery returns new names only; a variable a WITH left out is undefined, not an internal error. +AB2|gql 9b36882444 A CALL subquery returns new names only; a variable a WITH left out is undefined, not an internal error. +AB3|cypher 03e47cc203 A CALL subquery returns new names only; a variable a WITH left out is undefined, not an internal error. +AB3|gql 03e47cc203 A CALL subquery returns new names only; a variable a WITH left out is undefined, not an internal error. +AB4|cypher 3586a5f788 A CALL subquery returns new names only; a variable a WITH left out is undefined, not an internal error. +AB4|gql 3586a5f788 A CALL subquery returns new names only; a variable a WITH left out is undefined, not an internal error. +AB5|cypher d2e17893c7 A CALL subquery returns new names only; a variable a WITH left out is undefined, not an internal error. +AB5|gql d2e17893c7 A CALL subquery returns new names only; a variable a WITH left out is undefined, not an internal error. +AC10|cypher b9df7ad121 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC10|gql b9df7ad121 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC11|cypher 3a8d2d4ee5 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC11|gql 3a8d2d4ee5 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC12|cypher 04731a809e Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC12|gql 04731a809e Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC13|gql 04731a809e Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC1|cypher d792156d95 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC1|gql d792156d95 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC2|cypher 0768243410 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC2|gql 0768243410 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC3|cypher d0974271e7 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC3|gql d0974271e7 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC4|cypher a0bf6fdfb0 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC4|gql a0bf6fdfb0 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC5|cypher 5ad92dde94 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC6|cypher 992960337a Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC7|cypher 50005e4e24 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC7|gql 50005e4e24 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AC8|cypher bfdc2047f3 Subquery bodies take ORDER BY, SKIP, LIMIT and UNION; OPTIONAL MATCH works first and in Cypher EXISTS/COUNT. +AD1|cypher 03e47cc203 A subquery that imports nothing sees no outer variable (it read them as null). +AD1|gql 03e47cc203 A subquery that imports nothing sees no outer variable (it read them as null). +AD2|cypher 03e47cc203 A subquery that imports nothing sees no outer variable (it read them as null). +AD3|cypher c84f662233 A subquery that imports nothing sees no outer variable (it read them as null). +AD3|gql c84f662233 A subquery that imports nothing sees no outer variable (it read them as null). +AE1|cypher 9932bc4932 A later pattern after a shortest path comes back to its nodes; a subquery sharing nothing with the row is answered. +AE2|gql 9932bc4932 A later pattern after a shortest path comes back to its nodes; a subquery sharing nothing with the row is answered. +AE3|cypher 60c8e9a85a A later pattern after a shortest path comes back to its nodes; a subquery sharing nothing with the row is answered. +AE3|gql 60c8e9a85a A later pattern after a shortest path comes back to its nodes; a subquery sharing nothing with the row is answered. +AE4|cypher 9fc1b556b3 A later pattern after a shortest path comes back to its nodes; a subquery sharing nothing with the row is answered. +AE4|gql 9fc1b556b3 A later pattern after a shortest path comes back to its nodes; a subquery sharing nothing with the row is answered. +AF1|gql 199032d8c1 GQL NEXT passes the rows on; a VALUE subquery reads the outer row (and runs in WHERE). +AF2|gql 625b8e9fdb GQL NEXT passes the rows on; a VALUE subquery reads the outer row (and runs in WHERE). +AF3|gql 5cb2de68c2 GQL NEXT passes the rows on; a VALUE subquery reads the outer row (and runs in WHERE). +AF4|gql ab159b1cb9 GQL NEXT passes the rows on; a VALUE subquery reads the outer row (and runs in WHERE). +B10|cypher 53d503c3db Dotted access on startNode(r) fails with a clearer message. +B12|cypher 3023346a9b A node without w no longer returns the w of the edge with the same ID after WITH DISTINCT. +B16|cypher 6d8ebdc593 Collected and unwound edges keep their kind: x.w reads the edge, not the node with the same ID. +B6|gql 5d81268c40 An edge bound by LET is returned as an edge record, not 0. +B8|cypher 22ea1c58b5 A later MATCH through an edge bound earlier matches only that edge. +C13|cypher b62ffdb611 UNION in a Cypher CALL subquery runs (it failed to parse). +C16|cypher cd3e8f944c Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C16|gql cd3e8f944c Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C18|cypher d06f568726 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C18|gql d06f568726 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C19|cypher 44a244f56a Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C19|gql 44a244f56a Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C20|gql cb687ca938 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C23|cypher b9ec8c9563 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C23|gql b9ec8c9563 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C24|cypher ae3aa57892 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C24|gql ae3aa57892 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C5|cypher 3111cdd764 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C5|gql 3111cdd764 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C7|cypher 107e84a816 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C7|gql 107e84a816 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C8|gql 0ed77be421 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +C9|gql c048d20456 Set operations: each branch is planned on its own, so later branches return nodes and edges as records and UNION, EXCEPT and INTERSECT compare records (#481, #482). +D10|cypher 950b611aeb ORDER BY no longer adds its sort keys to the result as extra columns. +D10|gql 950b611aeb ORDER BY no longer adds its sort keys to the result as extra columns. +D16|cypher 0ce132321f ORDER BY no longer adds its sort keys to the result as extra columns. +D16|gql 0ce132321f ORDER BY no longer adds its sort keys to the result as extra columns. +D17|cypher 911c1f4d14 ORDER BY no longer adds its sort keys to the result as extra columns. +D17|gql 911c1f4d14 ORDER BY no longer adds its sort keys to the result as extra columns. +D18|cypher b03c11177d ORDER BY no longer adds its sort keys to the result as extra columns. +D18|gql b03c11177d ORDER BY no longer adds its sort keys to the result as extra columns. +D19|cypher d106acaafe ORDER BY no longer adds its sort keys to the result as extra columns. +D19|gql d106acaafe ORDER BY no longer adds its sort keys to the result as extra columns. +D1|cypher c51d3d1b0c ORDER BY no longer adds its sort keys to the result as extra columns. +D1|gql c51d3d1b0c ORDER BY no longer adds its sort keys to the result as extra columns. +D20|cypher 8033c6529b ORDER BY no longer adds its sort keys to the result as extra columns. +D20|gql 8033c6529b ORDER BY no longer adds its sort keys to the result as extra columns. +D22|cypher 953b72b25c ORDER BY no longer adds its sort keys to the result as extra columns. +D22|gql 953b72b25c ORDER BY no longer adds its sort keys to the result as extra columns. +D4|cypher bd1f72b4ad ORDER BY no longer adds its sort keys to the result as extra columns. +D4|gql bd1f72b4ad ORDER BY no longer adds its sort keys to the result as extra columns. +D5|cypher 950b611aeb ORDER BY no longer adds its sort keys to the result as extra columns. +D5|gql 950b611aeb ORDER BY no longer adds its sort keys to the result as extra columns. +D6|cypher b59d233609 ORDER BY no longer adds its sort keys to the result as extra columns. +D6|gql b59d233609 ORDER BY no longer adds its sort keys to the result as extra columns. +D8|cypher 05954ef876 ORDER BY no longer adds its sort keys to the result as extra columns. +D8|gql 05954ef876 ORDER BY no longer adds its sort keys to the result as extra columns. +E10|cypher 30cbbd107e A property read after WITH ... ORDER BY, SKIP, LIMIT or DISTINCT comes from the right node or edge (null where it has none). +E11|cypher c80dcbfaab A property read after WITH ... ORDER BY, SKIP, LIMIT or DISTINCT comes from the right node or edge (null where it has none). +E12|cypher f6e1c2ff61 A later MATCH through an edge bound earlier matches only that edge. +E14|cypher e54417a92b A property read after WITH ... ORDER BY, SKIP, LIMIT or DISTINCT comes from the right node or edge (null where it has none). +E15|cypher 90b8ce5686 keys() of an edge lists its keys instead of null. +E3|cypher 39c4f76178 A property read after WITH ... ORDER BY, SKIP, LIMIT or DISTINCT comes from the right node or edge (null where it has none). +E4|cypher 879f24a6f3 Properties after WITH ... ORDER BY come from the edge, not the node with the same ID. +E5|cypher 1e2b9d7838 A property read after WITH ... ORDER BY, SKIP, LIMIT or DISTINCT comes from the right node or edge (null where it has none). +E6|cypher bf61b6e064 A property read after WITH ... ORDER BY, SKIP, LIMIT or DISTINCT comes from the right node or edge (null where it has none). +F11|cypher bba3b9ab66 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). Also across row batches. +F11|gql bba3b9ab66 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). Also across row batches. +F13|cypher f5d9a5f906 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). Also across row batches. +F13|gql f5d9a5f906 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). Also across row batches. +F4|cypher 9d561aa59d Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). Also across row batches. +F4|gql 9d561aa59d Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). Also across row batches. +F6|cypher c8bda6e6e3 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). Also across row batches. +F6|gql c8bda6e6e3 Edges returned after ORDER BY, SKIP, LIMIT or DISTINCT are edge records, not 0 (#482). Also across row batches. +G1|cypher 099bb21d93 ORDER BY uses one total order, the openCypher one (maps, lists, strings, booleans, numbers, then null). +G1|gql 099bb21d93 ORDER BY uses one total order, the openCypher one (maps, lists, strings, booleans, numbers, then null). +G2|cypher cbcc57cba4 ORDER BY uses one total order, the openCypher one (maps, lists, strings, booleans, numbers, then null). +G2|gql cbcc57cba4 ORDER BY uses one total order, the openCypher one (maps, lists, strings, booleans, numbers, then null). +I1|cypher 84d111dabe RETURN * with ORDER BY works instead of failing with an internal error. +I1|gql 84d111dabe RETURN * with ORDER BY works instead of failing with an internal error. +I2|cypher 97b7747133 RETURN * with ORDER BY works instead of failing with an internal error. +I2|gql 97b7747133 RETURN * with ORDER BY works instead of failing with an internal error. +J2|cypher ddde428e93 An edge kept as a group key is returned as an edge record, not an ID. +J2|gql ddde428e93 An edge kept as a group key is returned as an edge record, not an ID. +K10|cypher 1564f1a991 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K10|gql 1564f1a991 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K11|cypher 5ca6923aa1 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K11|gql 5ca6923aa1 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K12|cypher 2fe3d07373 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K12|gql 2fe3d07373 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K15|cypher 97daaa6efb A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K15|gql 97daaa6efb A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K18|cypher 5df72cecce A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K18|gql 5df72cecce A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K1|cypher 3f9ac5693e A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K1|gql 3f9ac5693e A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K20|cypher e091683d87 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K20|gql e091683d87 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K21|cypher ce23c7ebe7 Properties after a later MATCH come from the right node: nodes without w no longer read an edge's w. +K21|gql ce23c7ebe7 Properties after a later MATCH come from the right node: nodes without w no longer read an edge's w. +K23|cypher 4495158f51 A later MATCH after WITH keeps the earlier variables (GQL failed with an internal error); a node without w reads null. +K23|gql 4495158f51 A later MATCH after WITH keeps the earlier variables (GQL failed with an internal error); a node without w reads null. +K25|cypher f7f2fcdc6e A later MATCH through an edge list bound earlier matches only that list. +K25|gql f7f2fcdc6e A later MATCH through an edge list bound earlier matches only that list. +K26|cypher b16b37e1c6 The same edge twice in one path can only be a loop. +K26|gql b16b37e1c6 The same edge twice in one path can only be a loop. +K2|cypher 52c4eb669a A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K2|gql 52c4eb669a A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K31|cypher f57d122a33 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K32|cypher df63ad5130 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K32|gql df63ad5130 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K3|cypher 4242b130f3 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K3|gql 4242b130f3 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K42|cypher 6dfc226ed5 An edge has no property i: r.i no longer reads the node with the same ID. +K43|cypher a57749766d An edge has no property i: r.i no longer reads the node with the same ID. +K4|cypher e32d209c71 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K4|gql e32d209c71 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K5|cypher b16b37e1c6 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K5|gql b16b37e1c6 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K6|cypher 3bd58d54a0 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K6|gql 3bd58d54a0 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K7|cypher 6bfbcda916 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K7|gql 6bfbcda916 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K8|cypher 6bfbcda916 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K8|gql 6bfbcda916 A later MATCH through a node or edge bound earlier matches only that node or edge, and properties after a later MATCH come from the right entity. +K9|cypher ec5822ecd4 A path back to an earlier variable inside CALL ends at the node the variable holds. +K9|gql ec5822ecd4 A path back to an earlier variable inside CALL ends at the node the variable holds. +L10|cypher 714306f254 ORDER BY uses one total order, the openCypher one: values of different types, lists, NaN after infinity, and NULLS FIRST/LAST hold with DESC. +L10|gql 714306f254 ORDER BY uses one total order, the openCypher one: values of different types, lists, NaN after infinity, and NULLS FIRST/LAST hold with DESC. +L11|cypher 16d6e49e75 ORDER BY uses one total order, the openCypher one: values of different types, lists, NaN after infinity, and NULLS FIRST/LAST hold with DESC. +L11|gql 16d6e49e75 ORDER BY uses one total order, the openCypher one: values of different types, lists, NaN after infinity, and NULLS FIRST/LAST hold with DESC. +L18|cypher 91773d0d1d ORDER BY uses one total order, the openCypher one: values of different types, lists, NaN after infinity, and NULLS FIRST/LAST hold with DESC. +L18|gql 91773d0d1d ORDER BY uses one total order, the openCypher one: values of different types, lists, NaN after infinity, and NULLS FIRST/LAST hold with DESC. +L1|cypher 6f905cbea6 ORDER BY uses one total order, the openCypher one: values of different types, lists, NaN after infinity, and NULLS FIRST/LAST hold with DESC. +L1|gql 6f905cbea6 ORDER BY uses one total order, the openCypher one: values of different types, lists, NaN after infinity, and NULLS FIRST/LAST hold with DESC. +L2|cypher 7c7e33916a ORDER BY uses one total order, the openCypher one: values of different types, lists, NaN after infinity, and NULLS FIRST/LAST hold with DESC. +L2|gql 7c7e33916a ORDER BY uses one total order, the openCypher one: values of different types, lists, NaN after infinity, and NULLS FIRST/LAST hold with DESC. +L5|gql 16a470385a ORDER BY uses one total order, the openCypher one: values of different types, lists, NaN after infinity, and NULLS FIRST/LAST hold with DESC. +L7|gql ad110d7c47 ORDER BY uses one total order, the openCypher one: values of different types, lists, NaN after infinity, and NULLS FIRST/LAST hold with DESC. +M10|cypher c978f9d66f EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M10|gql c978f9d66f EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M11|cypher 63dce8b5af EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M11|gql 63dce8b5af EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M12|cypher 0b481da4be EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M12|gql 0b481da4be EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M13|cypher 26d63acab1 A null node makes EXISTS false and COUNT 0 instead of null. +M13|gql 26d63acab1 A null node makes EXISTS false and COUNT 0 instead of null. +M14|cypher 3a406ffcd0 A null node makes NOT EXISTS true, so the row is kept. +M14|gql 3a406ffcd0 A null node makes NOT EXISTS true, so the row is kept. +M16|cypher f32d41cf48 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M16|gql f32d41cf48 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M17|cypher acf4b125d3 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M17|gql acf4b125d3 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M19|cypher c5b55220d4 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M19|gql c5b55220d4 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M20|cypher 1235dca336 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M20|gql 1235dca336 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M21|cypher 2ece2d5589 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M21|gql 2ece2d5589 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M22|cypher 3e8c510773 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M22|gql 3e8c510773 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M23|cypher 2ece2d5589 A Cypher MATCH (b:City) after the edge checks that b is a City. +M24|cypher e9e00f0208 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M2|cypher 1f8b9f073f EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M2|gql 1f8b9f073f EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M31|cypher 1f8b9f073f A pattern predicate in WHERE works (it failed before). +M32|cypher 90d8956923 A negated pattern predicate in WHERE works (it failed before). +M33|cypher 8442578b89 EXISTS uses the row's c, so each person only matches the city they live in. +M33|gql 8442578b89 EXISTS uses the row's c, so each person only matches the city they live in. +M36|cypher 1f8b9f073f COUNT compared in WHERE counts per row. +M36|gql 1f8b9f073f COUNT compared in WHERE counts per row. +M38|cypher 6354c87d40 NOT EXISTS uses both nodes of the row. +M38|gql 6354c87d40 NOT EXISTS uses both nodes of the row. +M39|cypher c9a189f776 COUNT and EXISTS in RETURN use the row's node. +M39|gql c9a189f776 COUNT and EXISTS in RETURN use the row's node. +M4|cypher 787943d3b5 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M4|gql 787943d3b5 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M5|cypher 045a78e29c EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M5|gql 045a78e29c EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M6|cypher d3a64ea3af EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M6|gql d3a64ea3af EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M7|cypher 41b584d5c4 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M7|gql 41b584d5c4 EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M9|cypher 0b481da4be EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +M9|gql 0b481da4be EXISTS and COUNT subqueries match their whole pattern and use the row's nodes and edges. +N10|cypher 2226820635 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N10|gql 2226820635 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N11|cypher 306a927d66 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N11|gql 306a927d66 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N1|cypher 35438c1583 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N1|gql 35438c1583 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N2|cypher 21d44f5fc1 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N2|gql 21d44f5fc1 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N3|cypher 7fb15da67b Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N3|gql 7fb15da67b Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N4|cypher 5ca71023f9 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N4|gql 5ca71023f9 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N5|cypher 3512a005c2 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N5|gql 3512a005c2 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N6|cypher f9fd9544be Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N6|gql f9fd9544be Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N8|cypher 9a485dd7dd Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N8|gql 9a485dd7dd Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N9|cypher 3e43b2f052 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +N9|gql 3e43b2f052 Nodes and edges keep their type through OPTIONAL MATCH, EXISTS, CALL, UNWIND and grouping, so properties come from the right entity. +O10|cypher 2ad8297ca6 Unwound edges keep their kind, so a later MATCH through them matches only those edges. +O10|gql 2ad8297ca6 Unwound edges keep their kind, so a later MATCH through them matches only those edges. +O11|cypher d3617ffdf5 An item of a collected edge list reads that edge's property. +O11|gql d3617ffdf5 An item of a collected edge list reads that edge's property. +O1|cypher 21e4702b26 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O1|gql 21e4702b26 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O2|cypher f9fd9544be Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O2|gql f9fd9544be Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O3|cypher 8c26362660 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O3|gql 8c26362660 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O4|cypher 01a4b3d404 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O4|gql 01a4b3d404 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O5|cypher 15ef32fd86 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O5|gql 15ef32fd86 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O6|cypher 5c84808b02 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O6|gql 5c84808b02 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O7|cypher 3eec7cfdf9 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O7|gql 3eec7cfdf9 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O8|cypher cbaff6f467 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O8|gql cbaff6f467 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O9|cypher ac62dd2342 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +O9|gql ac62dd2342 Lists of nodes and edges keep their kind through collect, grouping and UNWIND. +P1|cypher 78bd9436fa keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P1|gql 78bd9436fa keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P2|cypher 01a4b3d404 keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P2|gql 01a4b3d404 keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P3|cypher ad20846bc2 keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P3|gql ad20846bc2 keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P4|cypher cbf0e2357c keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P4|gql cbf0e2357c keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P5|cypher b08d890fc3 keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P5|gql b08d890fc3 keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P6|cypher cd8909e056 keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P7|cypher cbf0e2357c keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P7|gql cbf0e2357c keys() of an edge lists its keys; a variable used as both a node and an edge is rejected. +P8|cypher c3a135fc82 A CALL subquery can match an imported edge as an edge. +P8|gql c3a135fc82 A CALL subquery can match an imported edge as an edge. +Q1|cypher b4929f3bc0 EXISTS and COUNT subqueries tied to the outer row by a value are answered per row (#543). +Q1|gql b4929f3bc0 EXISTS and COUNT subqueries tied to the outer row by a value are answered per row (#543). +Q2|cypher d0a27b5194 EXISTS and COUNT subqueries tied to the outer row by a value are answered per row (#543). +Q2|gql d0a27b5194 EXISTS and COUNT subqueries tied to the outer row by a value are answered per row (#543). +Q3|cypher c6fcb6a64b EXISTS and COUNT subqueries tied to the outer row by a value are answered per row (#543). +Q3|gql c6fcb6a64b EXISTS and COUNT subqueries tied to the outer row by a value are answered per row (#543). +Q4|cypher 8dea950d1a EXISTS and COUNT subqueries tied to the outer row by a value are answered per row (#543). +Q4|gql 8dea950d1a EXISTS and COUNT subqueries tied to the outer row by a value are answered per row (#543). +Q5|cypher 0af437ca0d COUNT over a variable-length path counts every path, not only neighbors. +Q5|gql 0af437ca0d COUNT over a variable-length path counts every path, not only neighbors. +Q8|cypher f85f46c33e EXISTS and COUNT subqueries tied to the outer row by a value are answered per row (#543). +Q8|gql f85f46c33e EXISTS and COUNT subqueries tied to the outer row by a value are answered per row (#543). +R1|cypher d2dbc8421e EXISTS in WHERE tied to the outer row by a value is answered per row (#543). +R2|cypher 353aa3ace4 EXISTS in WHERE tied to the outer row by a value is answered per row (#543). +R4|cypher 51d167f03b EXISTS can compare an edge of the subquery with the row's edge. +R4|gql 51d167f03b EXISTS can compare an edge of the subquery with the row's edge. +R5|cypher 5ae5f14831 EXISTS in WHERE tied to the outer row by a value is answered per row (#543). +R5|gql 5ae5f14831 EXISTS in WHERE tied to the outer row by a value is answered per row (#543). +R6|cypher 642e7dcad0 A CALL subquery over two edges in a row runs again for each outer row. +S1|gql 2afbd4bc03 GQL VALUE { ... RETURN count(x) } counts only rows where x is not null. +S3|gql ca5f97075f GQL VALUE { ... RETURN count(x) } counts only rows where x is not null. +T1|cypher 792313fb51 The MATCH clauses of a GQL subquery go on from each other. +T1|gql 792313fb51 The MATCH clauses of a GQL subquery go on from each other. +T2|cypher 1c193e7c9b The MATCH clauses of a GQL subquery go on from each other. +T2|gql 1c193e7c9b The MATCH clauses of a GQL subquery go on from each other. +T3|gql f905cfffa8 The MATCH clauses of a GQL subquery go on from each other. +U1|cypher f5175a9387 A WHERE in a Cypher importing WITH is rejected, as in Neo4j (it was ignored). +U2|cypher 2f98f94f97 An alias in a Cypher importing WITH is rejected, as in Neo4j (it imported the wrong value). +V1|cypher d433a58f8c A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V1|gql d433a58f8c A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V2|cypher 0940d1f4d9 A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V2|gql 0940d1f4d9 A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V3|cypher 69dd0d0713 A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V3|gql 69dd0d0713 A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V4|cypher ff6d86a896 A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V4|gql ff6d86a896 A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V5|cypher 1a583cc22a A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V5|gql 1a583cc22a A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V6|cypher c7df6dd70c A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V6|gql c7df6dd70c A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V7|cypher ddf6a6aac7 A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V7|gql ddf6a6aac7 A CALL subquery returns its nodes and edges as references, so a later MATCH, equality, id() and type() see them. +V8|cypher 10f03920db A Cypher query no longer ends with a CALL subquery that returns rows, as in openCypher. +W1|gql ded0bcfaf3 A GQL CALL subquery sees the outer row's variables; CALL (a) { ... } and CALL () { ... } parse. +W2|gql c3a135fc82 A GQL CALL subquery sees the outer row's variables; CALL (a) { ... } and CALL () { ... } parse. +W3|gql c55eb136e9 A WITH in a GQL CALL subquery is an ordinary WITH, so its WHERE filters. +W4|gql ded0bcfaf3 A GQL CALL subquery sees the outer row's variables; CALL (a) { ... } and CALL () { ... } parse. +W5|gql c3a135fc82 A GQL CALL subquery sees the outer row's variables; CALL (a) { ... } and CALL () { ... } parse. +W6|gql 1f8ab4f00d A GQL CALL subquery sees the outer row's variables; CALL (a) { ... } and CALL () { ... } parse. +W7|gql 36f86522fe A GQL CALL subquery sees the outer row's variables; CALL (a) { ... } and CALL () { ... } parse. +X1|cypher ded0bcfaf3 Cypher CALL (a) { ... }, CALL (*) { ... } and CALL () { ... } parse. +X2|cypher ded0bcfaf3 Cypher CALL (a) { ... }, CALL (*) { ... } and CALL () { ... } parse. +X3|cypher c3a135fc82 Cypher CALL (a) { ... }, CALL (*) { ... } and CALL () { ... } parse. +X4|cypher 1f8ab4f00d Cypher CALL (a) { ... }, CALL (*) { ... } and CALL () { ... } parse. +X5|cypher c55eb136e9 Cypher CALL (a) { ... }, CALL (*) { ... } and CALL () { ... } parse. +Y1|cypher 51d167f03b RETURN * in a CALL subquery returns the variables the subquery binds. +Y1|gql 51d167f03b RETURN * in a CALL subquery returns the variables the subquery binds. +Y2|cypher 0940d1f4d9 RETURN * in a CALL subquery returns the variables the subquery binds. +Y2|gql 0940d1f4d9 RETURN * in a CALL subquery returns the variables the subquery binds. +Y4|gql ffe2a647a1 RETURN * in a CALL subquery returns the variables the subquery binds. +Z1|gql 9e1cb3dfb1 A CALL subquery's RETURN under ORDER BY and LIMIT also returns references. diff --git a/scripts/npm-publish.sh b/scripts/npm-publish.sh new file mode 100644 index 000000000..a6ecf103d --- /dev/null +++ b/scripts/npm-publish.sh @@ -0,0 +1,19 @@ +#!/usr/bin/env bash +# Publishes the npm package in the current directory, or skips it when this +# version is already on npm, so a re-run of a release is safe while a real +# failure still fails the job. Extra arguments go to `npm publish`. +# +# Publishing uses npm trusted publishing: the job needs `id-token: write` and +# npm 11.5.1 or later, and no token. +set -euo pipefail + +name=$(node -p "require('./package.json').name") +version=$(node -p "require('./package.json').version") + +if [ -n "$(npm view "${name}@${version}" version 2>/dev/null || true)" ]; then + echo "${name}@${version} is already on npm, skipping" + exit 0 +fi + +echo "Publishing ${name}@${version}" +npm publish --access public "$@" diff --git a/scripts/tests/test_difftest.py b/scripts/tests/test_difftest.py new file mode 100644 index 000000000..b35214580 --- /dev/null +++ b/scripts/tests/test_difftest.py @@ -0,0 +1,230 @@ +"""Tests for scripts/difftest (the comparison and the corpus; no grafeo build needed). + +Run: uv run --with pytest python -m pytest scripts/tests +""" + +from __future__ import annotations + +import json +import re +import sys +from pathlib import Path + +import pytest + +sys.path.insert(0, str(Path(__file__).resolve().parents[1] / "difftest")) + +import corpus # noqa: E402 +import difftest # noqa: E402 + + +def rows(*values, ordered=False, columns=("x",)): + return { + "query": "q", + "ordered": ordered, + "columns": list(columns), + "rows": [list(v) for v in values], + } + + +def error(text): + return {"query": "q", "ordered": False, "error": text} + + +def write_results(path: Path, results: dict, version="0.0.0") -> Path: + payload = {"meta": {"version": version, "commit": None}, "results": results} + path.write_text(json.dumps(payload), encoding="utf-8") + return path + + +# --- normalize --------------------------------------------------------------- + + +def test_normalize_makes_floats_comparable(): + assert difftest.normalize(0.1 + 0.2) == 0.3 + assert difftest.normalize(float("nan")) == {"float": "NaN"} + assert difftest.normalize(float("-inf")) == {"float": "-Infinity"} + assert difftest.normalize(True) is True + + +def test_normalize_sorts_map_keys_and_keeps_list_order(): + value = {"b": [3, 1], "a": (2.5, None)} + assert difftest.normalize(value) == {"a": [2.5, None], "b": [3, 1]} + assert list(difftest.normalize(value)) == ["a", "b"] + + +def test_normalize_uses_an_entities_dict_form(): + class Node: + def to_dict(self): + return {"name": "Alix", "id": 0} + + assert difftest.normalize(Node()) == {"type": "Node", "id": 0, "name": "Alix"} + + +# --- canonical and differing -------------------------------------------------- + + +def test_unordered_rows_compare_as_a_multiset(): + assert difftest.canonical(rows([1], [2])) == difftest.canonical(rows([2], [1])) + assert difftest.canonical(rows([1], [1])) != difftest.canonical(rows([1])) + + +def test_ordered_rows_compare_in_order(): + first = rows([1], [2], ordered=True) + second = rows([2], [1], ordered=True) + assert difftest.canonical(first) != difftest.canonical(second) + + +def test_a_value_that_changes_type_differs(): + old = {"A1|gql": rows([1]), "A2|gql": rows([1]), "A3|gql": rows([True])} + new = {"A1|gql": rows([1.0]), "A2|gql": rows([True]), "A3|gql": rows([True])} + assert difftest.differing(old, new) == ["A1|gql", "A2|gql"] + + +def test_differing_lists_changed_cases_in_corpus_order(): + old = { + "A10|gql": rows([1]), + "A2|gql": rows([1]), + "B1|cypher": rows([1]), + "C1|gql": rows([1]), + } + new = { + "A10|gql": rows([2]), + "A2|gql": error("boom"), + "B1|cypher": rows([1]), + "D1|gql": rows(), + } + assert difftest.differing(old, new) == ["A2|gql", "A10|gql"] + + +# --- the reviewed list -------------------------------------------------------- + + +def test_read_reviewed_skips_comments_and_blank_lines(tmp_path): + path = tmp_path / "reviewed.txt" + path.write_text( + "# 0.5.44 against 0.5.43\n\nA1|gql 0123456789 Edges keep their values (#482)\n", + encoding="utf-8", + ) + assert difftest.read_reviewed(path) == { + "A1|gql": ("0123456789", "Edges keep their values (#482)") + } + + +@pytest.mark.parametrize( + "text", + [ + "A1|gql 0123456789\n", # no reason + "A1 0123456789 no language\n", + "A1|gql 0123456789 one\nA1|gql 0123456789 two\n", # listed twice + ], +) +def test_read_reviewed_rejects_malformed_lines(tmp_path, text): + path = tmp_path / "reviewed.txt" + path.write_text(text, encoding="utf-8") + with pytest.raises(SystemExit): + difftest.read_reviewed(path) + + +def gate_files(tmp_path, reviewed_lines): + old = write_results( + tmp_path / "old.json", {"A1|gql": rows([1]), "A2|gql": rows([5])} + ) + new = write_results( + tmp_path / "new.json", {"A1|gql": rows([2]), "A2|gql": rows([5])} + ) + reviewed = tmp_path / "reviewed.txt" + reviewed.write_text( + "".join(f"{line}\n" for line in reviewed_lines), encoding="utf-8" + ) + return old, new, reviewed + + +def test_compare_passes_when_every_difference_is_reviewed(tmp_path, capsys): + digest = difftest.fingerprint(rows([2])) + old, new, reviewed = gate_files(tmp_path, [f"A1|gql {digest} A1 now returns 2"]) + assert difftest.compare(old, new, reviewed) == 0 + assert "all reviewed" in capsys.readouterr().out + + +def test_compare_fails_on_a_difference_nobody_reviewed(tmp_path, capsys): + old, new, reviewed = gate_files(tmp_path, []) + assert difftest.compare(old, new, reviewed) == 1 + out = capsys.readouterr().out + assert "[NOT REVIEWED]" in out + assert f"to review: A1|gql {difftest.fingerprint(rows([2]))}" in out + + +def test_compare_fails_when_a_reviewed_result_changed_since(tmp_path, capsys): + digest = difftest.fingerprint(rows([3])) + old, new, reviewed = gate_files( + tmp_path, [f"A1|gql {digest} A1 returned 3 when reviewed"] + ) + assert difftest.compare(old, new, reviewed) == 1 + assert "[CHANGED SINCE THE REVIEW]" in capsys.readouterr().out + + +def test_compare_fails_when_a_reviewed_difference_is_gone(tmp_path, capsys): + digest = difftest.fingerprint(rows([2])) + old, new, reviewed = gate_files( + tmp_path, + [f"A1|gql {digest} A1 now returns 2", f"A2|gql {digest} A2 changed once"], + ) + assert difftest.compare(old, new, reviewed) == 1 + assert "A2|gql [NO LONGER DIFFERENT]" in capsys.readouterr().out + + +def test_compare_without_a_reviewed_list_only_reports(tmp_path, capsys): + old, new, _ = gate_files(tmp_path, []) + assert difftest.compare(old, new) == 0 + out = capsys.readouterr().out + assert "=== A1|gql [changed]" in out + assert "REVIEW" not in out + + +# --- parity -------------------------------------------------------------------- + + +def test_parity_reports_languages_that_disagree(tmp_path, capsys): + results = { + "A1|gql": rows([1], columns=("n",)), + "A1|cypher": rows([1], columns=("a.n",)), # column names may differ + "A2|gql": rows([1]), + "A2|cypher": rows([2]), + "A3|gql": error("gql error"), + "A3|cypher": error("cypher error"), # both fail: they agree + "A4|gql": rows([1]), + "A4|cypher": error("unsupported"), + "A5|gql": rows([1]), # GQL only + } + path = write_results(tmp_path / "results.json", results) + assert difftest.parity(path) == 1 + out = capsys.readouterr().out + assert "=== A2:" in out and "=== A4:" in out + assert "=== A1:" not in out and "=== A3:" not in out + assert "2 cases where GQL and Cypher disagree" in out + + +# --- the corpus ------------------------------------------------------------------ + + +def test_case_ids_are_unique(): + ids = [case.id for case in corpus.CASES] + assert len(ids) == len(set(ids)) + + +def test_cases_name_known_languages_and_fixtures(): + for case in corpus.CASES: + assert case.languages and set(case.languages) <= corpus.LANGUAGES, case.id + assert case.fixture in corpus.FIXTURES, case.id + assert re.fullmatch(r"[A-Z]+\d+", case.id), case.id + + +def test_cases_do_not_write(): + # Each fixture is built once and shared by its cases, so a write would change + # what later cases see. + writes = re.compile( + r"\b(INSERT|CREATE|SET|REMOVE|DELETE|MERGE|DROP)\b", re.IGNORECASE + ) + for case in corpus.CASES: + assert not writes.search(case.query), case.id diff --git a/tests/spec/datasets/entity_kinds.setup b/tests/spec/datasets/entity_kinds.setup new file mode 100644 index 000000000..003b412da --- /dev/null +++ b/tests/spec/datasets/entity_kinds.setup @@ -0,0 +1,10 @@ +# Entity Kinds Dataset +# 5 Person and 3 City nodes, 5 KNOWS and 3 LIVES_IN edges. Nodes and edges +# both carry `w` (nodes 100 and up, edges 1 to 8) and their IDs overlap, so a +# node read as an edge, or the reverse, shows a value from the wrong range. +# Alix, Gus, Vincent (w 100, 101, 103), Jules and Mia; Amsterdam (w 104), +# Berlin, Paris (w 105). KNOWS w 1 to 5: Alix->Gus, Gus->Vincent, +# Vincent->Alix, Jules->Mia, Alix->Jules. LIVES_IN w 6 to 8: Alix->Amsterdam, +# Gus->Berlin, Mia->Paris. Jules, Mia and Berlin have no `w`. + +INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par) diff --git a/tests/spec/datasets/ldbc_snb_mini.setup b/tests/spec/datasets/ldbc_snb_mini.setup new file mode 100644 index 000000000..47198e1d1 --- /dev/null +++ b/tests/spec/datasets/ldbc_snb_mini.setup @@ -0,0 +1,389 @@ +# LDBC SNB Mini Dataset +# A small slice of the LDBC Social Network Benchmark Interactive v1 test data +# (github.com/ldbc/ldbc_snb_interactive_v1_impls, cypher/test-data/vanilla), for +# the LDBC SNB Interactive v1 queries in lpg/cypher/ldbc_snb_interactive.gtest, +# lpg/gql/ldbc_snb_interactive.gtest and lpg/sql_pgq/ldbc_snb_interactive.gtest. +# Every node, property value and edge is copied from that data unchanged; the +# slice only leaves things out. Labels and property names follow the LDBC Neo4j +# import (Place:City, Organisation:Company, Post:Message, Comment:Message, +# `speaks` and `email` as lists); dates are epoch milliseconds. One statement +# per node, with its outgoing edges. +# +# Persons: Asher Mamo (228, Adama), his friends Alfonso Alvarez (150), Jae-Jin +# Park (76), Aurora Cruz (2199023255712) and Philibert Roindefo (102), their +# friends Abdala Ndiaye (153), Rafael Fernández (4398046511333), Jie Yang +# (6597069766775) and Maria Alkaios (143), John Kumar (41), a friend of Maria, +# and, without messages of their own, Gary Hill (8796093022357, a friend of +# Asher), Abdullah Koksal (8796093022390), who likes a message of Asher, and +# Javed Khan (10995116277918), whom Gary befriends in an update. KNOWS: the +# 29 edges among them. +# Messages: 44 posts and 48 comments of these persons: their latest +# posts, replies between them (with the whole reply chain), the messages of +# Asher that they like, and five messages located outside the creator's +# country. Forums: the 22 forums of those posts with their members among +# these persons. Tags: the 113 tags of those messages, with their tag +# classes up to Thing; interests and forum tags only for these tags. +# Places and organisations: the ones these persons and messages refer to. Also +# the post Asher likes, the tags and companies referenced by the update events +# of the LDBC update stream that the gtests replay. + +INSERT (pl1454:Place:Continent {id: 1454, name: 'Asia', url: 'http://dbpedia.org/resource/Asia'}) +INSERT (pl1455:Place:Continent {id: 1455, name: 'Africa', url: 'http://dbpedia.org/resource/Africa'}) +INSERT (pl1456:Place:Continent {id: 1456, name: 'Europe', url: 'http://dbpedia.org/resource/Europe'}) +INSERT (pl1457:Place:Continent {id: 1457, name: 'South_America', url: 'http://dbpedia.org/resource/South_America'}) +INSERT (pl1458:Place:Continent {id: 1458, name: 'North_America', url: 'http://dbpedia.org/resource/North_America'}) +MATCH (pl1454:Place {id: 1454}) INSERT (pl0:Place:Country {id: 0, name: 'India', url: 'http://dbpedia.org/resource/India'})-[:IS_PART_OF]->(pl1454) +MATCH (pl1454:Place {id: 1454}) INSERT (pl1:Place:Country {id: 1, name: 'China', url: 'http://dbpedia.org/resource/China'})-[:IS_PART_OF]->(pl1454) +MATCH (pl1456:Place {id: 1456}) INSERT (pl14:Place:Country {id: 14, name: 'Finland', url: 'http://dbpedia.org/resource/Finland'})-[:IS_PART_OF]->(pl1456) +MATCH (pl1458:Place {id: 1458}) INSERT (pl30:Place:Country {id: 30, name: 'Nicaragua', url: 'http://dbpedia.org/resource/Nicaragua'})-[:IS_PART_OF]->(pl1458) +MATCH (pl1457:Place {id: 1457}) INSERT (pl47:Place:Country {id: 47, name: 'Uruguay', url: 'http://dbpedia.org/resource/Uruguay'})-[:IS_PART_OF]->(pl1457) +MATCH (pl1458:Place {id: 1458}) INSERT (pl53:Place:Country {id: 53, name: 'Mexico', url: 'http://dbpedia.org/resource/Mexico'})-[:IS_PART_OF]->(pl1458) +MATCH (pl1454:Place {id: 1454}) INSERT (pl54:Place:Country {id: 54, name: 'Pakistan', url: 'http://dbpedia.org/resource/Pakistan'})-[:IS_PART_OF]->(pl1454) +MATCH (pl1454:Place {id: 1454}) INSERT (pl55:Place:Country {id: 55, name: 'Philippines', url: 'http://dbpedia.org/resource/Philippines'})-[:IS_PART_OF]->(pl1454) +MATCH (pl1458:Place {id: 1458}) INSERT (pl57:Place:Country {id: 57, name: 'United_States', url: 'http://dbpedia.org/resource/United_States'})-[:IS_PART_OF]->(pl1458) +MATCH (pl1458:Place {id: 1458}) INSERT (pl66:Place:Country {id: 66, name: 'Canada', url: 'http://dbpedia.org/resource/Canada'})-[:IS_PART_OF]->(pl1458) +MATCH (pl1456:Place {id: 1456}) INSERT (pl75:Place:Country {id: 75, name: 'England', url: 'http://dbpedia.org/resource/England'})-[:IS_PART_OF]->(pl1456) +MATCH (pl1455:Place {id: 1455}) INSERT (pl76:Place:Country {id: 76, name: 'Ethiopia', url: 'http://dbpedia.org/resource/Ethiopia'})-[:IS_PART_OF]->(pl1455) +MATCH (pl1456:Place {id: 1456}) INSERT (pl78:Place:Country {id: 78, name: 'Greece', url: 'http://dbpedia.org/resource/Greece'})-[:IS_PART_OF]->(pl1456) +MATCH (pl1455:Place {id: 1455}) INSERT (pl84:Place:Country {id: 84, name: 'Madagascar', url: 'http://dbpedia.org/resource/Madagascar'})-[:IS_PART_OF]->(pl1455) +MATCH (pl1455:Place {id: 1455}) INSERT (pl90:Place:Country {id: 90, name: 'Niger', url: 'http://dbpedia.org/resource/Niger'})-[:IS_PART_OF]->(pl1455) +MATCH (pl1455:Place {id: 1455}) INSERT (pl96:Place:Country {id: 96, name: 'Senegal', url: 'http://dbpedia.org/resource/Senegal'})-[:IS_PART_OF]->(pl1455) +MATCH (pl1454:Place {id: 1454}) INSERT (pl98:Place:Country {id: 98, name: 'South_Korea', url: 'http://dbpedia.org/resource/South_Korea'})-[:IS_PART_OF]->(pl1454) +MATCH (pl1456:Place {id: 1456}) INSERT (pl99:Place:Country {id: 99, name: 'Spain', url: 'http://dbpedia.org/resource/Spain'})-[:IS_PART_OF]->(pl1456) +MATCH (pl1454:Place {id: 1454}) INSERT (pl105:Place:Country {id: 105, name: 'Turkey', url: 'http://dbpedia.org/resource/Turkey'})-[:IS_PART_OF]->(pl1454) +MATCH (pl0:Place {id: 0}) INSERT (pl176:Place:City {id: 176, name: 'Bangalore', url: 'http://dbpedia.org/resource/Bangalore'})-[:IS_PART_OF]->(pl0) +MATCH (pl0:Place {id: 0}) INSERT (pl185:Place:City {id: 185, name: 'Puttur', url: 'http://dbpedia.org/resource/Puttur'})-[:IS_PART_OF]->(pl0) +MATCH (pl1:Place {id: 1}) INSERT (pl443:Place:City {id: 443, name: 'Dafeng', url: 'http://dbpedia.org/resource/Dafeng'})-[:IS_PART_OF]->(pl1) +MATCH (pl1:Place {id: 1}) INSERT (pl469:Place:City {id: 469, name: 'Huainan', url: 'http://dbpedia.org/resource/Huainan'})-[:IS_PART_OF]->(pl1) +MATCH (pl53:Place {id: 53}) INSERT (pl732:Place:City {id: 732, name: 'Puebla', url: 'http://dbpedia.org/resource/Puebla'})-[:IS_PART_OF]->(pl53) +MATCH (pl53:Place {id: 53}) INSERT (pl745:Place:City {id: 745, name: 'San_Luis_Potosí', url: 'http://dbpedia.org/resource/San_Luis_Potosí'})-[:IS_PART_OF]->(pl53) +MATCH (pl54:Place {id: 54}) INSERT (pl771:Place:City {id: 771, name: 'Topi', url: 'http://dbpedia.org/resource/Topi'})-[:IS_PART_OF]->(pl54) +MATCH (pl54:Place {id: 54}) INSERT (pl780:Place:City {id: 780, name: 'Major_Cities', url: 'http://dbpedia.org/resource/Major_Cities'})-[:IS_PART_OF]->(pl54) +MATCH (pl55:Place {id: 55}) INSERT (pl801:Place:City {id: 801, name: 'Cebu_City', url: 'http://dbpedia.org/resource/Cebu_City'})-[:IS_PART_OF]->(pl55) +MATCH (pl55:Place {id: 55}) INSERT (pl826:Place:City {id: 826, name: 'Tagbilaran', url: 'http://dbpedia.org/resource/Tagbilaran'})-[:IS_PART_OF]->(pl55) +MATCH (pl66:Place {id: 66}) INSERT (pl1024:Place:City {id: 1024, name: 'London', url: 'http://dbpedia.org/resource/London'})-[:IS_PART_OF]->(pl66) +MATCH (pl75:Place {id: 75}) INSERT (pl1118:Place:City {id: 1118, name: 'Leeds', url: 'http://dbpedia.org/resource/Leeds'})-[:IS_PART_OF]->(pl75) +MATCH (pl76:Place {id: 76}) INSERT (pl1121:Place:City {id: 1121, name: 'Nekemte', url: 'http://dbpedia.org/resource/Nekemte'})-[:IS_PART_OF]->(pl76) +MATCH (pl76:Place {id: 76}) INSERT (pl1127:Place:City {id: 1127, name: 'Adama', url: 'http://dbpedia.org/resource/Adama'})-[:IS_PART_OF]->(pl76) +MATCH (pl78:Place {id: 78}) INSERT (pl1142:Place:City {id: 1142, name: 'Athens', url: 'http://dbpedia.org/resource/Athens'})-[:IS_PART_OF]->(pl78) +MATCH (pl84:Place {id: 84}) INSERT (pl1201:Place:City {id: 1201, name: 'Antananarivo', url: 'http://dbpedia.org/resource/Antananarivo'})-[:IS_PART_OF]->(pl84) +MATCH (pl84:Place {id: 84}) INSERT (pl1205:Place:City {id: 1205, name: 'Toliara', url: 'http://dbpedia.org/resource/Toliara'})-[:IS_PART_OF]->(pl84) +MATCH (pl96:Place {id: 96}) INSERT (pl1319:Place:City {id: 1319, name: 'Touba', url: 'http://dbpedia.org/resource/Touba'})-[:IS_PART_OF]->(pl96) +MATCH (pl96:Place {id: 96}) INSERT (pl1320:Place:City {id: 1320, name: 'Dakar', url: 'http://dbpedia.org/resource/Dakar'})-[:IS_PART_OF]->(pl96) +MATCH (pl98:Place {id: 98}) INSERT (pl1342:Place:City {id: 1342, name: 'Incheon', url: 'http://dbpedia.org/resource/Incheon'})-[:IS_PART_OF]->(pl98) +MATCH (pl99:Place {id: 99}) INSERT (pl1344:Place:City {id: 1344, name: 'Madrid', url: 'http://dbpedia.org/resource/Madrid'})-[:IS_PART_OF]->(pl99) +MATCH (pl99:Place {id: 99}) INSERT (pl1345:Place:City {id: 1345, name: 'Barcelona', url: 'http://dbpedia.org/resource/Barcelona'})-[:IS_PART_OF]->(pl99) +MATCH (pl105:Place {id: 105}) INSERT (pl1405:Place:City {id: 1405, name: 'Ankara', url: 'http://dbpedia.org/resource/Ankara'})-[:IS_PART_OF]->(pl105) +MATCH (pl105:Place {id: 105}) INSERT (pl1411:Place:City {id: 1411, name: 'Izmir', url: 'http://dbpedia.org/resource/Izmir'})-[:IS_PART_OF]->(pl105) +MATCH (pl76:Place {id: 76}) INSERT (o399:Organisation:Company {id: 399, name: 'Ethiopian_Airlines', url: 'http://dbpedia.org/resource/Ethiopian_Airlines'})-[:IS_LOCATED_IN]->(pl76) +MATCH (pl76:Place {id: 76}) INSERT (o400:Organisation:Company {id: 400, name: 'Trans_Nation_Airways', url: 'http://dbpedia.org/resource/Trans_Nation_Airways'})-[:IS_LOCATED_IN]->(pl76) +MATCH (pl78:Place {id: 78}) INSERT (o498:Organisation:Company {id: 498, name: 'Macedonian_Airlines', url: 'http://dbpedia.org/resource/Macedonian_Airlines'})-[:IS_LOCATED_IN]->(pl78) +MATCH (pl0:Place {id: 0}) INSERT (o538:Organisation:Company {id: 538, name: 'Jet_Airways', url: 'http://dbpedia.org/resource/Jet_Airways'})-[:IS_LOCATED_IN]->(pl0) +MATCH (pl0:Place {id: 0}) INSERT (o545:Organisation:Company {id: 545, name: 'Jagson_Airlines', url: 'http://dbpedia.org/resource/Jagson_Airlines'})-[:IS_LOCATED_IN]->(pl0) +MATCH (pl0:Place {id: 0}) INSERT (o553:Organisation:Company {id: 553, name: 'Deccan_360', url: 'http://dbpedia.org/resource/Deccan_360'})-[:IS_LOCATED_IN]->(pl0) +MATCH (pl30:Place {id: 30}) INSERT (o861:Organisation:Company {id: 861, name: 'Nicaragüense_de_Aviación', url: 'http://dbpedia.org/resource/Nicaragüense_de_Aviación'})-[:IS_LOCATED_IN]->(pl30) +MATCH (pl54:Place {id: 54}) INSERT (o887:Organisation:Company {id: 887, name: 'Pakistan_International_Airlines', url: 'http://dbpedia.org/resource/Pakistan_International_Airlines'})-[:IS_LOCATED_IN]->(pl54) +MATCH (pl54:Place {id: 54}) INSERT (o888:Organisation:Company {id: 888, name: 'Aero_Asia_International', url: 'http://dbpedia.org/resource/Aero_Asia_International'})-[:IS_LOCATED_IN]->(pl54) +MATCH (pl54:Place {id: 54}) INSERT (o892:Organisation:Company {id: 892, name: 'Safe_Air', url: 'http://dbpedia.org/resource/Safe_Air'})-[:IS_LOCATED_IN]->(pl54) +MATCH (pl55:Place {id: 55}) INSERT (o955:Organisation:Company {id: 955, name: 'Pacificair', url: 'http://dbpedia.org/resource/Pacificair'})-[:IS_LOCATED_IN]->(pl55) +MATCH (pl55:Place {id: 55}) INSERT (o956:Organisation:Company {id: 956, name: 'South_East_Asian_Airlines', url: 'http://dbpedia.org/resource/South_East_Asian_Airlines'})-[:IS_LOCATED_IN]->(pl55) +MATCH (pl55:Place {id: 55}) INSERT (o957:Organisation:Company {id: 957, name: 'Interisland_Airlines', url: 'http://dbpedia.org/resource/Interisland_Airlines'})-[:IS_LOCATED_IN]->(pl55) +MATCH (pl55:Place {id: 55}) INSERT (o958:Organisation:Company {id: 958, name: 'Filipinas_Orient_Airways', url: 'http://dbpedia.org/resource/Filipinas_Orient_Airways'})-[:IS_LOCATED_IN]->(pl55) +MATCH (pl55:Place {id: 55}) INSERT (o965:Organisation:Company {id: 965, name: 'Aerolift_Philippines', url: 'http://dbpedia.org/resource/Aerolift_Philippines'})-[:IS_LOCATED_IN]->(pl55) +MATCH (pl96:Place {id: 96}) INSERT (o1117:Organisation:Company {id: 1117, name: 'Air_Sénégal_International', url: 'http://dbpedia.org/resource/Air_Sénégal_International'})-[:IS_LOCATED_IN]->(pl96) +MATCH (pl96:Place {id: 96}) INSERT (o1118:Organisation:Company {id: 1118, name: 'Groupement_Aérien_Sénégalais', url: 'http://dbpedia.org/resource/Groupement_Aérien_Sénégalais'})-[:IS_LOCATED_IN]->(pl96) +MATCH (pl98:Place {id: 98}) INSERT (o1176:Organisation:Company {id: 1176, name: 'Jin_Air', url: 'http://dbpedia.org/resource/Jin_Air'})-[:IS_LOCATED_IN]->(pl98) +MATCH (pl98:Place {id: 98}) INSERT (o1178:Organisation:Company {id: 1178, name: 'Yeongnam_Air', url: 'http://dbpedia.org/resource/Yeongnam_Air'})-[:IS_LOCATED_IN]->(pl98) +MATCH (pl105:Place {id: 105}) INSERT (o1327:Organisation:Company {id: 1327, name: 'Air_Anatolia', url: 'http://dbpedia.org/resource/Air_Anatolia'})-[:IS_LOCATED_IN]->(pl105) +MATCH (pl105:Place {id: 105}) INSERT (o1349:Organisation:Company {id: 1349, name: 'Tailwind_Airlines', url: 'http://dbpedia.org/resource/Tailwind_Airlines'})-[:IS_LOCATED_IN]->(pl105) +MATCH (pl469:Place {id: 469}) INSERT (o2213:Organisation:University {id: 2213, name: 'Anhui_University_of_Science_and_Technology', url: 'http://dbpedia.org/resource/Anhui_University_of_Science_and_Technology'})-[:IS_LOCATED_IN]->(pl469) +MATCH (pl1024:Place {id: 1024}) INSERT (o2561:Organisation:University {id: 2561, name: 'St_Mary\'s_Hospital_Medical_School', url: 'http://dbpedia.org/resource/St_Mary\'s_Hospital_Medical_School'})-[:IS_LOCATED_IN]->(pl1024) +MATCH (pl1121:Place {id: 1121}) INSERT (o2649:Organisation:University {id: 2649, name: 'Wollega_University', url: 'http://dbpedia.org/resource/Wollega_University'})-[:IS_LOCATED_IN]->(pl1121) +MATCH (pl1142:Place {id: 1142}) INSERT (o2941:Organisation:University {id: 2941, name: 'National_and_Kapodistrian_University_of_Athens', url: 'http://dbpedia.org/resource/National_and_Kapodistrian_University_of_Athens'})-[:IS_LOCATED_IN]->(pl1142) +MATCH (pl176:Place {id: 176}) INSERT (o3010:Organisation:University {id: 3010, name: 'The_Oxford_Educational_Institutions', url: 'http://dbpedia.org/resource/The_Oxford_Educational_Institutions'})-[:IS_LOCATED_IN]->(pl176) +MATCH (pl1201:Place {id: 1201}) INSERT (o4991:Organisation:University {id: 4991, name: 'Madagascar_Institute_of_Political_Studies', url: 'http://dbpedia.org/resource/Madagascar_Institute_of_Political_Studies'})-[:IS_LOCATED_IN]->(pl1201) +MATCH (pl732:Place {id: 732}) INSERT (o5040:Organisation:University {id: 5040, name: 'Instituto_Tecnológico_de_Puebla', url: 'http://dbpedia.org/resource/Instituto_Tecnológico_de_Puebla'})-[:IS_LOCATED_IN]->(pl732) +MATCH (pl771:Place {id: 771}) INSERT (o5268:Organisation:University {id: 5268, name: 'Ghulam_Ishaq_Khan_Institute_of_Engineering_Sciences_and_Technology', url: 'http://dbpedia.org/resource/Ghulam_Ishaq_Khan_Institute_of_Engineering_Sciences_and_Technology'})-[:IS_LOCATED_IN]->(pl771) +MATCH (pl801:Place {id: 801}) INSERT (o5522:Organisation:University {id: 5522, name: 'College_of_Technological_Sciences–Cebu', url: 'http://dbpedia.org/resource/College_of_Technological_Sciences–Cebu'})-[:IS_LOCATED_IN]->(pl801) +MATCH (pl1320:Place {id: 1320}) INSERT (o6148:Organisation:University {id: 6148, name: 'Cheikh_Anta_Diop_University', url: 'http://dbpedia.org/resource/Cheikh_Anta_Diop_University'})-[:IS_LOCATED_IN]->(pl1320) +MATCH (pl1344:Place {id: 1344}) INSERT (o6302:Organisation:University {id: 6302, name: 'Autonomous_University_of_Madrid', url: 'http://dbpedia.org/resource/Autonomous_University_of_Madrid'})-[:IS_LOCATED_IN]->(pl1344) +MATCH (pl1405:Place {id: 1405}) INSERT (o6557:Organisation:University {id: 6557, name: 'Bilkent_University_Faculty_of_Law', url: 'http://dbpedia.org/resource/Bilkent_University_Faculty_of_Law'})-[:IS_LOCATED_IN]->(pl1405) +INSERT (tc0:TagClass {id: 0, name: 'Thing', url: 'http://www.w3.org/2002/07/owl#Thing'}) +MATCH (tc0:TagClass {id: 0}) INSERT (tc188:TagClass {id: 188, name: 'Work', url: 'http://dbpedia.org/ontology/Work'})-[:IS_SUBCLASS_OF]->(tc0) +MATCH (tc0:TagClass {id: 0}) INSERT (tc239:TagClass {id: 239, name: 'Agent', url: 'http://dbpedia.org/ontology/Agent'})-[:IS_SUBCLASS_OF]->(tc0) +MATCH (tc0:TagClass {id: 0}) INSERT (tc303:TagClass {id: 303, name: 'Place', url: 'http://dbpedia.org/ontology/Place'})-[:IS_SUBCLASS_OF]->(tc0) +MATCH (tc188:TagClass {id: 188}) INSERT (tc123:TagClass {id: 123, name: 'MusicalWork', url: 'http://dbpedia.org/ontology/MusicalWork'})-[:IS_SUBCLASS_OF]->(tc188) +MATCH (tc239:TagClass {id: 239}) INSERT (tc211:TagClass {id: 211, name: 'Person', url: 'http://dbpedia.org/ontology/Person'})-[:IS_SUBCLASS_OF]->(tc239) +MATCH (tc303:TagClass {id: 303}) INSERT (tc318:TagClass {id: 318, name: 'PopulatedPlace', url: 'http://dbpedia.org/ontology/PopulatedPlace'})-[:IS_SUBCLASS_OF]->(tc303) +MATCH (tc318:TagClass {id: 318}) INSERT (tc62:TagClass {id: 62, name: 'Country', url: 'http://dbpedia.org/ontology/Country'})-[:IS_SUBCLASS_OF]->(tc318) +MATCH (tc211:TagClass {id: 211}) INSERT (tc95:TagClass {id: 95, name: 'Politician', url: 'http://dbpedia.org/ontology/Politician'})-[:IS_SUBCLASS_OF]->(tc211) +MATCH (tc211:TagClass {id: 211}) INSERT (tc98:TagClass {id: 98, name: 'Monarch', url: 'http://dbpedia.org/ontology/Monarch'})-[:IS_SUBCLASS_OF]->(tc211) +MATCH (tc211:TagClass {id: 211}) INSERT (tc109:TagClass {id: 109, name: 'Cleric', url: 'http://dbpedia.org/ontology/Cleric'})-[:IS_SUBCLASS_OF]->(tc211) +MATCH (tc211:TagClass {id: 211}) INSERT (tc149:TagClass {id: 149, name: 'Athlete', url: 'http://dbpedia.org/ontology/Athlete'})-[:IS_SUBCLASS_OF]->(tc211) +MATCH (tc211:TagClass {id: 211}) INSERT (tc155:TagClass {id: 155, name: 'Royalty', url: 'http://dbpedia.org/ontology/Royalty'})-[:IS_SUBCLASS_OF]->(tc211) +MATCH (tc123:TagClass {id: 123}) INSERT (tc180:TagClass {id: 180, name: 'Song', url: 'http://dbpedia.org/ontology/Song'})-[:IS_SUBCLASS_OF]->(tc123) +MATCH (tc123:TagClass {id: 123}) INSERT (tc182:TagClass {id: 182, name: 'Album', url: 'http://dbpedia.org/ontology/Album'})-[:IS_SUBCLASS_OF]->(tc123) +MATCH (tc211:TagClass {id: 211}) INSERT (tc250:TagClass {id: 250, name: 'Artist', url: 'http://dbpedia.org/ontology/Artist'})-[:IS_SUBCLASS_OF]->(tc211) +MATCH (tc211:TagClass {id: 211}) INSERT (tc279:TagClass {id: 279, name: 'Scientist', url: 'http://dbpedia.org/ontology/Scientist'})-[:IS_SUBCLASS_OF]->(tc211) +MATCH (tc211:TagClass {id: 211}) INSERT (tc287:TagClass {id: 287, name: 'Criminal', url: 'http://dbpedia.org/ontology/Criminal'})-[:IS_SUBCLASS_OF]->(tc211) +MATCH (tc211:TagClass {id: 211}) INSERT (tc295:TagClass {id: 295, name: 'FictionalCharacter', url: 'http://dbpedia.org/ontology/FictionalCharacter'})-[:IS_SUBCLASS_OF]->(tc211) +MATCH (tc123:TagClass {id: 123}) INSERT (tc342:TagClass {id: 342, name: 'Single', url: 'http://dbpedia.org/ontology/Single'})-[:IS_SUBCLASS_OF]->(tc123) +MATCH (tc211:TagClass {id: 211}) INSERT (tc349:TagClass {id: 349, name: 'OfficeHolder', url: 'http://dbpedia.org/ontology/OfficeHolder'})-[:IS_SUBCLASS_OF]->(tc211) +MATCH (tc95:TagClass {id: 95}) INSERT (tc57:TagClass {id: 57, name: 'President', url: 'http://dbpedia.org/ontology/President'})-[:IS_SUBCLASS_OF]->(tc95) +MATCH (tc149:TagClass {id: 149}) INSERT (tc59:TagClass {id: 59, name: 'TennisPlayer', url: 'http://dbpedia.org/ontology/TennisPlayer'})-[:IS_SUBCLASS_OF]->(tc149) +MATCH (tc250:TagClass {id: 250}) INSERT (tc88:TagClass {id: 88, name: 'Writer', url: 'http://dbpedia.org/ontology/Writer'})-[:IS_SUBCLASS_OF]->(tc250) +MATCH (tc250:TagClass {id: 250}) INSERT (tc115:TagClass {id: 115, name: 'MusicalArtist', url: 'http://dbpedia.org/ontology/MusicalArtist'})-[:IS_SUBCLASS_OF]->(tc250) +MATCH (tc109:TagClass {id: 109}) INSERT (tc193:TagClass {id: 193, name: 'Saint', url: 'http://dbpedia.org/ontology/Saint'})-[:IS_SUBCLASS_OF]->(tc109) +MATCH (tc109:TagClass {id: 109}) INSERT (tc332:TagClass {id: 332, name: 'ChristianBishop', url: 'http://dbpedia.org/ontology/ChristianBishop'})-[:IS_SUBCLASS_OF]->(tc109) +MATCH (tc155:TagClass {id: 155}) INSERT (tc336:TagClass {id: 336, name: 'BritishRoyalty', url: 'http://dbpedia.org/ontology/BritishRoyalty'})-[:IS_SUBCLASS_OF]->(tc155) +MATCH (tc349:TagClass {id: 349}) INSERT (t0:Tag {id: 0, name: 'Hamid_Karzai', url: 'http://dbpedia.org/resource/Hamid_Karzai'})-[:HAS_TYPE]->(tc349) +MATCH (tc211:TagClass {id: 211}) INSERT (t26:Tag {id: 26, name: 'Daniel_Barenboim', url: 'http://dbpedia.org/resource/Daniel_Barenboim'})-[:HAS_TYPE]->(tc211) +MATCH (tc349:TagClass {id: 349}) INSERT (t61:Tag {id: 61, name: 'Kevin_Rudd', url: 'http://dbpedia.org/resource/Kevin_Rudd'})-[:HAS_TYPE]->(tc349) +MATCH (tc349:TagClass {id: 349}) INSERT (t273:Tag {id: 273, name: 'Aung_San_Suu_Kyi', url: 'http://dbpedia.org/resource/Aung_San_Suu_Kyi'})-[:HAS_TYPE]->(tc349) +MATCH (tc349:TagClass {id: 349}) INSERT (t294:Tag {id: 294, name: 'William_Lyon_Mackenzie_King', url: 'http://dbpedia.org/resource/William_Lyon_Mackenzie_King'})-[:HAS_TYPE]->(tc349) +MATCH (tc88:TagClass {id: 88}) INSERT (t573:Tag {id: 573, name: 'Victor_Hugo', url: 'http://dbpedia.org/resource/Victor_Hugo'})-[:HAS_TYPE]->(tc88) +MATCH (tc193:TagClass {id: 193}) INSERT (t579:Tag {id: 579, name: 'Joan_of_Arc', url: 'http://dbpedia.org/resource/Joan_of_Arc'})-[:HAS_TYPE]->(tc193) +MATCH (tc279:TagClass {id: 279}) INSERT (t596:Tag {id: 596, name: 'Jean-Baptiste_Lamarck', url: 'http://dbpedia.org/resource/Jean-Baptiste_Lamarck'})-[:HAS_TYPE]->(tc279) +MATCH (tc59:TagClass {id: 59}) INSERT (t1169:Tag {id: 1169, name: 'Dudi_Sela', url: 'http://dbpedia.org/resource/Dudi_Sela'})-[:HAS_TYPE]->(tc59) +MATCH (tc332:TagClass {id: 332}) INSERT (t1174:Tag {id: 1174, name: 'Pope_Paul_VI', url: 'http://dbpedia.org/resource/Pope_Paul_VI'})-[:HAS_TYPE]->(tc332) +MATCH (tc250:TagClass {id: 250}) INSERT (t1176:Tag {id: 1176, name: 'Leonardo_da_Vinci', url: 'http://dbpedia.org/resource/Leonardo_da_Vinci'})-[:HAS_TYPE]->(tc250) +MATCH (tc332:TagClass {id: 332}) INSERT (t1185:Tag {id: 1185, name: 'Pope_Leo_XIII', url: 'http://dbpedia.org/resource/Pope_Leo_XIII'})-[:HAS_TYPE]->(tc332) +MATCH (tc88:TagClass {id: 88}) INSERT (t1201:Tag {id: 1201, name: 'Horace', url: 'http://dbpedia.org/resource/Horace'})-[:HAS_TYPE]->(tc88) +MATCH (tc115:TagClass {id: 115}) INSERT (t1203:Tag {id: 1203, name: 'Ennio_Morricone', url: 'http://dbpedia.org/resource/Ennio_Morricone'})-[:HAS_TYPE]->(tc115) +MATCH (tc211:TagClass {id: 211}) INSERT (t1410:Tag {id: 1410, name: 'Muammar_Gaddafi', url: 'http://dbpedia.org/resource/Muammar_Gaddafi'})-[:HAS_TYPE]->(tc211) +MATCH (tc115:TagClass {id: 115}) INSERT (t1419:Tag {id: 1419, name: 'Guy_Sebastian', url: 'http://dbpedia.org/resource/Guy_Sebastian'})-[:HAS_TYPE]->(tc115) +MATCH (tc57:TagClass {id: 57}) INSERT (t1526:Tag {id: 1526, name: 'Manuel_Noriega', url: 'http://dbpedia.org/resource/Manuel_Noriega'})-[:HAS_TYPE]->(tc57) +MATCH (tc59:TagClass {id: 59}) INSERT (t1530:Tag {id: 1530, name: 'Luis_Horna', url: 'http://dbpedia.org/resource/Luis_Horna'})-[:HAS_TYPE]->(tc59) +MATCH (tc57:TagClass {id: 57}) INSERT (t1538:Tag {id: 1538, name: 'Emilio_Aguinaldo', url: 'http://dbpedia.org/resource/Emilio_Aguinaldo'})-[:HAS_TYPE]->(tc57) +MATCH (tc88:TagClass {id: 88}) INSERT (t1614:Tag {id: 1614, name: 'Oscar_Wilde', url: 'http://dbpedia.org/resource/Oscar_Wilde'})-[:HAS_TYPE]->(tc88) +MATCH (tc57:TagClass {id: 57}) INSERT (t1671:Tag {id: 1671, name: 'Mikhail_Gorbachev', url: 'http://dbpedia.org/resource/Mikhail_Gorbachev'})-[:HAS_TYPE]->(tc57) +MATCH (tc115:TagClass {id: 115}) INSERT (t1761:Tag {id: 1761, name: 'Enrique_Iglesias', url: 'http://dbpedia.org/resource/Enrique_Iglesias'})-[:HAS_TYPE]->(tc115) +MATCH (tc349:TagClass {id: 349}) INSERT (t1987:Tag {id: 1987, name: 'Winston_Churchill', url: 'http://dbpedia.org/resource/Winston_Churchill'})-[:HAS_TYPE]->(tc349) +MATCH (tc349:TagClass {id: 349}) INSERT (t2005:Tag {id: 2005, name: 'Arthur_Wellesley,_1st_Duke_of_Wellington', url: 'http://dbpedia.org/resource/Arthur_Wellesley,_1st_Duke_of_Wellington'})-[:HAS_TYPE]->(tc349) +MATCH (tc349:TagClass {id: 349}) INSERT (t2029:Tag {id: 2029, name: 'William_Ewart_Gladstone', url: 'http://dbpedia.org/resource/William_Ewart_Gladstone'})-[:HAS_TYPE]->(tc349) +MATCH (tc336:TagClass {id: 336}) INSERT (t2044:Tag {id: 2044, name: 'Anne,_Queen_of_Great_Britain', url: 'http://dbpedia.org/resource/Anne,_Queen_of_Great_Britain'})-[:HAS_TYPE]->(tc336) +MATCH (tc88:TagClass {id: 88}) INSERT (t2059:Tag {id: 2059, name: 'William_Wordsworth', url: 'http://dbpedia.org/resource/William_Wordsworth'})-[:HAS_TYPE]->(tc88) +MATCH (tc211:TagClass {id: 211}) INSERT (t2076:Tag {id: 2076, name: 'William_Morris', url: 'http://dbpedia.org/resource/William_Morris'})-[:HAS_TYPE]->(tc211) +MATCH (tc211:TagClass {id: 211}) INSERT (t2084:Tag {id: 2084, name: 'William_Penn', url: 'http://dbpedia.org/resource/William_Penn'})-[:HAS_TYPE]->(tc211) +MATCH (tc211:TagClass {id: 211}) INSERT (t2114:Tag {id: 2114, name: 'Christopher_Lee', url: 'http://dbpedia.org/resource/Christopher_Lee'})-[:HAS_TYPE]->(tc211) +MATCH (tc88:TagClass {id: 88}) INSERT (t2120:Tag {id: 2120, name: 'Thomas_Hardy', url: 'http://dbpedia.org/resource/Thomas_Hardy'})-[:HAS_TYPE]->(tc88) +MATCH (tc349:TagClass {id: 349}) INSERT (t2795:Tag {id: 2795, name: 'Harry_S._Truman', url: 'http://dbpedia.org/resource/Harry_S._Truman'})-[:HAS_TYPE]->(tc349) +MATCH (tc349:TagClass {id: 349}) INSERT (t2797:Tag {id: 2797, name: 'Thomas_Jefferson', url: 'http://dbpedia.org/resource/Thomas_Jefferson'})-[:HAS_TYPE]->(tc349) +MATCH (tc115:TagClass {id: 115}) INSERT (t2798:Tag {id: 2798, name: 'Stevie_Wonder', url: 'http://dbpedia.org/resource/Stevie_Wonder'})-[:HAS_TYPE]->(tc115) +MATCH (tc349:TagClass {id: 349}) INSERT (t2805:Tag {id: 2805, name: 'Woodrow_Wilson', url: 'http://dbpedia.org/resource/Woodrow_Wilson'})-[:HAS_TYPE]->(tc349) +MATCH (tc115:TagClass {id: 115}) INSERT (t2834:Tag {id: 2834, name: 'Frank_Zappa', url: 'http://dbpedia.org/resource/Frank_Zappa'})-[:HAS_TYPE]->(tc115) +MATCH (tc211:TagClass {id: 211}) INSERT (t2841:Tag {id: 2841, name: 'Martin_Scorsese', url: 'http://dbpedia.org/resource/Martin_Scorsese'})-[:HAS_TYPE]->(tc211) +MATCH (tc115:TagClass {id: 115}) INSERT (t2848:Tag {id: 2848, name: 'Diana_Ross', url: 'http://dbpedia.org/resource/Diana_Ross'})-[:HAS_TYPE]->(tc115) +MATCH (tc115:TagClass {id: 115}) INSERT (t2855:Tag {id: 2855, name: 'Tina_Turner', url: 'http://dbpedia.org/resource/Tina_Turner'})-[:HAS_TYPE]->(tc115) +MATCH (tc349:TagClass {id: 349}) INSERT (t2870:Tag {id: 2870, name: 'John_Adams', url: 'http://dbpedia.org/resource/John_Adams'})-[:HAS_TYPE]->(tc349) +MATCH (tc115:TagClass {id: 115}) INSERT (t2912:Tag {id: 2912, name: 'LL_Cool_J', url: 'http://dbpedia.org/resource/LL_Cool_J'})-[:HAS_TYPE]->(tc115) +MATCH (tc349:TagClass {id: 349}) INSERT (t2922:Tag {id: 2922, name: 'Jefferson_Davis', url: 'http://dbpedia.org/resource/Jefferson_Davis'})-[:HAS_TYPE]->(tc349) +MATCH (tc115:TagClass {id: 115}) INSERT (t2945:Tag {id: 2945, name: 'Ne-Yo', url: 'http://dbpedia.org/resource/Ne-Yo'})-[:HAS_TYPE]->(tc115) +MATCH (tc211:TagClass {id: 211}) INSERT (t2950:Tag {id: 2950, name: 'Bill_Gates', url: 'http://dbpedia.org/resource/Bill_Gates'})-[:HAS_TYPE]->(tc211) +MATCH (tc211:TagClass {id: 211}) INSERT (t2973:Tag {id: 2973, name: 'Al_Pacino', url: 'http://dbpedia.org/resource/Al_Pacino'})-[:HAS_TYPE]->(tc211) +MATCH (tc88:TagClass {id: 88}) INSERT (t2992:Tag {id: 2992, name: 'Philip_K._Dick', url: 'http://dbpedia.org/resource/Philip_K._Dick'})-[:HAS_TYPE]->(tc88) +MATCH (tc59:TagClass {id: 59}) INSERT (t3009:Tag {id: 3009, name: 'Venus_Williams', url: 'http://dbpedia.org/resource/Venus_Williams'})-[:HAS_TYPE]->(tc59) +MATCH (tc211:TagClass {id: 211}) INSERT (t3075:Tag {id: 3075, name: 'Joss_Whedon', url: 'http://dbpedia.org/resource/Joss_Whedon'})-[:HAS_TYPE]->(tc211) +MATCH (tc287:TagClass {id: 287}) INSERT (t3111:Tag {id: 3111, name: 'Al_Capone', url: 'http://dbpedia.org/resource/Al_Capone'})-[:HAS_TYPE]->(tc287) +MATCH (tc349:TagClass {id: 349}) INSERT (t4854:Tag {id: 4854, name: 'Julia_Gillard', url: 'http://dbpedia.org/resource/Julia_Gillard'})-[:HAS_TYPE]->(tc349) +MATCH (tc182:TagClass {id: 182}) INSERT (t4975:Tag {id: 4975, name: 'Niandra_Lades_and_Usually_Just_a_T-Shirt', url: 'http://dbpedia.org/resource/Niandra_Lades_and_Usually_Just_a_T-Shirt'})-[:HAS_TYPE]->(tc182) +MATCH (tc62:TagClass {id: 62}) INSERT (t5059:Tag {id: 5059, name: 'Djibouti', url: 'http://dbpedia.org/resource/Djibouti'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t5068:Tag {id: 5068, name: 'Paraguay', url: 'http://dbpedia.org/resource/Paraguay'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t5077:Tag {id: 5077, name: 'South_Vietnam', url: 'http://dbpedia.org/resource/South_Vietnam'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t5080:Tag {id: 5080, name: 'Thailand', url: 'http://dbpedia.org/resource/Thailand'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t5100:Tag {id: 5100, name: 'Ecuador', url: 'http://dbpedia.org/resource/Ecuador'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t5106:Tag {id: 5106, name: 'Iceland', url: 'http://dbpedia.org/resource/Iceland'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t5108:Tag {id: 5108, name: 'Lebanon', url: 'http://dbpedia.org/resource/Lebanon'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t5109:Tag {id: 5109, name: 'Liberia', url: 'http://dbpedia.org/resource/Liberia'})-[:HAS_TYPE]->(tc62) +MATCH (tc211:TagClass {id: 211}) INSERT (t5117:Tag {id: 5117, name: 'Robert_Altman', url: 'http://dbpedia.org/resource/Robert_Altman'})-[:HAS_TYPE]->(tc211) +MATCH (tc349:TagClass {id: 349}) INSERT (t5132:Tag {id: 5132, name: 'Alexander_Hamilton', url: 'http://dbpedia.org/resource/Alexander_Hamilton'})-[:HAS_TYPE]->(tc349) +MATCH (tc0:TagClass {id: 0}) INSERT (t5159:Tag {id: 5159, name: 'Georges_Bizet', url: 'http://dbpedia.org/resource/Georges_Bizet'})-[:HAS_TYPE]->(tc0) +MATCH (tc62:TagClass {id: 62}) INSERT (t5167:Tag {id: 5167, name: 'Democratic_Republic_of_the_Congo', url: 'http://dbpedia.org/resource/Democratic_Republic_of_the_Congo'})-[:HAS_TYPE]->(tc62) +MATCH (tc182:TagClass {id: 182}) INSERT (t5276:Tag {id: 5276, name: 'Return_of_Saturn', url: 'http://dbpedia.org/resource/Return_of_Saturn'})-[:HAS_TYPE]->(tc182) +MATCH (tc342:TagClass {id: 342}) INSERT (t5417:Tag {id: 5417, name: 'Hold_It_Against_Me', url: 'http://dbpedia.org/resource/Hold_It_Against_Me'})-[:HAS_TYPE]->(tc342) +MATCH (tc211:TagClass {id: 211}) INSERT (t5418:Tag {id: 5418, name: 'J._P._Morgan', url: 'http://dbpedia.org/resource/J._P._Morgan'})-[:HAS_TYPE]->(tc211) +MATCH (tc115:TagClass {id: 115}) INSERT (t5445:Tag {id: 5445, name: 'Brian_Wilson', url: 'http://dbpedia.org/resource/Brian_Wilson'})-[:HAS_TYPE]->(tc115) +MATCH (tc342:TagClass {id: 342}) INSERT (t5800:Tag {id: 5800, name: 'Waiting_for_the_End', url: 'http://dbpedia.org/resource/Waiting_for_the_End'})-[:HAS_TYPE]->(tc342) +MATCH (tc62:TagClass {id: 62}) INSERT (t6373:Tag {id: 6373, name: 'Guam', url: 'http://dbpedia.org/resource/Guam'})-[:HAS_TYPE]->(tc62) +MATCH (tc115:TagClass {id: 115}) INSERT (t6394:Tag {id: 6394, name: 'George_Harrison', url: 'http://dbpedia.org/resource/George_Harrison'})-[:HAS_TYPE]->(tc115) +MATCH (tc62:TagClass {id: 62}) INSERT (t6402:Tag {id: 6402, name: 'Chile', url: 'http://dbpedia.org/resource/Chile'})-[:HAS_TYPE]->(tc62) +MATCH (tc115:TagClass {id: 115}) INSERT (t6458:Tag {id: 6458, name: 'A._R._Rahman', url: 'http://dbpedia.org/resource/A._R._Rahman'})-[:HAS_TYPE]->(tc115) +MATCH (tc0:TagClass {id: 0}) INSERT (t6940:Tag {id: 6940, name: 'Leonard_Bernstein', url: 'http://dbpedia.org/resource/Leonard_Bernstein'})-[:HAS_TYPE]->(tc0) +MATCH (tc295:TagClass {id: 295}) INSERT (t6959:Tag {id: 6959, name: 'Peter_Pan', url: 'http://dbpedia.org/resource/Peter_Pan'})-[:HAS_TYPE]->(tc295) +MATCH (tc62:TagClass {id: 62}) INSERT (t6969:Tag {id: 6969, name: 'Venezuela', url: 'http://dbpedia.org/resource/Venezuela'})-[:HAS_TYPE]->(tc62) +MATCH (tc182:TagClass {id: 182}) INSERT (t7010:Tag {id: 7010, name: 'Pet_Sounds', url: 'http://dbpedia.org/resource/Pet_Sounds'})-[:HAS_TYPE]->(tc182) +MATCH (tc62:TagClass {id: 62}) INSERT (t7018:Tag {id: 7018, name: 'South_Korea', url: 'http://dbpedia.org/resource/South_Korea'})-[:HAS_TYPE]->(tc62) +MATCH (tc342:TagClass {id: 342}) INSERT (t7106:Tag {id: 7106, name: 'Candy_Shop', url: 'http://dbpedia.org/resource/Candy_Shop'})-[:HAS_TYPE]->(tc342) +MATCH (tc115:TagClass {id: 115}) INSERT (t7538:Tag {id: 7538, name: 'Stephen_Sondheim', url: 'http://dbpedia.org/resource/Stephen_Sondheim'})-[:HAS_TYPE]->(tc115) +MATCH (tc62:TagClass {id: 62}) INSERT (t7555:Tag {id: 7555, name: 'France', url: 'http://dbpedia.org/resource/France'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t7559:Tag {id: 7559, name: 'Israel', url: 'http://dbpedia.org/resource/Israel'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t7566:Tag {id: 7566, name: 'Monaco', url: 'http://dbpedia.org/resource/Monaco'})-[:HAS_TYPE]->(tc62) +MATCH (tc342:TagClass {id: 342}) INSERT (t7658:Tag {id: 7658, name: 'Head_Like_a_Hole', url: 'http://dbpedia.org/resource/Head_Like_a_Hole'})-[:HAS_TYPE]->(tc342) +MATCH (tc342:TagClass {id: 342}) INSERT (t7670:Tag {id: 7670, name: 'The_First_Night', url: 'http://dbpedia.org/resource/The_First_Night'})-[:HAS_TYPE]->(tc342) +MATCH (tc342:TagClass {id: 342}) INSERT (t7919:Tag {id: 7919, name: 'Hot_for_Teacher', url: 'http://dbpedia.org/resource/Hot_for_Teacher'})-[:HAS_TYPE]->(tc342) +MATCH (tc182:TagClass {id: 182}) INSERT (t8115:Tag {id: 8115, name: 'I_Say_I_Say_I_Say', url: 'http://dbpedia.org/resource/I_Say_I_Say_I_Say'})-[:HAS_TYPE]->(tc182) +MATCH (tc62:TagClass {id: 62}) INSERT (t9034:Tag {id: 9034, name: 'Bukovina', url: 'http://dbpedia.org/resource/Bukovina'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t9061:Tag {id: 9061, name: 'Nigeria', url: 'http://dbpedia.org/resource/Nigeria'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t9074:Tag {id: 9074, name: 'Empire_of_Japan', url: 'http://dbpedia.org/resource/Empire_of_Japan'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t9147:Tag {id: 9147, name: 'League_of_Nations', url: 'http://dbpedia.org/resource/League_of_Nations'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t9159:Tag {id: 9159, name: 'German_Empire', url: 'http://dbpedia.org/resource/German_Empire'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t9164:Tag {id: 9164, name: 'Republic_of_the_Congo', url: 'http://dbpedia.org/resource/Republic_of_the_Congo'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t9249:Tag {id: 9249, name: 'Gibraltar', url: 'http://dbpedia.org/resource/Gibraltar'})-[:HAS_TYPE]->(tc62) +MATCH (tc336:TagClass {id: 336}) INSERT (t9261:Tag {id: 9261, name: 'George_III_of_the_United_Kingdom', url: 'http://dbpedia.org/resource/George_III_of_the_United_Kingdom'})-[:HAS_TYPE]->(tc336) +MATCH (tc342:TagClass {id: 342}) INSERT (t9572:Tag {id: 9572, name: 'Bad,_Bad_Leroy_Brown', url: 'http://dbpedia.org/resource/Bad,_Bad_Leroy_Brown'})-[:HAS_TYPE]->(tc342) +MATCH (tc342:TagClass {id: 342}) INSERT (t10144:Tag {id: 10144, name: 'California_King_Bed', url: 'http://dbpedia.org/resource/California_King_Bed'})-[:HAS_TYPE]->(tc342) +MATCH (tc182:TagClass {id: 182}) INSERT (t10526:Tag {id: 10526, name: 'Live_in_São_Paulo', url: 'http://dbpedia.org/resource/Live_in_São_Paulo'})-[:HAS_TYPE]->(tc182) +MATCH (tc182:TagClass {id: 182}) INSERT (t11035:Tag {id: 11035, name: 'Getback', url: 'http://dbpedia.org/resource/Getback'})-[:HAS_TYPE]->(tc182) +MATCH (tc182:TagClass {id: 182}) INSERT (t11269:Tag {id: 11269, name: 'The_Magic_Touch', url: 'http://dbpedia.org/resource/The_Magic_Touch'})-[:HAS_TYPE]->(tc182) +MATCH (tc180:TagClass {id: 180}) INSERT (t11358:Tag {id: 11358, name: 'Chattanooga_Choo_Choo', url: 'http://dbpedia.org/resource/Chattanooga_Choo_Choo'})-[:HAS_TYPE]->(tc180) +MATCH (tc62:TagClass {id: 62}) INSERT (t11543:Tag {id: 11543, name: 'Xiongnu', url: 'http://dbpedia.org/resource/Xiongnu'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t11630:Tag {id: 11630, name: 'Portuguese_Empire', url: 'http://dbpedia.org/resource/Portuguese_Empire'})-[:HAS_TYPE]->(tc62) +MATCH (tc98:TagClass {id: 98}) INSERT (t11644:Tag {id: 11644, name: 'Timur', url: 'http://dbpedia.org/resource/Timur'})-[:HAS_TYPE]->(tc98) +MATCH (tc62:TagClass {id: 62}) INSERT (t11647:Tag {id: 11647, name: 'Abbasid_Caliphate', url: 'http://dbpedia.org/resource/Abbasid_Caliphate'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t11686:Tag {id: 11686, name: 'Sassanid_Empire', url: 'http://dbpedia.org/resource/Sassanid_Empire'})-[:HAS_TYPE]->(tc62) +MATCH (tc62:TagClass {id: 62}) INSERT (t12004:Tag {id: 12004, name: 'Almoravid_dynasty', url: 'http://dbpedia.org/resource/Almoravid_dynasty'})-[:HAS_TYPE]->(tc62) +MATCH (tc182:TagClass {id: 182}) INSERT (t12078:Tag {id: 12078, name: 'Trilogy:_Past_Present_Future', url: 'http://dbpedia.org/resource/Trilogy:_Past_Present_Future'})-[:HAS_TYPE]->(tc182) +MATCH (tc182:TagClass {id: 182}) INSERT (t12742:Tag {id: 12742, name: 'Hot_Streets', url: 'http://dbpedia.org/resource/Hot_Streets'})-[:HAS_TYPE]->(tc182) +MATCH (tc342:TagClass {id: 342}) INSERT (t12763:Tag {id: 12763, name: 'Morning_Has_Broken', url: 'http://dbpedia.org/resource/Morning_Has_Broken'})-[:HAS_TYPE]->(tc342) +MATCH (tc342:TagClass {id: 342}) INSERT (t12950:Tag {id: 12950, name: 'Spill_the_Wine', url: 'http://dbpedia.org/resource/Spill_the_Wine'})-[:HAS_TYPE]->(tc342) +MATCH (tc182:TagClass {id: 182}) INSERT (t13010:Tag {id: 13010, name: 'After_the_Gold_Rush', url: 'http://dbpedia.org/resource/After_the_Gold_Rush'})-[:HAS_TYPE]->(tc182) +MATCH (tc342:TagClass {id: 342}) INSERT (t13191:Tag {id: 13191, name: 'Soul_Makossa', url: 'http://dbpedia.org/resource/Soul_Makossa'})-[:HAS_TYPE]->(tc342) +MATCH (tc182:TagClass {id: 182}) INSERT (t13379:Tag {id: 13379, name: 'Saturday_Nights_&_Sunday_Mornings', url: 'http://dbpedia.org/resource/Saturday_Nights_&_Sunday_Mornings'})-[:HAS_TYPE]->(tc182) +MATCH (o3010:Organisation {id: 3010}), (o538:Organisation {id: 538}), (o545:Organisation {id: 545}), (o553:Organisation {id: 553}), (pl185:Place {id: 185}), (t1410:Tag {id: 1410}), (t1671:Tag {id: 1671}) INSERT (p41:Person {id: 41, firstName: 'John', lastName: 'Kumar', gender: 'male', birthday: 527731200000, creationDate: 1266276257359, locationIP: '27.116.33.147', browserUsed: 'Safari', speaks: ['gu', 'mr', 'en'], email: ['John41@gmail.com', 'John41@jizan.cc', 'John41@yahoo.com', 'John41@zoho.com']})-[:IS_LOCATED_IN]->(pl185), (p41)-[:HAS_INTEREST]->(t1410), (p41)-[:HAS_INTEREST]->(t1671), (p41)-[:STUDY_AT {classYear: 2004}]->(o3010), (p41)-[:WORK_AT {workFrom: 2005}]->(o538), (p41)-[:WORK_AT {workFrom: 2005}]->(o545), (p41)-[:WORK_AT {workFrom: 2006}]->(o553) +MATCH (o1176:Organisation {id: 1176}), (o1178:Organisation {id: 1178}), (o861:Organisation {id: 861}), (pl1342:Place {id: 1342}), (t1538:Tag {id: 1538}) INSERT (p76:Person {id: 76, firstName: 'Jae-Jin', lastName: 'Park', gender: 'male', birthday: 608169600000, creationDate: 1267292326198, locationIP: '27.35.111.48', browserUsed: 'Chrome', speaks: ['en'], email: ['Jae-Jin76@gmail.com', 'Jae-Jin76@oujda.cc']})-[:IS_LOCATED_IN]->(pl1342), (p76)-[:HAS_INTEREST]->(t1538), (p76)-[:WORK_AT {workFrom: 2008}]->(o861), (p76)-[:WORK_AT {workFrom: 2009}]->(o1176), (p76)-[:WORK_AT {workFrom: 2007}]->(o1178) +MATCH (o4991:Organisation {id: 4991}), (pl1205:Place {id: 1205}), (t2084:Tag {id: 2084}) INSERT (p102:Person {id: 102, firstName: 'Philibert', lastName: 'Roindefo', gender: 'female', birthday: 547516800000, creationDate: 1263725040059, locationIP: '41.204.102.171', browserUsed: 'Safari', speaks: ['mg', 'en'], email: ['Philibert102@gmail.com', 'Philibert102@gmx.com']})-[:IS_LOCATED_IN]->(pl1205), (p102)-[:HAS_INTEREST]->(t2084), (p102)-[:STUDY_AT {classYear: 2005}]->(o4991) +MATCH (o2941:Organisation {id: 2941}), (o498:Organisation {id: 498}), (p102:Person {id: 102}), (p41:Person {id: 41}), (pl1142:Place {id: 1142}), (t273:Tag {id: 273}), (t2798:Tag {id: 2798}), (t6394:Tag {id: 6394}) INSERT (p143:Person {id: 143, firstName: 'Maria', lastName: 'Alkaios', gender: 'female', birthday: 410659200000, creationDate: 1262456643976, locationIP: '62.217.119.183', browserUsed: 'Firefox', speaks: ['fr', 'en'], email: ['Maria143@gmail.com']})-[:IS_LOCATED_IN]->(pl1142), (p143)-[:HAS_INTEREST]->(t273), (p143)-[:HAS_INTEREST]->(t2798), (p143)-[:HAS_INTEREST]->(t6394), (p143)-[:STUDY_AT {classYear: 2003}]->(o2941), (p143)-[:WORK_AT {workFrom: 2004}]->(o498), (p41)-[:KNOWS {creationDate: 1267781946984}]->(p143), (p102)-[:KNOWS {creationDate: 1265333297464}]->(p143) +MATCH (o5040:Organisation {id: 5040}), (p76:Person {id: 76}), (pl745:Place {id: 745}) INSERT (p150:Person {id: 150, firstName: 'Alfonso', lastName: 'Alvarez', gender: 'female', birthday: 410227200000, creationDate: 1262602398117, locationIP: '148.240.94.143', browserUsed: 'Firefox', speaks: ['es', 'en'], email: ['Alfonso150@yahoo.com']})-[:IS_LOCATED_IN]->(pl745), (p150)-[:STUDY_AT {classYear: 2002}]->(o5040), (p76)-[:KNOWS {creationDate: 1267480779453}]->(p150) +MATCH (o1117:Organisation {id: 1117}), (o1118:Organisation {id: 1118}), (o6148:Organisation {id: 6148}), (p143:Person {id: 143}), (p150:Person {id: 150}), (pl1319:Place {id: 1319}), (t0:Tag {id: 0}) INSERT (p153:Person {id: 153, firstName: 'Abdala', lastName: 'Ndiaye', gender: 'female', birthday: 345513600000, creationDate: 1266688948654, locationIP: '196.1.98.252', browserUsed: 'Firefox', speaks: ['fr', 'wo', 'en'], email: ['Abdala153@gmail.com']})-[:IS_LOCATED_IN]->(pl1319), (p153)-[:HAS_INTEREST]->(t0), (p153)-[:STUDY_AT {classYear: 1999}]->(o6148), (p153)-[:WORK_AT {workFrom: 2000}]->(o1117), (p153)-[:WORK_AT {workFrom: 2000}]->(o1118), (p143)-[:KNOWS {creationDate: 1267456810473}]->(p153), (p150)-[:KNOWS {creationDate: 1268069961266}]->(p153) +MATCH (o2649:Organisation {id: 2649}), (o399:Organisation {id: 399}), (o400:Organisation {id: 400}), (p102:Person {id: 102}), (p150:Person {id: 150}), (p76:Person {id: 76}), (pl1127:Place {id: 1127}), (t2797:Tag {id: 2797}), (t2848:Tag {id: 2848}), (t294:Tag {id: 294}) INSERT (p228:Person {id: 228, firstName: 'Asher', lastName: 'Mamo', gender: 'female', birthday: 524016000000, creationDate: 1266720721912, locationIP: '213.55.93.153', browserUsed: 'Chrome', speaks: ['ar', 'en'], email: ['Asher228@gmail.com', 'Asher228@yahoo.com', 'Asher228@zoho.com']})-[:IS_LOCATED_IN]->(pl1127), (p228)-[:HAS_INTEREST]->(t294), (p228)-[:HAS_INTEREST]->(t2797), (p228)-[:HAS_INTEREST]->(t2848), (p228)-[:STUDY_AT {classYear: 2005}]->(o2649), (p228)-[:WORK_AT {workFrom: 2005}]->(o399), (p228)-[:WORK_AT {workFrom: 2006}]->(o400), (p76)-[:KNOWS {creationDate: 1267889385714}]->(p228), (p102)-[:KNOWS {creationDate: 1268755084867}]->(p228), (p150)-[:KNOWS {creationDate: 1267126413921}]->(p228) +MATCH (o5522:Organisation {id: 5522}), (o955:Organisation {id: 955}), (o956:Organisation {id: 956}), (o958:Organisation {id: 958}), (p228:Person {id: 228}), (p76:Person {id: 76}), (pl826:Place {id: 826}), (t1201:Tag {id: 1201}), (t2841:Tag {id: 2841}) INSERT (p2199023255712:Person {id: 2199023255712, firstName: 'Aurora', lastName: 'Cruz', gender: 'female', birthday: 589680000000, creationDate: 1271017218227, locationIP: '115.84.173.212', browserUsed: 'Chrome', speaks: ['en'], email: ['Aurora2199023255712@gmail.com']})-[:IS_LOCATED_IN]->(pl826), (p2199023255712)-[:HAS_INTEREST]->(t1201), (p2199023255712)-[:HAS_INTEREST]->(t2841), (p2199023255712)-[:STUDY_AT {classYear: 2006}]->(o5522), (p2199023255712)-[:WORK_AT {workFrom: 2008}]->(o955), (p2199023255712)-[:WORK_AT {workFrom: 2007}]->(o956), (p2199023255712)-[:WORK_AT {workFrom: 2007}]->(o958), (p76)-[:KNOWS {creationDate: 1271645962049}]->(p2199023255712), (p228)-[:KNOWS {creationDate: 1271536640884}]->(p2199023255712) +MATCH (o6302:Organisation {id: 6302}), (p143:Person {id: 143}), (p150:Person {id: 150}), (p153:Person {id: 153}), (p76:Person {id: 76}), (pl1345:Place {id: 1345}), (t1419:Tag {id: 1419}) INSERT (p4398046511333:Person {id: 4398046511333, firstName: 'Rafael', lastName: 'Fernández', gender: 'female', birthday: 334540800000, creationDate: 1275959471971, locationIP: '31.24.152.190', browserUsed: 'Chrome', speaks: ['es', 'en'], email: ['Rafael4398046511333@gmail.com', 'Rafael4398046511333@yahoo.com', 'Rafael4398046511333@zoho.com']})-[:IS_LOCATED_IN]->(pl1345), (p4398046511333)-[:HAS_INTEREST]->(t1419), (p4398046511333)-[:STUDY_AT {classYear: 2002}]->(o6302), (p76)-[:KNOWS {creationDate: 1276156139184}]->(p4398046511333), (p143)-[:KNOWS {creationDate: 1278082483463}]->(p4398046511333), (p150)-[:KNOWS {creationDate: 1277029819364}]->(p4398046511333), (p153)-[:KNOWS {creationDate: 1278269640274}]->(p4398046511333) +MATCH (o2213:Organisation {id: 2213}), (p143:Person {id: 143}), (p2199023255712:Person {id: 2199023255712}), (p4398046511333:Person {id: 4398046511333}), (p76:Person {id: 76}), (pl443:Place {id: 443}), (t1987:Tag {id: 1987}) INSERT (p6597069766775:Person {id: 6597069766775, firstName: 'Jie', lastName: 'Yang', gender: 'male', birthday: 426038400000, creationDate: 1280509125052, locationIP: '1.4.5.93', browserUsed: 'Internet Explorer', speaks: ['zh', 'en'], email: ['Jie6597069766775@gmail.com']})-[:IS_LOCATED_IN]->(pl443), (p6597069766775)-[:HAS_INTEREST]->(t1987), (p6597069766775)-[:STUDY_AT {classYear: 2004}]->(o2213), (p76)-[:KNOWS {creationDate: 1282384958210}]->(p6597069766775), (p143)-[:KNOWS {creationDate: 1282543488007}]->(p6597069766775), (p2199023255712)-[:KNOWS {creationDate: 1282282926918}]->(p6597069766775), (p4398046511333)-[:KNOWS {creationDate: 1281036177770}]->(p6597069766775) +MATCH (o2561:Organisation {id: 2561}), (p143:Person {id: 143}), (p150:Person {id: 150}), (p228:Person {id: 228}), (p76:Person {id: 76}), (pl1118:Place {id: 1118}), (t1987:Tag {id: 1987}), (t573:Tag {id: 573}), (t9261:Tag {id: 9261}) INSERT (p8796093022357:Person {id: 8796093022357, firstName: 'Gary', lastName: 'Hill', gender: 'male', birthday: 410918400000, creationDate: 1287361150632, locationIP: '31.12.85.157', browserUsed: 'Chrome', speaks: ['en'], email: ['Gary8796093022357@gmail.com']})-[:IS_LOCATED_IN]->(pl1118), (p8796093022357)-[:HAS_INTEREST]->(t573), (p8796093022357)-[:HAS_INTEREST]->(t1987), (p8796093022357)-[:HAS_INTEREST]->(t9261), (p8796093022357)-[:STUDY_AT {classYear: 2005}]->(o2561), (p76)-[:KNOWS {creationDate: 1288699821804}]->(p8796093022357), (p143)-[:KNOWS {creationDate: 1289395343429}]->(p8796093022357), (p150)-[:KNOWS {creationDate: 1287899871416}]->(p8796093022357), (p228)-[:KNOWS {creationDate: 1288580721183}]->(p8796093022357) +MATCH (o1327:Organisation {id: 1327}), (o1349:Organisation {id: 1349}), (o6557:Organisation {id: 6557}), (p143:Person {id: 143}), (p4398046511333:Person {id: 4398046511333}), (p6597069766775:Person {id: 6597069766775}), (p76:Person {id: 76}), (pl1411:Place {id: 1411}) INSERT (p8796093022390:Person {id: 8796093022390, firstName: 'Abdullah', lastName: 'Koksal', gender: 'male', birthday: 605404800000, creationDate: 1282578456825, locationIP: '46.31.114.181', browserUsed: 'Chrome', speaks: ['tr', 'ku', 'en'], email: ['Abdullah8796093022390@gmail.com', 'Abdullah8796093022390@zoho.com']})-[:IS_LOCATED_IN]->(pl1411), (p8796093022390)-[:STUDY_AT {classYear: 2008}]->(o6557), (p8796093022390)-[:WORK_AT {workFrom: 2009}]->(o1327), (p8796093022390)-[:WORK_AT {workFrom: 2008}]->(o1349), (p76)-[:KNOWS {creationDate: 1283826740087}]->(p8796093022390), (p143)-[:KNOWS {creationDate: 1283553972986}]->(p8796093022390), (p4398046511333)-[:KNOWS {creationDate: 1284251676299}]->(p8796093022390), (p6597069766775)-[:KNOWS {creationDate: 1282784928244}]->(p8796093022390) +MATCH (o5268:Organisation {id: 5268}), (o887:Organisation {id: 887}), (o888:Organisation {id: 888}), (o892:Organisation {id: 892}), (p150:Person {id: 150}), (p4398046511333:Person {id: 4398046511333}), (p76:Person {id: 76}), (pl780:Place {id: 780}), (t2798:Tag {id: 2798}), (t2848:Tag {id: 2848}), (t2945:Tag {id: 2945}), (t573:Tag {id: 573}) INSERT (p10995116277918:Person {id: 10995116277918, firstName: 'Javed', lastName: 'Khan', gender: 'male', birthday: 524102400000, creationDate: 1288977435923, locationIP: '42.83.84.215', browserUsed: 'Chrome', speaks: ['en'], email: ['Javed10995116277918@gmail.com', 'Javed10995116277918@hotmail.com']})-[:IS_LOCATED_IN]->(pl780), (p10995116277918)-[:HAS_INTEREST]->(t573), (p10995116277918)-[:HAS_INTEREST]->(t2798), (p10995116277918)-[:HAS_INTEREST]->(t2848), (p10995116277918)-[:HAS_INTEREST]->(t2945), (p10995116277918)-[:STUDY_AT {classYear: 2006}]->(o5268), (p10995116277918)-[:WORK_AT {workFrom: 2008}]->(o887), (p10995116277918)-[:WORK_AT {workFrom: 2006}]->(o888), (p10995116277918)-[:WORK_AT {workFrom: 2006}]->(o892), (p76)-[:KNOWS {creationDate: 1289299684330}]->(p10995116277918), (p150)-[:KNOWS {creationDate: 1290531879144}]->(p10995116277918), (p4398046511333)-[:KNOWS {creationDate: 1290670426514}]->(p10995116277918) +MATCH (p102:Person {id: 102}), (p143:Person {id: 143}), (p228:Person {id: 228}), (t2084:Tag {id: 2084}) INSERT (f200:Forum {id: 200, title: 'Wall of Philibert Roindefo', creationDate: 1263725050059})-[:HAS_MODERATOR]->(p102), (f200)-[:HAS_MEMBER {joinDate: 1265333307464}]->(p143), (f200)-[:HAS_MEMBER {joinDate: 1268755094867}]->(p228), (f200)-[:HAS_TAG]->(t2084) +MATCH (p102:Person {id: 102}), (p143:Person {id: 143}), (p153:Person {id: 153}), (p41:Person {id: 41}), (p4398046511333:Person {id: 4398046511333}), (p6597069766775:Person {id: 6597069766775}), (p8796093022357:Person {id: 8796093022357}), (p8796093022390:Person {id: 8796093022390}), (t273:Tag {id: 273}), (t2798:Tag {id: 2798}), (t6394:Tag {id: 6394}) INSERT (f567:Forum {id: 567, title: 'Wall of Maria Alkaios', creationDate: 1262456653976})-[:HAS_MODERATOR]->(p143), (f567)-[:HAS_MEMBER {joinDate: 1267781956984}]->(p41), (f567)-[:HAS_MEMBER {joinDate: 1265333307464}]->(p102), (f567)-[:HAS_MEMBER {joinDate: 1267456820473}]->(p153), (f567)-[:HAS_MEMBER {joinDate: 1278082493463}]->(p4398046511333), (f567)-[:HAS_MEMBER {joinDate: 1282543498007}]->(p6597069766775), (f567)-[:HAS_MEMBER {joinDate: 1289395353429}]->(p8796093022357), (f567)-[:HAS_MEMBER {joinDate: 1283553982986}]->(p8796093022390), (f567)-[:HAS_TAG]->(t273), (f567)-[:HAS_TAG]->(t2798), (f567)-[:HAS_TAG]->(t6394) +MATCH (p143:Person {id: 143}), (p41:Person {id: 41}), (t1410:Tag {id: 1410}), (t1671:Tag {id: 1671}) INSERT (f689:Forum {id: 689, title: 'Wall of John Kumar', creationDate: 1266276267359})-[:HAS_MODERATOR]->(p41), (f689)-[:HAS_MEMBER {joinDate: 1267781956984}]->(p143), (f689)-[:HAS_TAG]->(t1410), (f689)-[:HAS_TAG]->(t1671) +MATCH (p10995116277918:Person {id: 10995116277918}), (p150:Person {id: 150}), (p2199023255712:Person {id: 2199023255712}), (p228:Person {id: 228}), (p4398046511333:Person {id: 4398046511333}), (p6597069766775:Person {id: 6597069766775}), (p76:Person {id: 76}), (p8796093022357:Person {id: 8796093022357}), (p8796093022390:Person {id: 8796093022390}), (t1538:Tag {id: 1538}) INSERT (f767:Forum {id: 767, title: 'Wall of Jae-Jin Park', creationDate: 1267292336198})-[:HAS_MODERATOR]->(p76), (f767)-[:HAS_MEMBER {joinDate: 1267480789453}]->(p150), (f767)-[:HAS_MEMBER {joinDate: 1267889395714}]->(p228), (f767)-[:HAS_MEMBER {joinDate: 1271645972049}]->(p2199023255712), (f767)-[:HAS_MEMBER {joinDate: 1276156149184}]->(p4398046511333), (f767)-[:HAS_MEMBER {joinDate: 1282384968210}]->(p6597069766775), (f767)-[:HAS_MEMBER {joinDate: 1288699831804}]->(p8796093022357), (f767)-[:HAS_MEMBER {joinDate: 1283826750087}]->(p8796093022390), (f767)-[:HAS_MEMBER {joinDate: 1289299694330}]->(p10995116277918), (f767)-[:HAS_TAG]->(t1538) +MATCH (p102:Person {id: 102}), (p150:Person {id: 150}), (p2199023255712:Person {id: 2199023255712}), (p228:Person {id: 228}), (p76:Person {id: 76}), (p8796093022357:Person {id: 8796093022357}), (t2797:Tag {id: 2797}), (t2848:Tag {id: 2848}), (t294:Tag {id: 294}) INSERT (f872:Forum {id: 872, title: 'Wall of Asher Mamo', creationDate: 1266720731912})-[:HAS_MODERATOR]->(p228), (f872)-[:HAS_MEMBER {joinDate: 1267889395714}]->(p76), (f872)-[:HAS_MEMBER {joinDate: 1268755094867}]->(p102), (f872)-[:HAS_MEMBER {joinDate: 1267126423921}]->(p150), (f872)-[:HAS_MEMBER {joinDate: 1271536650884}]->(p2199023255712), (f872)-[:HAS_MEMBER {joinDate: 1288580731183}]->(p8796093022357), (f872)-[:HAS_TAG]->(t294), (f872)-[:HAS_TAG]->(t2797), (f872)-[:HAS_TAG]->(t2848) +MATCH (p2199023255712:Person {id: 2199023255712}), (p228:Person {id: 228}), (p76:Person {id: 76}), (t294:Tag {id: 294}) INSERT (f882:Forum {id: 882, title: 'Album 9 of Asher Mamo', creationDate: 1267290583839})-[:HAS_MODERATOR]->(p228), (f882)-[:HAS_MEMBER {joinDate: 1272266253286}]->(p76), (f882)-[:HAS_MEMBER {joinDate: 1289474312288}]->(p2199023255712), (f882)-[:HAS_TAG]->(t294) +MATCH (p10995116277918:Person {id: 10995116277918}), (p150:Person {id: 150}), (p153:Person {id: 153}), (p228:Person {id: 228}), (p4398046511333:Person {id: 4398046511333}), (p76:Person {id: 76}), (p8796093022357:Person {id: 8796093022357}) INSERT (f900:Forum {id: 900, title: 'Wall of Alfonso Alvarez', creationDate: 1262602408117})-[:HAS_MODERATOR]->(p150), (f900)-[:HAS_MEMBER {joinDate: 1267480789453}]->(p76), (f900)-[:HAS_MEMBER {joinDate: 1268069971266}]->(p153), (f900)-[:HAS_MEMBER {joinDate: 1267126423921}]->(p228), (f900)-[:HAS_MEMBER {joinDate: 1277029829364}]->(p4398046511333), (f900)-[:HAS_MEMBER {joinDate: 1287899881416}]->(p8796093022357), (f900)-[:HAS_MEMBER {joinDate: 1290531889144}]->(p10995116277918) +MATCH (p143:Person {id: 143}), (p150:Person {id: 150}), (p153:Person {id: 153}), (p4398046511333:Person {id: 4398046511333}), (t0:Tag {id: 0}) INSERT (f913:Forum {id: 913, title: 'Wall of Abdala Ndiaye', creationDate: 1266688958654})-[:HAS_MODERATOR]->(p153), (f913)-[:HAS_MEMBER {joinDate: 1267456820473}]->(p143), (f913)-[:HAS_MEMBER {joinDate: 1268069971266}]->(p150), (f913)-[:HAS_MEMBER {joinDate: 1278269650274}]->(p4398046511333), (f913)-[:HAS_TAG]->(t0) +MATCH (p2199023255712:Person {id: 2199023255712}), (p228:Person {id: 228}), (p6597069766775:Person {id: 6597069766775}), (p76:Person {id: 76}), (t1201:Tag {id: 1201}), (t2841:Tag {id: 2841}) INSERT (f68719476937:Forum {id: 68719476937, title: 'Wall of Aurora Cruz', creationDate: 1271017228227})-[:HAS_MODERATOR]->(p2199023255712), (f68719476937)-[:HAS_MEMBER {joinDate: 1271645972049}]->(p76), (f68719476937)-[:HAS_MEMBER {joinDate: 1271536650884}]->(p228), (f68719476937)-[:HAS_MEMBER {joinDate: 1282282936918}]->(p6597069766775), (f68719476937)-[:HAS_TAG]->(t1201), (f68719476937)-[:HAS_TAG]->(t2841) +MATCH (p2199023255712:Person {id: 2199023255712}), (p228:Person {id: 228}), (p76:Person {id: 76}), (t2797:Tag {id: 2797}) INSERT (f68719477612:Forum {id: 68719477612, title: 'Album 3 of Asher Mamo', creationDate: 1267972822000})-[:HAS_MODERATOR]->(p228), (f68719477612)-[:HAS_MEMBER {joinDate: 1277068031267}]->(p76), (f68719477612)-[:HAS_MEMBER {joinDate: 1287646295226}]->(p2199023255712), (f68719477612)-[:HAS_TAG]->(t2797) +MATCH (p2199023255712:Person {id: 2199023255712}), (p76:Person {id: 76}) INSERT (f137438953677:Forum {id: 137438953677, title: 'Album 1 of Aurora Cruz', creationDate: 1276274297999})-[:HAS_MODERATOR]->(p2199023255712), (f137438953677)-[:HAS_MEMBER {joinDate: 1290191794151}]->(p76) +MATCH (p2199023255712:Person {id: 2199023255712}), (p6597069766775:Person {id: 6597069766775}), (p76:Person {id: 76}) INSERT (f137438953678:Forum {id: 137438953678, title: 'Album 2 of Aurora Cruz', creationDate: 1276553280898})-[:HAS_MODERATOR]->(p2199023255712), (f137438953678)-[:HAS_MEMBER {joinDate: 1287524920093}]->(p76), (f137438953678)-[:HAS_MEMBER {joinDate: 1284501690372}]->(p6597069766775) +MATCH (p10995116277918:Person {id: 10995116277918}), (p143:Person {id: 143}), (p150:Person {id: 150}), (p153:Person {id: 153}), (p4398046511333:Person {id: 4398046511333}), (p6597069766775:Person {id: 6597069766775}), (p76:Person {id: 76}), (p8796093022390:Person {id: 8796093022390}), (t1419:Tag {id: 1419}) INSERT (f137438953769:Forum {id: 137438953769, title: 'Wall of Rafael Fernández', creationDate: 1275959481971})-[:HAS_MODERATOR]->(p4398046511333), (f137438953769)-[:HAS_MEMBER {joinDate: 1276156149184}]->(p76), (f137438953769)-[:HAS_MEMBER {joinDate: 1278082493463}]->(p143), (f137438953769)-[:HAS_MEMBER {joinDate: 1277029829364}]->(p150), (f137438953769)-[:HAS_MEMBER {joinDate: 1278269650274}]->(p153), (f137438953769)-[:HAS_MEMBER {joinDate: 1281036187770}]->(p6597069766775), (f137438953769)-[:HAS_MEMBER {joinDate: 1284251686299}]->(p8796093022390), (f137438953769)-[:HAS_MEMBER {joinDate: 1290670436514}]->(p10995116277918), (f137438953769)-[:HAS_TAG]->(t1419) +MATCH (p2199023255712:Person {id: 2199023255712}), (p228:Person {id: 228}), (p6597069766775:Person {id: 6597069766775}) INSERT (f206158430419:Forum {id: 206158430419, title: 'Album 7 of Aurora Cruz', creationDate: 1277597388173})-[:HAS_MODERATOR]->(p2199023255712), (f206158430419)-[:HAS_MEMBER {joinDate: 1282657841030}]->(p228), (f206158430419)-[:HAS_MEMBER {joinDate: 1288201837158}]->(p6597069766775) +MATCH (p143:Person {id: 143}), (p2199023255712:Person {id: 2199023255712}), (p4398046511333:Person {id: 4398046511333}), (p6597069766775:Person {id: 6597069766775}), (p76:Person {id: 76}), (p8796093022390:Person {id: 8796093022390}), (t1987:Tag {id: 1987}) INSERT (f206158430557:Forum {id: 206158430557, title: 'Wall of Jie Yang', creationDate: 1280509135052})-[:HAS_MODERATOR]->(p6597069766775), (f206158430557)-[:HAS_MEMBER {joinDate: 1282384968210}]->(p76), (f206158430557)-[:HAS_MEMBER {joinDate: 1282543498007}]->(p143), (f206158430557)-[:HAS_MEMBER {joinDate: 1282282936918}]->(p2199023255712), (f206158430557)-[:HAS_MEMBER {joinDate: 1281036187770}]->(p4398046511333), (f206158430557)-[:HAS_MEMBER {joinDate: 1282784938244}]->(p8796093022390), (f206158430557)-[:HAS_TAG]->(t1987) +MATCH (p102:Person {id: 102}), (p150:Person {id: 150}), (p228:Person {id: 228}) INSERT (f206158431081:Forum {id: 206158431081, title: 'Album 0 of Asher Mamo', creationDate: 1279208408298})-[:HAS_MODERATOR]->(p228), (f206158431081)-[:HAS_MEMBER {joinDate: 1286080705395}]->(p102), (f206158431081)-[:HAS_MEMBER {joinDate: 1282187173420}]->(p150) +MATCH (p2199023255712:Person {id: 2199023255712}) INSERT (f274877907153:Forum {id: 274877907153, title: 'Album 5 of Aurora Cruz', creationDate: 1286564647418})-[:HAS_MODERATOR]->(p2199023255712) +MATCH (p2199023255712:Person {id: 2199023255712}) INSERT (f343597383890:Forum {id: 343597383890, title: 'Album 6 of Aurora Cruz', creationDate: 1288453273991})-[:HAS_MODERATOR]->(p2199023255712) +MATCH (p41:Person {id: 41}) INSERT (f343597384373:Forum {id: 343597384373, title: 'Album 3 of John Kumar', creationDate: 1287673926664})-[:HAS_MODERATOR]->(p41) +MATCH (p228:Person {id: 228}) INSERT (f343597384554:Forum {id: 343597384554, title: 'Album 1 of Asher Mamo', creationDate: 1290590882265})-[:HAS_MODERATOR]->(p228) +MATCH (p150:Person {id: 150}), (p153:Person {id: 153}), (p228:Person {id: 228}), (p8796093022357:Person {id: 8796093022357}) INSERT (f343597384587:Forum {id: 343597384587, title: 'Album 6 of Alfonso Alvarez', creationDate: 1288654813102})-[:HAS_MODERATOR]->(p150), (f343597384587)-[:HAS_MEMBER {joinDate: 1289272714058}]->(p153), (f343597384587)-[:HAS_MEMBER {joinDate: 1289330192261}]->(p228), (f343597384587)-[:HAS_MEMBER {joinDate: 1289524640765}]->(p8796093022357) +MATCH (p153:Person {id: 153}), (p4398046511333:Person {id: 4398046511333}) INSERT (f343597384602:Forum {id: 343597384602, title: 'Album 8 of Abdala Ndiaye', creationDate: 1288058786120})-[:HAS_MODERATOR]->(p153), (f343597384602)-[:HAS_MEMBER {joinDate: 1289285868911}]->(p4398046511333) +MATCH (f882:Forum {id: 882}), (p228:Person {id: 228}), (pl76:Place {id: 76}) INSERT (m10169:Post:Message {id: 10169, imageFile: 'photo10169.jpg', creationDate: 1267290597839, locationIP: '213.55.93.153', browserUsed: 'Chrome', length: 0})-[:HAS_CREATOR]->(p228), (m10169)-[:IS_LOCATED_IN]->(pl76), (f882)-[:CONTAINER_OF]->(m10169) +MATCH (f882:Forum {id: 882}), (p228:Person {id: 228}), (pl76:Place {id: 76}) INSERT (m10174:Post:Message {id: 10174, imageFile: 'photo10174.jpg', creationDate: 1267290602839, locationIP: '213.55.93.153', browserUsed: 'Chrome', length: 0})-[:HAS_CREATOR]->(p228), (m10174)-[:IS_LOCATED_IN]->(pl76), (f882)-[:CONTAINER_OF]->(m10174) +MATCH (f200:Forum {id: 200}), (p102:Person {id: 102}), (pl84:Place {id: 84}), (t2084:Tag {id: 2084}) INSERT (m68719478399:Post:Message {id: 68719478399, creationDate: 1269848470732, locationIP: '41.204.102.171', browserUsed: 'Safari', language: 'tk', content: 'About William Penn, September 1670) was an English admiral and politician who sat in the House of Commons from 1660 to 1670. He was the ', length: 136})-[:HAS_CREATOR]->(p102), (m68719478399)-[:IS_LOCATED_IN]->(pl84), (m68719478399)-[:HAS_TAG]->(t2084), (f200)-[:CONTAINER_OF]->(m68719478399) +MATCH (f200:Forum {id: 200}), (p102:Person {id: 102}), (pl84:Place {id: 84}), (t1169:Tag {id: 1169}), (t2950:Tag {id: 2950}), (t2973:Tag {id: 2973}), (t5106:Tag {id: 5106}), (t5159:Tag {id: 5159}) INSERT (m68719478438:Post:Message {id: 68719478438, creationDate: 1271801431189, locationIP: '41.204.102.171', browserUsed: 'Safari', language: 'tk', content: 'About Dudi Sela, onal tennis player. About Bill Gates, n he was ranked thirAbout Al Pacino, winn', length: 97})-[:HAS_CREATOR]->(p102), (m68719478438)-[:IS_LOCATED_IN]->(pl84), (m68719478438)-[:HAS_TAG]->(t1169), (m68719478438)-[:HAS_TAG]->(t2950), (m68719478438)-[:HAS_TAG]->(t2973), (m68719478438)-[:HAS_TAG]->(t5106), (m68719478438)-[:HAS_TAG]->(t5159), (f200)-[:CONTAINER_OF]->(m68719478438) +MATCH (f200:Forum {id: 200}), (p102:Person {id: 102}), (pl84:Place {id: 84}), (t1169:Tag {id: 1169}), (t4975:Tag {id: 4975}), (t5167:Tag {id: 5167}), (t6373:Tag {id: 6373}), (t7555:Tag {id: 7555}), (t7919:Tag {id: 7919}), (t9034:Tag {id: 9034}) INSERT (m68719478441:Post:Message {id: 68719478441, creationDate: 1271788336189, locationIP: '41.204.102.171', browserUsed: 'Safari', language: 'tk', content: 'About Dudi Sela, the 2003 FrencAbout Niandra Lades and Usually Just a T-Shirt, -of-consciousneAbout De', length: 103})-[:HAS_CREATOR]->(p102), (m68719478441)-[:IS_LOCATED_IN]->(pl84), (m68719478441)-[:HAS_TAG]->(t1169), (m68719478441)-[:HAS_TAG]->(t4975), (m68719478441)-[:HAS_TAG]->(t5167), (m68719478441)-[:HAS_TAG]->(t6373), (m68719478441)-[:HAS_TAG]->(t7555), (m68719478441)-[:HAS_TAG]->(t7919), (m68719478441)-[:HAS_TAG]->(t9034), (f200)-[:CONTAINER_OF]->(m68719478441) +MATCH (f68719476937:Forum {id: 68719476937}), (p2199023255712:Person {id: 2199023255712}), (pl55:Place {id: 55}), (t1169:Tag {id: 1169}), (t12763:Tag {id: 12763}), (t13191:Tag {id: 13191}), (t2044:Tag {id: 2044}), (t2912:Tag {id: 2912}), (t9572:Tag {id: 9572}) INSERT (m68719478453:Post:Message {id: 68719478453, creationDate: 1271771731189, locationIP: '115.84.173.212', browserUsed: 'Chrome', language: 'uz', content: 'About Dudi Sela, h his doubles partneAbout Anne, Queen of Great Britain, tinued as sole monarAbout LL Cool J, der,', length: 114})-[:HAS_CREATOR]->(p2199023255712), (m68719478453)-[:IS_LOCATED_IN]->(pl55), (m68719478453)-[:HAS_TAG]->(t1169), (m68719478453)-[:HAS_TAG]->(t2044), (m68719478453)-[:HAS_TAG]->(t2912), (m68719478453)-[:HAS_TAG]->(t9572), (m68719478453)-[:HAS_TAG]->(t12763), (m68719478453)-[:HAS_TAG]->(t13191), (f68719476937)-[:CONTAINER_OF]->(m68719478453) +MATCH (f767:Forum {id: 767}), (p76:Person {id: 76}), (pl98:Place {id: 98}), (t1538:Tag {id: 1538}) INSERT (m68719485292:Post:Message {id: 68719485292, creationDate: 1268175368595, locationIP: '27.35.111.48', browserUsed: 'Chrome', language: 'uz', content: 'About Emilio Aguinaldo, 22, 1869 – February 6, 1964) was a Filipino general, politician, ', length: 90})-[:HAS_CREATOR]->(p76), (m68719485292)-[:IS_LOCATED_IN]->(pl98), (m68719485292)-[:HAS_TAG]->(t1538), (f767)-[:CONTAINER_OF]->(m68719485292) +MATCH (f68719477612:Forum {id: 68719477612}), (p228:Person {id: 228}), (pl76:Place {id: 76}) INSERT (m68719486854:Post:Message {id: 68719486854, imageFile: 'photo68719486854.jpg', creationDate: 1267972838000, locationIP: '213.55.93.153', browserUsed: 'Chrome', length: 0})-[:HAS_CREATOR]->(p228), (m68719486854)-[:IS_LOCATED_IN]->(pl76), (f68719477612)-[:CONTAINER_OF]->(m68719486854) +MATCH (f913:Forum {id: 913}), (p153:Person {id: 153}), (pl96:Place {id: 96}), (t11630:Tag {id: 11630}), (t1169:Tag {id: 1169}), (t12004:Tag {id: 12004}), (t3009:Tag {id: 3009}), (t596:Tag {id: 596}), (t9159:Tag {id: 9159}), (t9261:Tag {id: 9261}) INSERT (m68719487300:Post:Message {id: 68719487300, creationDate: 1271748511189, locationIP: '196.1.98.252', browserUsed: 'Firefox', language: 'tk', content: 'About Jean-Baptiste Lamarck, dapted them to local envAbout Dudi Sela, l\'s top men\'s singles plAbout Venus Williams, treak since January 1, 2About German Empire, and i', length: 166})-[:HAS_CREATOR]->(p153), (m68719487300)-[:IS_LOCATED_IN]->(pl96), (m68719487300)-[:HAS_TAG]->(t596), (m68719487300)-[:HAS_TAG]->(t1169), (m68719487300)-[:HAS_TAG]->(t3009), (m68719487300)-[:HAS_TAG]->(t9159), (m68719487300)-[:HAS_TAG]->(t9261), (m68719487300)-[:HAS_TAG]->(t11630), (m68719487300)-[:HAS_TAG]->(t12004), (f913)-[:CONTAINER_OF]->(m68719487300) +MATCH (f913:Forum {id: 913}), (p153:Person {id: 153}), (pl96:Place {id: 96}), (t1169:Tag {id: 1169}), (t1203:Tag {id: 1203}), (t13010:Tag {id: 13010}), (t573:Tag {id: 573}), (t6458:Tag {id: 6458}), (t7010:Tag {id: 7010}) INSERT (m68719487313:Post:Message {id: 68719487313, creationDate: 1271670256189, locationIP: '196.1.98.252', browserUsed: 'Firefox', language: 'tk', content: 'About Victor Hugo, his poetry butAbout Dudi Sela, d Ferrer in strAbout Ennio Morrico', length: 85})-[:HAS_CREATOR]->(p153), (m68719487313)-[:IS_LOCATED_IN]->(pl96), (m68719487313)-[:HAS_TAG]->(t573), (m68719487313)-[:HAS_TAG]->(t1169), (m68719487313)-[:HAS_TAG]->(t1203), (m68719487313)-[:HAS_TAG]->(t6458), (m68719487313)-[:HAS_TAG]->(t7010), (m68719487313)-[:HAS_TAG]->(t13010), (f913)-[:CONTAINER_OF]->(m68719487313) +MATCH (f913:Forum {id: 913}), (p153:Person {id: 153}), (pl96:Place {id: 96}), (t10526:Tag {id: 10526}), (t1169:Tag {id: 1169}), (t2059:Tag {id: 2059}) INSERT (m68719487330:Post:Message {id: 68719487330, creationDate: 1271808766189, locationIP: '196.1.98.252', browserUsed: 'Firefox', language: 'tk', content: 'About Dudi Sela, ently Israel\'s top men\'s singles pAbout William Wordsworth, rature with the 1798 jo', length: 100})-[:HAS_CREATOR]->(p153), (m68719487330)-[:IS_LOCATED_IN]->(pl96), (m68719487330)-[:HAS_TAG]->(t1169), (m68719487330)-[:HAS_TAG]->(t2059), (m68719487330)-[:HAS_TAG]->(t10526), (f913)-[:CONTAINER_OF]->(m68719487330) +MATCH (f137438953677:Forum {id: 137438953677}), (p2199023255712:Person {id: 2199023255712}), (pl55:Place {id: 55}) INSERT (m137438955264:Post:Message {id: 137438955264, imageFile: 'photo137438955264.jpg', creationDate: 1276274322999, locationIP: '115.84.173.212', browserUsed: 'Chrome', length: 0})-[:HAS_CREATOR]->(p2199023255712), (m137438955264)-[:IS_LOCATED_IN]->(pl55), (f137438953677)-[:CONTAINER_OF]->(m137438955264) +MATCH (f137438953678:Forum {id: 137438953678}), (p2199023255712:Person {id: 2199023255712}), (pl47:Place {id: 47}) INSERT (m137438955277:Post:Message {id: 137438955277, imageFile: 'photo137438955277.jpg', creationDate: 1276553299898, locationIP: '186.8.225.25', browserUsed: 'Chrome', length: 0})-[:HAS_CREATOR]->(p2199023255712), (m137438955277)-[:IS_LOCATED_IN]->(pl47), (f137438953678)-[:CONTAINER_OF]->(m137438955277) +MATCH (f137438953678:Forum {id: 137438953678}), (p2199023255712:Person {id: 2199023255712}), (pl66:Place {id: 66}) INSERT (m137438955279:Post:Message {id: 137438955279, imageFile: 'photo137438955279.jpg', creationDate: 1276553301898, locationIP: '64.15.79.8', browserUsed: 'Chrome', length: 0})-[:HAS_CREATOR]->(p2199023255712), (m137438955279)-[:IS_LOCATED_IN]->(pl66), (f137438953678)-[:CONTAINER_OF]->(m137438955279) +MATCH (f567:Forum {id: 567}), (p143:Person {id: 143}), (pl78:Place {id: 78}), (t273:Tag {id: 273}), (t2798:Tag {id: 2798}), (t6394:Tag {id: 6394}) INSERT (m137438958563:Post:Message {id: 137438958563, creationDate: 1277062890260, locationIP: '62.217.119.183', browserUsed: 'Firefox', language: 'uz', content: 'About Aung San Suu Kyi, e elections. She remained underAbout Stevie Wonder, ver awarded to', length: 90})-[:HAS_CREATOR]->(p143), (m137438958563)-[:IS_LOCATED_IN]->(pl78), (m137438958563)-[:HAS_TAG]->(t273), (m137438958563)-[:HAS_TAG]->(t2798), (m137438958563)-[:HAS_TAG]->(t6394), (f567)-[:CONTAINER_OF]->(m137438958563) +MATCH (f872:Forum {id: 872}), (p228:Person {id: 228}), (pl90:Place {id: 90}), (t2797:Tag {id: 2797}), (t294:Tag {id: 294}) INSERT (m137438963499:Post:Message {id: 137438963499, creationDate: 1272740097725, locationIP: '41.190.228.205', browserUsed: 'Chrome', language: 'tk', content: 'About William Lyon Mackenzie King, allowed his intense spirituality to distort About T', length: 86})-[:HAS_CREATOR]->(p228), (m137438963499)-[:IS_LOCATED_IN]->(pl90), (m137438963499)-[:HAS_TAG]->(t294), (m137438963499)-[:HAS_TAG]->(t2797), (f872)-[:CONTAINER_OF]->(m137438963499) +MATCH (f872:Forum {id: 872}), (p228:Person {id: 228}), (pl76:Place {id: 76}), (t2797:Tag {id: 2797}), (t2848:Tag {id: 2848}), (t294:Tag {id: 294}) INSERT (m137438963511:Post:Message {id: 137438963511, creationDate: 1277076649771, locationIP: '213.55.93.153', browserUsed: 'Chrome', language: 'tk', content: 'About William Lyon Mackenzie King, y. King worked to bring compromise andAbout Thomas Jefferson, erson, his wife', length: 112})-[:HAS_CREATOR]->(p228), (m137438963511)-[:IS_LOCATED_IN]->(pl76), (m137438963511)-[:HAS_TAG]->(t294), (m137438963511)-[:HAS_TAG]->(t2797), (m137438963511)-[:HAS_TAG]->(t2848), (f872)-[:CONTAINER_OF]->(m137438963511) +MATCH (f900:Forum {id: 900}), (p150:Person {id: 150}), (pl53:Place {id: 53}), (t1174:Tag {id: 1174}), (t1987:Tag {id: 1987}), (t4854:Tag {id: 4854}), (t5077:Tag {id: 5077}), (t7566:Tag {id: 7566}) INSERT (m137438963740:Post:Message {id: 137438963740, creationDate: 1273649436061, locationIP: '148.240.94.143', browserUsed: 'Firefox', language: 'tk', content: 'About Pope Paul VI, Church life during his pontificate excAbout Winston Churchill, United Kingdom during the Second WorldAbout Julia Gillard, ing Mitcham Demonstration School and UnAbout S', length: 190})-[:HAS_CREATOR]->(p150), (m137438963740)-[:IS_LOCATED_IN]->(pl53), (m137438963740)-[:HAS_TAG]->(t1174), (m137438963740)-[:HAS_TAG]->(t1987), (m137438963740)-[:HAS_TAG]->(t4854), (m137438963740)-[:HAS_TAG]->(t5077), (m137438963740)-[:HAS_TAG]->(t7566), (f900)-[:CONTAINER_OF]->(m137438963740) +MATCH (f900:Forum {id: 900}), (p150:Person {id: 150}), (pl53:Place {id: 53}), (t4854:Tag {id: 4854}), (t6402:Tag {id: 6402}), (t7018:Tag {id: 7018}), (t9061:Tag {id: 9061}) INSERT (m137438963751:Post:Message {id: 137438963751, creationDate: 1273617306061, locationIP: '148.240.94.143', browserUsed: 'Firefox', language: 'tk', content: 'About Julia Gillard, ister upon Labor\'s victory in the 2007 federal electionAbout Chile, Republic of Chile, is a country in South America occupyAbout South Korea, ith production focusing on electronics, automobiles, ', length: 216})-[:HAS_CREATOR]->(p150), (m137438963751)-[:IS_LOCATED_IN]->(pl53), (m137438963751)-[:HAS_TAG]->(t4854), (m137438963751)-[:HAS_TAG]->(t6402), (m137438963751)-[:HAS_TAG]->(t7018), (m137438963751)-[:HAS_TAG]->(t9061), (f900)-[:CONTAINER_OF]->(m137438963751) +MATCH (f200:Forum {id: 200}), (p102:Person {id: 102}), (pl84:Place {id: 84}), (t2084:Tag {id: 2084}) INSERT (m206158431892:Post:Message {id: 206158431892, creationDate: 1278511907643, locationIP: '41.204.102.171', browserUsed: 'Safari', language: 'tk', content: 'About William Penn, – 16 September 1670) was an English admiral and politician who sat in the House of Commons from 1660 to 1670. He was the father of Wi', length: 153})-[:HAS_CREATOR]->(p102), (m206158431892)-[:IS_LOCATED_IN]->(pl84), (m206158431892)-[:HAS_TAG]->(t2084), (f200)-[:CONTAINER_OF]->(m206158431892) +MATCH (f206158430419:Forum {id: 206158430419}), (p2199023255712:Person {id: 2199023255712}), (pl66:Place {id: 66}) INSERT (m206158432090:Post:Message {id: 206158432090, imageFile: 'photo206158432090.jpg', creationDate: 1277597415173, locationIP: '64.7.145.210', browserUsed: 'Chrome', length: 0})-[:HAS_CREATOR]->(p2199023255712), (m206158432090)-[:IS_LOCATED_IN]->(pl66), (f206158430419)-[:CONTAINER_OF]->(m206158432090) +MATCH (f137438953769:Forum {id: 137438953769}), (p4398046511333:Person {id: 4398046511333}), (pl99:Place {id: 99}), (t1419:Tag {id: 1419}) INSERT (m206158432782:Post:Message {id: 206158432782, creationDate: 1280894717004, locationIP: '31.24.152.190', browserUsed: 'Chrome', language: 'uz', content: 'About Guy Sebastian, ur Asian countries and New Zealand. Sebastian had a second number one in New Zealand with Who\'s That Girl, two other top ten singles and a number three album, and gained four platinum and two gold certifications there. He ha', length: 245})-[:HAS_CREATOR]->(p4398046511333), (m206158432782)-[:IS_LOCATED_IN]->(pl99), (m206158432782)-[:HAS_TAG]->(t1419), (f137438953769)-[:CONTAINER_OF]->(m206158432782) +MATCH (f206158431081:Forum {id: 206158431081}), (p228:Person {id: 228}), (pl76:Place {id: 76}) INSERT (m206158440292:Post:Message {id: 206158440292, imageFile: 'photo206158440292.jpg', creationDate: 1279208419298, locationIP: '213.55.93.153', browserUsed: 'Chrome', length: 0})-[:HAS_CREATOR]->(p228), (m206158440292)-[:IS_LOCATED_IN]->(pl76), (f206158431081)-[:CONTAINER_OF]->(m206158440292) +MATCH (f274877907153:Forum {id: 274877907153}), (p2199023255712:Person {id: 2199023255712}), (pl55:Place {id: 55}) INSERT (m274877908808:Post:Message {id: 274877908808, imageFile: 'photo274877908808.jpg', creationDate: 1286564676418, locationIP: '115.84.173.212', browserUsed: 'Chrome', length: 0})-[:HAS_CREATOR]->(p2199023255712), (m274877908808)-[:IS_LOCATED_IN]->(pl55), (f274877907153)-[:CONTAINER_OF]->(m274877908808) +MATCH (f137438953769:Forum {id: 137438953769}), (p4398046511333:Person {id: 4398046511333}), (pl99:Place {id: 99}), (t1419:Tag {id: 1419}) INSERT (m274877909514:Post:Message {id: 274877909514, creationDate: 1283465660488, locationIP: '31.24.152.190', browserUsed: 'Chrome', language: 'uz', content: 'About Guy Sebastian, 08 Australian tour. Like It Like That has three tracks with John Mayer o', length: 93})-[:HAS_CREATOR]->(p4398046511333), (m274877909514)-[:IS_LOCATED_IN]->(pl99), (m274877909514)-[:HAS_TAG]->(t1419), (f137438953769)-[:CONTAINER_OF]->(m274877909514) +MATCH (f206158430557:Forum {id: 206158430557}), (p6597069766775:Person {id: 6597069766775}), (pl1:Place {id: 1}), (t1526:Tag {id: 1526}) INSERT (m274877909919:Post:Message {id: 274877909919, creationDate: 1285682584114, locationIP: '1.4.5.93', browserUsed: 'Internet Explorer', language: 'tk', content: 'About Manuel Noriega, uest in April 2010. He arrived in Paris on April 27, 2010, and after a re-trial as a condition', length: 116})-[:HAS_CREATOR]->(p6597069766775), (m274877909919)-[:IS_LOCATED_IN]->(pl1), (m274877909919)-[:HAS_TAG]->(t1526), (f206158430557)-[:CONTAINER_OF]->(m274877909919) +MATCH (f206158430557:Forum {id: 206158430557}), (p6597069766775:Person {id: 6597069766775}), (pl1:Place {id: 1}), (t1526:Tag {id: 1526}), (t2005:Tag {id: 2005}), (t7658:Tag {id: 7658}) INSERT (m274877909924:Post:Message {id: 274877909924, creationDate: 1285664359114, locationIP: '1.4.5.93', browserUsed: 'Internet Explorer', language: 'tk', content: 'About Manuel Noriega, g, racketeering, and money laundering About Arthur Wellesley, 1st Duke of Wellington, m. He', length: 113})-[:HAS_CREATOR]->(p6597069766775), (m274877909924)-[:IS_LOCATED_IN]->(pl1), (m274877909924)-[:HAS_TAG]->(t1526), (m274877909924)-[:HAS_TAG]->(t2005), (m274877909924)-[:HAS_TAG]->(t7658), (f206158430557)-[:CONTAINER_OF]->(m274877909924) +MATCH (f567:Forum {id: 567}), (p143:Person {id: 143}), (pl78:Place {id: 78}), (t1185:Tag {id: 1185}), (t1201:Tag {id: 1201}), (t2029:Tag {id: 2029}), (t2922:Tag {id: 2922}), (t5080:Tag {id: 5080}) INSERT (m274877912121:Post:Message {id: 274877912121, creationDate: 1287511214129, locationIP: '62.217.119.183', browserUsed: 'Firefox', language: 'uz', content: 'About Pope Leo XIII, both the rosary and the scapular. He issued a reAbout Horace, n. He is now largely remembered for Strawberry HiAbout William Ewart Gladstone, cond ministry, which saw crises in Egypt (culminaAbout Jefferson Davis, d Uni', length: 241})-[:HAS_CREATOR]->(p143), (m274877912121)-[:IS_LOCATED_IN]->(pl78), (m274877912121)-[:HAS_TAG]->(t1185), (m274877912121)-[:HAS_TAG]->(t1201), (m274877912121)-[:HAS_TAG]->(t2029), (m274877912121)-[:HAS_TAG]->(t2922), (m274877912121)-[:HAS_TAG]->(t5080), (f567)-[:CONTAINER_OF]->(m274877912121) +MATCH (f567:Forum {id: 567}), (p143:Person {id: 143}), (pl78:Place {id: 78}), (t10144:Tag {id: 10144}), (t1761:Tag {id: 1761}) INSERT (m274877912139:Post:Message {id: 274877912139, creationDate: 1284593996258, locationIP: '62.217.119.183', browserUsed: 'Firefox', language: 'uz', content: 'About Enrique Iglesias, hits on the various Billboard charts. Billboard has called him The King of Latin Pop and The King of DanAbout California King Bed, ive reviews from music critics, who praised Rihanna', length: 206})-[:HAS_CREATOR]->(p143), (m274877912139)-[:IS_LOCATED_IN]->(pl78), (m274877912139)-[:HAS_TAG]->(t1761), (m274877912139)-[:HAS_TAG]->(t10144), (f567)-[:CONTAINER_OF]->(m274877912139) +MATCH (f689:Forum {id: 689}), (p41:Person {id: 41}), (pl0:Place {id: 0}), (t1410:Tag {id: 1410}) INSERT (m274877913521:Post:Message {id: 274877913521, creationDate: 1283268110966, locationIP: '27.116.33.147', browserUsed: 'Safari', language: 'uz', content: 'About Muammar Gaddafi, inister of Libya In office16 January 1970 – 16 July 1972 Preceded by Mahmud Sulayman al-Magh', length: 115})-[:HAS_CREATOR]->(p41), (m274877913521)-[:IS_LOCATED_IN]->(pl0), (m274877913521)-[:HAS_TAG]->(t1410), (f689)-[:CONTAINER_OF]->(m274877913521) +MATCH (f343597383890:Forum {id: 343597383890}), (p2199023255712:Person {id: 2199023255712}), (pl55:Place {id: 55}) INSERT (m343597385545:Post:Message {id: 343597385545, imageFile: 'photo343597385545.jpg', creationDate: 1288453284991, locationIP: '115.84.173.212', browserUsed: 'Chrome', length: 0})-[:HAS_CREATOR]->(p2199023255712), (m343597385545)-[:IS_LOCATED_IN]->(pl55), (f343597383890)-[:CONTAINER_OF]->(m343597385545) +MATCH (f567:Forum {id: 567}), (p143:Person {id: 143}), (pl78:Place {id: 78}), (t273:Tag {id: 273}) INSERT (m343597388806:Post:Message {id: 343597388806, creationDate: 1288427336182, locationIP: '62.217.119.183', browserUsed: 'Firefox', language: 'uz', content: 'About Aung San Suu Kyi, u Award for International Understanding by the government of India and the International Simón Bolívar Prize from the government of Venezuela. In 2007, the Government of Canada made her an honorary citizen of that country; ', length: 247})-[:HAS_CREATOR]->(p143), (m343597388806)-[:IS_LOCATED_IN]->(pl78), (m343597388806)-[:HAS_TAG]->(t273), (f567)-[:CONTAINER_OF]->(m343597388806) +MATCH (f689:Forum {id: 689}), (p41:Person {id: 41}), (pl0:Place {id: 0}), (t1410:Tag {id: 1410}) INSERT (m343597390255:Post:Message {id: 343597390255, creationDate: 1288363961759, locationIP: '27.116.33.147', browserUsed: 'Safari', language: 'uz', content: 'About Muammar Gaddafi, l prices and extraction in Libya led to increasing revenues. By ', length: 87})-[:HAS_CREATOR]->(p41), (m343597390255)-[:IS_LOCATED_IN]->(pl0), (m343597390255)-[:HAS_TAG]->(t1410), (f689)-[:CONTAINER_OF]->(m343597390255) +MATCH (f343597384373:Forum {id: 343597384373}), (p41:Person {id: 41}), (pl0:Place {id: 0}) INSERT (m343597390340:Post:Message {id: 343597390340, imageFile: 'photo343597390340.jpg', creationDate: 1287673946664, locationIP: '27.116.33.147', browserUsed: 'Safari', length: 0})-[:HAS_CREATOR]->(p41), (m343597390340)-[:IS_LOCATED_IN]->(pl0), (f343597384373)-[:CONTAINER_OF]->(m343597390340) +MATCH (f767:Forum {id: 767}), (p76:Person {id: 76}), (pl98:Place {id: 98}), (t1538:Tag {id: 1538}) INSERT (m343597392282:Post:Message {id: 343597392282, creationDate: 1289863438482, locationIP: '27.35.111.48', browserUsed: 'Chrome', language: 'uz', content: 'About Emilio Aguinaldo, ne-American War or War of Philippine Independence that resisted Amer', length: 92})-[:HAS_CREATOR]->(p76), (m343597392282)-[:IS_LOCATED_IN]->(pl98), (m343597392282)-[:HAS_TAG]->(t1538), (f767)-[:CONTAINER_OF]->(m343597392282) +MATCH (f767:Forum {id: 767}), (p76:Person {id: 76}), (pl98:Place {id: 98}), (t1530:Tag {id: 1530}), (t5117:Tag {id: 5117}), (t5445:Tag {id: 5445}), (t7559:Tag {id: 7559}) INSERT (m343597392324:Post:Message {id: 343597392324, creationDate: 1289034280205, locationIP: '27.35.111.48', browserUsed: 'Chrome', language: 'uz', content: 'About Luis Horna, clay. He was the Men\'s DoAbout Robert Altman, my of Motion Picture ArtsAbout Br', length: 97})-[:HAS_CREATOR]->(p76), (m343597392324)-[:IS_LOCATED_IN]->(pl98), (m343597392324)-[:HAS_TAG]->(t1530), (m343597392324)-[:HAS_TAG]->(t5117), (m343597392324)-[:HAS_TAG]->(t5445), (m343597392324)-[:HAS_TAG]->(t7559), (f767)-[:CONTAINER_OF]->(m343597392324) +MATCH (f767:Forum {id: 767}), (p76:Person {id: 76}), (pl98:Place {id: 98}), (t11543:Tag {id: 11543}), (t1530:Tag {id: 1530}), (t2795:Tag {id: 2795}), (t9164:Tag {id: 9164}), (t9249:Tag {id: 9249}) INSERT (m343597392338:Post:Message {id: 343597392338, creationDate: 1289038915205, locationIP: '27.35.111.48', browserUsed: 'Chrome', language: 'uz', content: 'About Luis Horna, ofessional in 1998. HoAbout Harry S. Truman, rld struggling for freAbout Republic of the C', length: 108})-[:HAS_CREATOR]->(p76), (m343597392338)-[:IS_LOCATED_IN]->(pl98), (m343597392338)-[:HAS_TAG]->(t1530), (m343597392338)-[:HAS_TAG]->(t2795), (m343597392338)-[:HAS_TAG]->(t9164), (m343597392338)-[:HAS_TAG]->(t9249), (m343597392338)-[:HAS_TAG]->(t11543), (f767)-[:CONTAINER_OF]->(m343597392338) +MATCH (f872:Forum {id: 872}), (p228:Person {id: 228}), (pl76:Place {id: 76}), (t1530:Tag {id: 1530}), (t5059:Tag {id: 5059}), (t6940:Tag {id: 6940}), (t7670:Tag {id: 7670}) INSERT (m343597393747:Post:Message {id: 343597393747, creationDate: 1289066095205, locationIP: '213.55.93.153', browserUsed: 'Chrome', language: 'tk', content: 'About Luis Horna, as a strong serve for a relatively shoAbout Djibouti, uti National Army and its sub-branchesAbout Leonard Bernstein, hilharmonic, ', length: 148})-[:HAS_CREATOR]->(p228), (m343597393747)-[:IS_LOCATED_IN]->(pl76), (m343597393747)-[:HAS_TAG]->(t1530), (m343597393747)-[:HAS_TAG]->(t5059), (m343597393747)-[:HAS_TAG]->(t6940), (m343597393747)-[:HAS_TAG]->(t7670), (f872)-[:CONTAINER_OF]->(m343597393747) +MATCH (f343597384554:Forum {id: 343597384554}), (p228:Person {id: 228}), (pl76:Place {id: 76}) INSERT (m343597393778:Post:Message {id: 343597393778, imageFile: 'photo343597393778.jpg', creationDate: 1290590905265, locationIP: '213.55.93.153', browserUsed: 'Chrome', length: 0})-[:HAS_CREATOR]->(p228), (m343597393778)-[:IS_LOCATED_IN]->(pl76), (f343597384554)-[:CONTAINER_OF]->(m343597393778) +MATCH (f343597384554:Forum {id: 343597384554}), (p228:Person {id: 228}), (pl76:Place {id: 76}) INSERT (m343597393779:Post:Message {id: 343597393779, imageFile: 'photo343597393779.jpg', creationDate: 1290590906265, locationIP: '213.55.93.153', browserUsed: 'Chrome', length: 0})-[:HAS_CREATOR]->(p228), (m343597393779)-[:IS_LOCATED_IN]->(pl76), (f343597384554)-[:CONTAINER_OF]->(m343597393779) +MATCH (f343597384587:Forum {id: 343597384587}), (p150:Person {id: 150}), (pl53:Place {id: 53}) INSERT (m343597394049:Post:Message {id: 343597394049, imageFile: 'photo343597394049.jpg', creationDate: 1288654829102, locationIP: '148.240.94.143', browserUsed: 'Firefox', length: 0})-[:HAS_CREATOR]->(p150), (m343597394049)-[:IS_LOCATED_IN]->(pl53), (f343597384587)-[:CONTAINER_OF]->(m343597394049) +MATCH (f343597384587:Forum {id: 343597384587}), (p150:Person {id: 150}), (pl53:Place {id: 53}) INSERT (m343597394050:Post:Message {id: 343597394050, imageFile: 'photo343597394050.jpg', creationDate: 1288654830102, locationIP: '148.240.94.143', browserUsed: 'Firefox', length: 0})-[:HAS_CREATOR]->(p150), (m343597394050)-[:IS_LOCATED_IN]->(pl53), (f343597384587)-[:CONTAINER_OF]->(m343597394050) +MATCH (f913:Forum {id: 913}), (p153:Person {id: 153}), (pl96:Place {id: 96}), (t0:Tag {id: 0}) INSERT (m343597394146:Post:Message {id: 343597394146, creationDate: 1288454579368, locationIP: '196.1.98.252', browserUsed: 'Firefox', language: 'tk', content: 'About Hamid Karzai, he removal of the Taliban regime in late 2001. During the December 2001 International Conference on Afghanistan in Germany, Karzai was selected by promi', length: 172})-[:HAS_CREATOR]->(p153), (m343597394146)-[:IS_LOCATED_IN]->(pl96), (m343597394146)-[:HAS_TAG]->(t0), (f913)-[:CONTAINER_OF]->(m343597394146) +MATCH (f343597384602:Forum {id: 343597384602}), (p153:Person {id: 153}), (pl96:Place {id: 96}) INSERT (m343597394406:Post:Message {id: 343597394406, imageFile: 'photo343597394406.jpg', creationDate: 1288058816120, locationIP: '196.1.98.252', browserUsed: 'Firefox', length: 0})-[:HAS_CREATOR]->(p153), (m343597394406)-[:IS_LOCATED_IN]->(pl96), (f343597384602)-[:CONTAINER_OF]->(m343597394406) +MATCH (m68719478399:Message {id: 68719478399}), (p143:Person {id: 143}), (pl78:Place {id: 78}), (t2084:Tag {id: 2084}), (t5132:Tag {id: 5132}) INSERT (m68719478400:Comment:Message {id: 68719478400, creationDate: 1269860075146, locationIP: '62.217.119.183', browserUsed: 'Firefox', content: 'About William Penn, 670. He was the father of William Penn,About Alexander Ha', length: 77})-[:HAS_CREATOR]->(p143), (m68719478400)-[:IS_LOCATED_IN]->(pl78), (m68719478400)-[:HAS_TAG]->(t2084), (m68719478400)-[:HAS_TAG]->(t5132), (m68719478400)-[:REPLY_OF]->(m68719478399) +MATCH (m68719478441:Message {id: 68719478441}), (p228:Person {id: 228}), (pl76:Place {id: 76}), (t11035:Tag {id: 11035}), (t12950:Tag {id: 12950}), (t13379:Tag {id: 13379}), (t2076:Tag {id: 2076}), (t6373:Tag {id: 6373}), (t7555:Tag {id: 7555}) INSERT (m68719478445:Comment:Message {id: 68719478445, creationDate: 1271801672648, locationIP: '213.55.93.153', browserUsed: 'Chrome', content: 'About William Morris, the end of thAbout Guam, ted Nations. TAbout France, ymath,', length: 82})-[:HAS_CREATOR]->(p228), (m68719478445)-[:IS_LOCATED_IN]->(pl76), (m68719478445)-[:HAS_TAG]->(t2076), (m68719478445)-[:HAS_TAG]->(t6373), (m68719478445)-[:HAS_TAG]->(t7555), (m68719478445)-[:HAS_TAG]->(t11035), (m68719478445)-[:HAS_TAG]->(t12950), (m68719478445)-[:HAS_TAG]->(t13379), (m68719478445)-[:REPLY_OF]->(m68719478441) +MATCH (m68719478453:Message {id: 68719478453}), (p76:Person {id: 76}), (pl98:Place {id: 98}) INSERT (m68719478455:Comment:Message {id: 68719478455, creationDate: 1271842486160, locationIP: '27.35.111.48', browserUsed: 'Chrome', content: 'roflol', length: 6})-[:HAS_CREATOR]->(p76), (m68719478455)-[:IS_LOCATED_IN]->(pl98), (m68719478455)-[:REPLY_OF]->(m68719478453) +MATCH (m68719478453:Message {id: 68719478453}), (p228:Person {id: 228}), (pl76:Place {id: 76}), (t11686:Tag {id: 11686}), (t12763:Tag {id: 12763}), (t13191:Tag {id: 13191}), (t1614:Tag {id: 1614}), (t2044:Tag {id: 2044}), (t5108:Tag {id: 5108}), (t9572:Tag {id: 9572}) INSERT (m68719478457:Comment:Message {id: 68719478457, creationDate: 1271850332021, locationIP: '213.55.93.153', browserUsed: 'Chrome', content: 'About Oscar Wilde, intellectuals. Their sAbout Anne, Queen of Great Britain, II became joint monarcAbout Lebanon, ld national infrastrucAbout Bad, Bad L', length: 152})-[:HAS_CREATOR]->(p228), (m68719478457)-[:IS_LOCATED_IN]->(pl76), (m68719478457)-[:HAS_TAG]->(t1614), (m68719478457)-[:HAS_TAG]->(t2044), (m68719478457)-[:HAS_TAG]->(t5108), (m68719478457)-[:HAS_TAG]->(t9572), (m68719478457)-[:HAS_TAG]->(t11686), (m68719478457)-[:HAS_TAG]->(t12763), (m68719478457)-[:HAS_TAG]->(t13191), (m68719478457)-[:REPLY_OF]->(m68719478453) +MATCH (m68719485292:Message {id: 68719485292}), (p150:Person {id: 150}), (pl53:Place {id: 53}), (t2841:Tag {id: 2841}) INSERT (m68719485303:Comment:Message {id: 68719485303, creationDate: 1268236134952, locationIP: '148.240.94.143', browserUsed: 'Firefox', content: 'About Martin Scorsese, Scorsese is hailed as one of the most significant and influe', length: 83})-[:HAS_CREATOR]->(p150), (m68719485303)-[:IS_LOCATED_IN]->(pl53), (m68719485303)-[:HAS_TAG]->(t2841), (m68719485303)-[:REPLY_OF]->(m68719485292) +MATCH (m68719487300:Message {id: 68719487300}), (p150:Person {id: 150}), (pl53:Place {id: 53}) INSERT (m68719487303:Comment:Message {id: 68719487303, creationDate: 1271762570742, locationIP: '148.240.94.143', browserUsed: 'Firefox', content: 'LOL', length: 3})-[:HAS_CREATOR]->(p150), (m68719487303)-[:IS_LOCATED_IN]->(pl53), (m68719487303)-[:REPLY_OF]->(m68719487300) +MATCH (m68719487313:Message {id: 68719487313}), (p150:Person {id: 150}), (pl53:Place {id: 53}), (t11358:Tag {id: 11358}), (t11644:Tag {id: 11644}), (t5100:Tag {id: 5100}), (t6458:Tag {id: 6458}), (t7010:Tag {id: 7010}) INSERT (m68719487322:Comment:Message {id: 68719487322, creationDate: 1271697588629, locationIP: '148.240.94.143', browserUsed: 'Firefox', content: 'About Ecuador, quil. The historicAbout A. R. Rahman, BAFTA Award, a GoAbout Pet Sound', length: 86})-[:HAS_CREATOR]->(p150), (m68719487322)-[:IS_LOCATED_IN]->(pl53), (m68719487322)-[:HAS_TAG]->(t5100), (m68719487322)-[:HAS_TAG]->(t6458), (m68719487322)-[:HAS_TAG]->(t7010), (m68719487322)-[:HAS_TAG]->(t11358), (m68719487322)-[:HAS_TAG]->(t11644), (m68719487322)-[:REPLY_OF]->(m68719487313) +MATCH (m68719487313:Message {id: 68719487313}), (p143:Person {id: 143}), (pl47:Place {id: 47}), (t1169:Tag {id: 1169}), (t13010:Tag {id: 13010}), (t5418:Tag {id: 5418}), (t573:Tag {id: 573}), (t6458:Tag {id: 6458}), (t8115:Tag {id: 8115}), (t9074:Tag {id: 9074}) INSERT (m68719487323:Comment:Message {id: 68719487323, creationDate: 1271686755090, locationIP: '164.73.117.203', browserUsed: 'Firefox', content: 'About Victor Hugo, essayist, viAbout Dudi Sela, r in straighAbout J. P. Morgan', length: 78})-[:HAS_CREATOR]->(p143), (m68719487323)-[:IS_LOCATED_IN]->(pl47), (m68719487323)-[:HAS_TAG]->(t573), (m68719487323)-[:HAS_TAG]->(t1169), (m68719487323)-[:HAS_TAG]->(t5418), (m68719487323)-[:HAS_TAG]->(t6458), (m68719487323)-[:HAS_TAG]->(t8115), (m68719487323)-[:HAS_TAG]->(t9074), (m68719487323)-[:HAS_TAG]->(t13010), (m68719487323)-[:REPLY_OF]->(m68719487313) +MATCH (m68719487330:Message {id: 68719487330}), (p143:Person {id: 143}), (pl78:Place {id: 78}), (t1169:Tag {id: 1169}), (t2992:Tag {id: 2992}), (t5068:Tag {id: 5068}) INSERT (m68719487334:Comment:Message {id: 68719487334, creationDate: 1271872503696, locationIP: '62.217.119.183', browserUsed: 'Firefox', content: 'About Dudi Sela, Amir Weintraub. As a 17-yeAbout Philip K. Dick, of his caree', length: 77})-[:HAS_CREATOR]->(p143), (m68719487334)-[:IS_LOCATED_IN]->(pl78), (m68719487334)-[:HAS_TAG]->(t1169), (m68719487334)-[:HAS_TAG]->(t2992), (m68719487334)-[:HAS_TAG]->(t5068), (m68719487334)-[:REPLY_OF]->(m68719487330) +MATCH (m137438958563:Message {id: 137438958563}), (p102:Person {id: 102}), (pl84:Place {id: 84}) INSERT (m137438958564:Comment:Message {id: 137438958564, creationDate: 1277084387523, locationIP: '41.204.102.171', browserUsed: 'Safari', content: 'cool', length: 4})-[:HAS_CREATOR]->(p102), (m137438958564)-[:IS_LOCATED_IN]->(pl84), (m137438958564)-[:REPLY_OF]->(m137438958563) +MATCH (m137438963499:Message {id: 137438963499}), (p102:Person {id: 102}), (pl84:Place {id: 84}), (t2834:Tag {id: 2834}) INSERT (m137438963501:Comment:Message {id: 137438963501, creationDate: 1272755669544, locationIP: '41.204.102.171', browserUsed: 'Safari', content: 'About Frank Zappa, ollages. His later albums shared this eclectic and experimental approach, irrespectiv', length: 104})-[:HAS_CREATOR]->(p102), (m137438963501)-[:IS_LOCATED_IN]->(pl84), (m137438963501)-[:HAS_TAG]->(t2834), (m137438963501)-[:REPLY_OF]->(m137438963499) +MATCH (m137438963499:Message {id: 137438963499}), (p150:Person {id: 150}), (pl53:Place {id: 53}), (t2120:Tag {id: 2120}), (t2797:Tag {id: 2797}) INSERT (m137438963506:Comment:Message {id: 137438963506, creationDate: 1272757138003, locationIP: '148.240.94.143', browserUsed: 'Firefox', content: 'About Thomas Hardy, y published as serials in magazines, weAbout Thomas Jeffe', length: 77})-[:HAS_CREATOR]->(p150), (m137438963506)-[:IS_LOCATED_IN]->(pl53), (m137438963506)-[:HAS_TAG]->(t2120), (m137438963506)-[:HAS_TAG]->(t2797), (m137438963506)-[:REPLY_OF]->(m137438963499) +MATCH (m137438963511:Message {id: 137438963511}), (p2199023255712:Person {id: 2199023255712}), (pl55:Place {id: 55}) INSERT (m137438963517:Comment:Message {id: 137438963517, creationDate: 1277076710526, locationIP: '115.84.173.212', browserUsed: 'Chrome', content: 'I see', length: 5})-[:HAS_CREATOR]->(p2199023255712), (m137438963517)-[:IS_LOCATED_IN]->(pl55), (m137438963517)-[:REPLY_OF]->(m137438963511) +MATCH (m137438963740:Message {id: 137438963740}), (p228:Person {id: 228}), (pl57:Place {id: 57}), (t2114:Tag {id: 2114}), (t2855:Tag {id: 2855}), (t5077:Tag {id: 5077}), (t7106:Tag {id: 7106}) INSERT (m137438963744:Comment:Message {id: 137438963744, creationDate: 1273650346399, locationIP: '24.31.31.41', browserUsed: 'Chrome', content: 'About Christopher Lee, or services to dramAbout Tina Turner, inning with a', length: 74})-[:HAS_CREATOR]->(p228), (m137438963744)-[:IS_LOCATED_IN]->(pl57), (m137438963744)-[:HAS_TAG]->(t2114), (m137438963744)-[:HAS_TAG]->(t2855), (m137438963744)-[:HAS_TAG]->(t5077), (m137438963744)-[:HAS_TAG]->(t7106), (m137438963744)-[:REPLY_OF]->(m137438963740) +MATCH (m137438963751:Message {id: 137438963751}), (p76:Person {id: 76}), (pl14:Place {id: 14}) INSERT (m137438963765:Comment:Message {id: 137438963765, creationDate: 1273696787135, locationIP: '80.221.178.252', browserUsed: 'Chrome', content: 'right', length: 5})-[:HAS_CREATOR]->(p76), (m137438963765)-[:IS_LOCATED_IN]->(pl14), (m137438963765)-[:REPLY_OF]->(m137438963751) +MATCH (m137438963751:Message {id: 137438963751}), (p153:Person {id: 153}), (pl96:Place {id: 96}) INSERT (m137438963766:Comment:Message {id: 137438963766, creationDate: 1273627813453, locationIP: '196.1.98.252', browserUsed: 'Firefox', content: 'thanks', length: 6})-[:HAS_CREATOR]->(p153), (m137438963766)-[:IS_LOCATED_IN]->(pl96), (m137438963766)-[:REPLY_OF]->(m137438963751) +MATCH (m206158431892:Message {id: 206158431892}), (p228:Person {id: 228}), (pl76:Place {id: 76}) INSERT (m206158431893:Comment:Message {id: 206158431893, creationDate: 1278531351427, locationIP: '213.55.93.153', browserUsed: 'Chrome', content: 'roflol', length: 6})-[:HAS_CREATOR]->(p228), (m206158431893)-[:IS_LOCATED_IN]->(pl76), (m206158431893)-[:REPLY_OF]->(m206158431892) +MATCH (m206158431892:Message {id: 206158431892}), (p143:Person {id: 143}), (pl78:Place {id: 78}) INSERT (m206158431896:Comment:Message {id: 206158431896, creationDate: 1278512054636, locationIP: '62.217.119.183', browserUsed: 'Firefox', content: 'yes', length: 3})-[:HAS_CREATOR]->(p143), (m206158431896)-[:IS_LOCATED_IN]->(pl78), (m206158431896)-[:REPLY_OF]->(m206158431892) +MATCH (m206158432782:Message {id: 206158432782}), (p153:Person {id: 153}), (pl96:Place {id: 96}) INSERT (m206158432792:Comment:Message {id: 206158432792, creationDate: 1280914332302, locationIP: '196.1.98.252', browserUsed: 'Firefox', content: 'duh', length: 3})-[:HAS_CREATOR]->(p153), (m206158432792)-[:IS_LOCATED_IN]->(pl96), (m206158432792)-[:REPLY_OF]->(m206158432782) +MATCH (m206158432782:Message {id: 206158432782}), (p76:Person {id: 76}), (pl98:Place {id: 98}) INSERT (m206158432797:Comment:Message {id: 206158432797, creationDate: 1280907556754, locationIP: '27.35.111.48', browserUsed: 'Chrome', content: 'thx', length: 3})-[:HAS_CREATOR]->(p76), (m206158432797)-[:IS_LOCATED_IN]->(pl98), (m206158432797)-[:REPLY_OF]->(m206158432782) +MATCH (m274877909919:Message {id: 274877909919}), (p143:Person {id: 143}), (pl78:Place {id: 78}) INSERT (m274877909920:Comment:Message {id: 274877909920, creationDate: 1285685102274, locationIP: '62.217.119.183', browserUsed: 'Firefox', content: 'roflol', length: 6})-[:HAS_CREATOR]->(p143), (m274877909920)-[:IS_LOCATED_IN]->(pl78), (m274877909920)-[:REPLY_OF]->(m274877909919) +MATCH (m274877909924:Message {id: 274877909924}), (p76:Person {id: 76}), (pl98:Place {id: 98}) INSERT (m274877909925:Comment:Message {id: 274877909925, creationDate: 1285676860041, locationIP: '27.35.111.48', browserUsed: 'Chrome', content: 'thx', length: 3})-[:HAS_CREATOR]->(p76), (m274877909925)-[:IS_LOCATED_IN]->(pl98), (m274877909925)-[:REPLY_OF]->(m274877909924) +MATCH (m274877909924:Message {id: 274877909924}), (p4398046511333:Person {id: 4398046511333}), (pl99:Place {id: 99}), (t1526:Tag {id: 1526}), (t1671:Tag {id: 1671}), (t2870:Tag {id: 2870}) INSERT (m274877909927:Comment:Message {id: 274877909927, creationDate: 1285665306179, locationIP: '31.24.152.190', browserUsed: 'Chrome', content: 'About Manuel Noriega, trafficking, racketeering, andAbout Mikhail Gorbachev, was the onl', length: 89})-[:HAS_CREATOR]->(p4398046511333), (m274877909927)-[:IS_LOCATED_IN]->(pl99), (m274877909927)-[:HAS_TAG]->(t1526), (m274877909927)-[:HAS_TAG]->(t1671), (m274877909927)-[:HAS_TAG]->(t2870), (m274877909927)-[:REPLY_OF]->(m274877909924) +MATCH (m274877912121:Message {id: 274877912121}), (p6597069766775:Person {id: 6597069766775}), (pl1:Place {id: 1}) INSERT (m274877912122:Comment:Message {id: 274877912122, creationDate: 1287513991672, locationIP: '1.4.5.93', browserUsed: 'Internet Explorer', content: 'no way!', length: 7})-[:HAS_CREATOR]->(p6597069766775), (m274877912122)-[:IS_LOCATED_IN]->(pl1), (m274877912122)-[:REPLY_OF]->(m274877912121) +MATCH (m274877912121:Message {id: 274877912121}), (p153:Person {id: 153}), (pl96:Place {id: 96}), (t11647:Tag {id: 11647}), (t1185:Tag {id: 1185}), (t2922:Tag {id: 2922}), (t5109:Tag {id: 5109}), (t579:Tag {id: 579}) INSERT (m274877912131:Comment:Message {id: 274877912131, creationDate: 1287512409156, locationIP: '196.1.98.252', browserUsed: 'Firefox', content: 'About Joan of Arc, ne guidance, she led tAbout Pope Leo XIII, – 20 July 1903), born About Jefferson Davis, ', length: 107})-[:HAS_CREATOR]->(p153), (m274877912131)-[:IS_LOCATED_IN]->(pl96), (m274877912131)-[:HAS_TAG]->(t579), (m274877912131)-[:HAS_TAG]->(t1185), (m274877912131)-[:HAS_TAG]->(t2922), (m274877912131)-[:HAS_TAG]->(t5109), (m274877912131)-[:HAS_TAG]->(t11647), (m274877912131)-[:REPLY_OF]->(m274877912121) +MATCH (m274877912121:Message {id: 274877912121}), (p41:Person {id: 41}), (pl0:Place {id: 0}) INSERT (m274877912136:Comment:Message {id: 274877912136, creationDate: 1287526240684, locationIP: '27.116.33.147', browserUsed: 'Safari', content: 'LOL', length: 3})-[:HAS_CREATOR]->(p41), (m274877912136)-[:IS_LOCATED_IN]->(pl0), (m274877912136)-[:REPLY_OF]->(m274877912121) +MATCH (m274877912139:Message {id: 274877912139}), (p4398046511333:Person {id: 4398046511333}), (pl99:Place {id: 99}) INSERT (m274877912142:Comment:Message {id: 274877912142, creationDate: 1284613964426, locationIP: '31.24.152.190', browserUsed: 'Chrome', content: 'no', length: 2})-[:HAS_CREATOR]->(p4398046511333), (m274877912142)-[:IS_LOCATED_IN]->(pl99), (m274877912142)-[:REPLY_OF]->(m274877912139) +MATCH (m274877912139:Message {id: 274877912139}), (p41:Person {id: 41}), (pl0:Place {id: 0}), (t10144:Tag {id: 10144}), (t5276:Tag {id: 5276}) INSERT (m274877912146:Comment:Message {id: 274877912146, creationDate: 1284597715993, locationIP: '27.116.33.147', browserUsed: 'Safari', content: 'About Return of Saturn, many of the songs describe singer Gwen StefanAbout California Kin', length: 89})-[:HAS_CREATOR]->(p41), (m274877912146)-[:IS_LOCATED_IN]->(pl0), (m274877912146)-[:HAS_TAG]->(t5276), (m274877912146)-[:HAS_TAG]->(t10144), (m274877912146)-[:REPLY_OF]->(m274877912139) +MATCH (m274877913521:Message {id: 274877913521}), (p143:Person {id: 143}), (pl78:Place {id: 78}) INSERT (m274877913531:Comment:Message {id: 274877913531, creationDate: 1283303511849, locationIP: '62.217.119.183', browserUsed: 'Firefox', content: 'right', length: 5})-[:HAS_CREATOR]->(p143), (m274877913531)-[:IS_LOCATED_IN]->(pl78), (m274877913531)-[:REPLY_OF]->(m274877913521) +MATCH (m343597392282:Message {id: 343597392282}), (p4398046511333:Person {id: 4398046511333}), (pl99:Place {id: 99}) INSERT (m343597392285:Comment:Message {id: 343597392285, creationDate: 1289863576755, locationIP: '31.24.152.190', browserUsed: 'Chrome', content: 'thanks', length: 6})-[:HAS_CREATOR]->(p4398046511333), (m343597392285)-[:IS_LOCATED_IN]->(pl99), (m343597392285)-[:REPLY_OF]->(m343597392282) +MATCH (m343597392324:Message {id: 343597392324}), (p228:Person {id: 228}), (pl76:Place {id: 76}), (t2945:Tag {id: 2945}), (t5445:Tag {id: 5445}), (t6969:Tag {id: 6969}), (t7559:Tag {id: 7559}) INSERT (m343597392325:Comment:Message {id: 343597392325, creationDate: 1289049183991, locationIP: '213.55.93.153', browserUsed: 'Chrome', content: 'About Ne-Yo, November 22, 2010. InAbout Brian Wilson, elled for various reaAbout V', length: 82})-[:HAS_CREATOR]->(p228), (m343597392325)-[:IS_LOCATED_IN]->(pl76), (m343597392325)-[:HAS_TAG]->(t2945), (m343597392325)-[:HAS_TAG]->(t5445), (m343597392325)-[:HAS_TAG]->(t6969), (m343597392325)-[:HAS_TAG]->(t7559), (m343597392325)-[:REPLY_OF]->(m343597392324) +MATCH (m343597392324:Message {id: 343597392324}), (p2199023255712:Person {id: 2199023255712}), (pl55:Place {id: 55}) INSERT (m343597392334:Comment:Message {id: 343597392334, creationDate: 1289057495769, locationIP: '115.84.173.212', browserUsed: 'Chrome', content: 'LOL', length: 3})-[:HAS_CREATOR]->(p2199023255712), (m343597392334)-[:IS_LOCATED_IN]->(pl55), (m343597392334)-[:REPLY_OF]->(m343597392324) +MATCH (m343597392324:Message {id: 343597392324}), (p6597069766775:Person {id: 6597069766775}), (pl1:Place {id: 1}) INSERT (m343597392336:Comment:Message {id: 343597392336, creationDate: 1289111036572, locationIP: '1.4.5.93', browserUsed: 'Internet Explorer', content: 'I see', length: 5})-[:HAS_CREATOR]->(p6597069766775), (m343597392336)-[:IS_LOCATED_IN]->(pl1), (m343597392336)-[:REPLY_OF]->(m343597392324) +MATCH (m343597393747:Message {id: 343597393747}), (p76:Person {id: 76}), (pl98:Place {id: 98}), (t12078:Tag {id: 12078}), (t1530:Tag {id: 1530}), (t5800:Tag {id: 5800}) INSERT (m343597393750:Comment:Message {id: 343597393750, creationDate: 1289135518830, locationIP: '27.35.111.48', browserUsed: 'Chrome', content: 'About Luis Horna, e surface is clay. He was the About Waiting for the End, ough it was ', length: 87})-[:HAS_CREATOR]->(p76), (m343597393750)-[:IS_LOCATED_IN]->(pl98), (m343597393750)-[:HAS_TAG]->(t1530), (m343597393750)-[:HAS_TAG]->(t5800), (m343597393750)-[:HAS_TAG]->(t12078), (m343597393750)-[:REPLY_OF]->(m343597393747) +MATCH (m343597393747:Message {id: 343597393747}), (p102:Person {id: 102}), (pl84:Place {id: 84}), (t11269:Tag {id: 11269}), (t3111:Tag {id: 3111}), (t5059:Tag {id: 5059}), (t6940:Tag {id: 6940}) INSERT (m343597393755:Comment:Message {id: 343597393755, creationDate: 1289090250676, locationIP: '41.204.102.171', browserUsed: 'Safari', content: 'About Al Capone, ing the money he made from his actAbout Djibouti, military forces of Djibouti and coAbout Leonard Bernstein, talent', length: 132})-[:HAS_CREATOR]->(p102), (m343597393755)-[:IS_LOCATED_IN]->(pl84), (m343597393755)-[:HAS_TAG]->(t3111), (m343597393755)-[:HAS_TAG]->(t5059), (m343597393755)-[:HAS_TAG]->(t6940), (m343597393755)-[:HAS_TAG]->(t11269), (m343597393755)-[:REPLY_OF]->(m343597393747) +MATCH (m68719478400:Message {id: 68719478400}), (p228:Person {id: 228}), (pl76:Place {id: 76}), (t1176:Tag {id: 1176}), (t5132:Tag {id: 5132}) INSERT (m68719478401:Comment:Message {id: 68719478401, creationDate: 1269860333890, locationIP: '213.55.93.153', browserUsed: 'Chrome', content: 'About Leonardo da Vinci, ne painter, Verrocchio. Much of his earAbout Alexand', length: 77})-[:HAS_CREATOR]->(p228), (m68719478401)-[:IS_LOCATED_IN]->(pl76), (m68719478401)-[:HAS_TAG]->(t1176), (m68719478401)-[:HAS_TAG]->(t5132), (m68719478401)-[:REPLY_OF]->(m68719478400) +MATCH (m68719478445:Message {id: 68719478445}), (p143:Person {id: 143}), (pl78:Place {id: 78}), (t11035:Tag {id: 11035}), (t12742:Tag {id: 12742}), (t2076:Tag {id: 2076}), (t61:Tag {id: 61}), (t7555:Tag {id: 7555}), (t9147:Tag {id: 9147}) INSERT (m68719478447:Comment:Message {id: 68719478447, creationDate: 1271884315385, locationIP: '62.217.119.183', browserUsed: 'Firefox', content: 'About Kevin Rudd, ia\'s remaining About William Morris, traditional teAbout France, c', length: 85})-[:HAS_CREATOR]->(p143), (m68719478447)-[:IS_LOCATED_IN]->(pl78), (m68719478447)-[:HAS_TAG]->(t61), (m68719478447)-[:HAS_TAG]->(t2076), (m68719478447)-[:HAS_TAG]->(t7555), (m68719478447)-[:HAS_TAG]->(t9147), (m68719478447)-[:HAS_TAG]->(t11035), (m68719478447)-[:HAS_TAG]->(t12742), (m68719478447)-[:REPLY_OF]->(m68719478445) +MATCH (m68719487322:Message {id: 68719487322}), (p143:Person {id: 143}), (pl78:Place {id: 78}), (t11644:Tag {id: 11644}), (t2805:Tag {id: 2805}), (t5100:Tag {id: 5100}), (t5417:Tag {id: 5417}), (t6959:Tag {id: 6959}), (t7010:Tag {id: 7010}) INSERT (m68719487324:Comment:Message {id: 68719487324, creationDate: 1271715688325, locationIP: '62.217.119.183', browserUsed: 'Firefox', content: 'About Woodrow Wilson, rimarily in tAbout Ecuador, mericas. EcuaAbout Hold ', length: 74})-[:HAS_CREATOR]->(p143), (m68719487324)-[:IS_LOCATED_IN]->(pl78), (m68719487324)-[:HAS_TAG]->(t2805), (m68719487324)-[:HAS_TAG]->(t5100), (m68719487324)-[:HAS_TAG]->(t5417), (m68719487324)-[:HAS_TAG]->(t6959), (m68719487324)-[:HAS_TAG]->(t7010), (m68719487324)-[:HAS_TAG]->(t11644), (m68719487324)-[:REPLY_OF]->(m68719487322) +MATCH (m68719487334:Message {id: 68719487334}), (p150:Person {id: 150}), (pl53:Place {id: 53}) INSERT (m68719487342:Comment:Message {id: 68719487342, creationDate: 1271910621171, locationIP: '148.240.94.143', browserUsed: 'Firefox', content: 'ok', length: 2})-[:HAS_CREATOR]->(p150), (m68719487342)-[:IS_LOCATED_IN]->(pl53), (m68719487342)-[:REPLY_OF]->(m68719487334) +MATCH (m137438963501:Message {id: 137438963501}), (p150:Person {id: 150}), (pl53:Place {id: 53}), (t7538:Tag {id: 7538}) INSERT (m137438963502:Comment:Message {id: 137438963502, creationDate: 1272758679985, locationIP: '148.240.94.143', browserUsed: 'Firefox', content: 'About Stephen Sondheim, an composer and lyricist known for his contributions to musical th', length: 90})-[:HAS_CREATOR]->(p150), (m137438963502)-[:IS_LOCATED_IN]->(pl53), (m137438963502)-[:HAS_TAG]->(t7538), (m137438963502)-[:REPLY_OF]->(m137438963501) +MATCH (m137438963506:Message {id: 137438963506}), (p2199023255712:Person {id: 2199023255712}), (pl55:Place {id: 55}) INSERT (m137438963510:Comment:Message {id: 137438963510, creationDate: 1272813417041, locationIP: '115.84.173.212', browserUsed: 'Chrome', content: 'great', length: 5})-[:HAS_CREATOR]->(p2199023255712), (m137438963510)-[:IS_LOCATED_IN]->(pl55), (m137438963510)-[:REPLY_OF]->(m137438963506) +MATCH (m274877909927:Message {id: 274877909927}), (p2199023255712:Person {id: 2199023255712}), (pl55:Place {id: 55}) INSERT (m274877909928:Comment:Message {id: 274877909928, creationDate: 1285681173411, locationIP: '115.84.173.212', browserUsed: 'Firefox', content: 'no way!', length: 7})-[:HAS_CREATOR]->(p2199023255712), (m274877909928)-[:IS_LOCATED_IN]->(pl55), (m274877909928)-[:REPLY_OF]->(m274877909927) +MATCH (m274877909927:Message {id: 274877909927}), (p143:Person {id: 143}), (pl78:Place {id: 78}) INSERT (m274877909929:Comment:Message {id: 274877909929, creationDate: 1285665988168, locationIP: '62.217.119.183', browserUsed: 'Firefox', content: 'I see', length: 5})-[:HAS_CREATOR]->(p143), (m274877909929)-[:IS_LOCATED_IN]->(pl78), (m274877909929)-[:REPLY_OF]->(m274877909927) +MATCH (m274877912146:Message {id: 274877912146}), (p102:Person {id: 102}), (pl84:Place {id: 84}) INSERT (m274877912148:Comment:Message {id: 274877912148, creationDate: 1284597772686, locationIP: '41.204.102.171', browserUsed: 'Safari', content: 'fine', length: 4})-[:HAS_CREATOR]->(p102), (m274877912148)-[:IS_LOCATED_IN]->(pl84), (m274877912148)-[:REPLY_OF]->(m274877912146) +MATCH (m343597392325:Message {id: 343597392325}), (p6597069766775:Person {id: 6597069766775}), (pl1:Place {id: 1}) INSERT (m343597392326:Comment:Message {id: 343597392326, creationDate: 1289092584789, locationIP: '1.4.5.93', browserUsed: 'Internet Explorer', content: 'no way!', length: 7})-[:HAS_CREATOR]->(p6597069766775), (m343597392326)-[:IS_LOCATED_IN]->(pl1), (m343597392326)-[:REPLY_OF]->(m343597392325) +MATCH (m343597393750:Message {id: 343597393750}), (p76:Person {id: 76}), (pl66:Place {id: 66}) INSERT (m343597393752:Comment:Message {id: 343597393752, creationDate: 1289136847840, locationIP: '64.40.245.250', browserUsed: 'Chrome', content: 'good', length: 4})-[:HAS_CREATOR]->(p76), (m343597393752)-[:IS_LOCATED_IN]->(pl66), (m343597393752)-[:REPLY_OF]->(m343597393750) +MATCH (m343597393755:Message {id: 343597393755}), (p150:Person {id: 150}), (pl53:Place {id: 53}) INSERT (m343597393763:Comment:Message {id: 343597393763, creationDate: 1289095711228, locationIP: '148.240.94.143', browserUsed: 'Firefox', content: 'roflol', length: 6})-[:HAS_CREATOR]->(p150), (m343597393763)-[:IS_LOCATED_IN]->(pl53), (m343597393763)-[:REPLY_OF]->(m343597393755) +MATCH (m137438963502:Message {id: 137438963502}), (p102:Person {id: 102}), (pl84:Place {id: 84}) INSERT (m137438963508:Comment:Message {id: 137438963508, creationDate: 1272758692818, locationIP: '41.204.102.171', browserUsed: 'Safari', content: 'no way!', length: 7})-[:HAS_CREATOR]->(p102), (m137438963508)-[:IS_LOCATED_IN]->(pl84), (m137438963508)-[:REPLY_OF]->(m137438963502) +MATCH (m10174:Message {id: 10174}), (p76:Person {id: 76}) INSERT (p76)-[:LIKES {creationDate: 1272582754260}]->(m10174) +MATCH (m137438955264:Message {id: 137438955264}), (p76:Person {id: 76}) INSERT (p76)-[:LIKES {creationDate: 1290453175260}]->(m137438955264) +MATCH (m68719478400:Message {id: 68719478400}), (p143:Person {id: 143}) INSERT (p143)-[:LIKES {creationDate: 1269973906897}]->(m68719478400) +MATCH (m206158440292:Message {id: 206158440292}), (p150:Person {id: 150}) INSERT (p150)-[:LIKES {creationDate: 1282198715860}]->(m206158440292) +MATCH (m68719478400:Message {id: 68719478400}), (p228:Person {id: 228}) INSERT (p228)-[:LIKES {creationDate: 1269984806390}]->(m68719478400) +MATCH (m10169:Message {id: 10169}), (p2199023255712:Person {id: 2199023255712}) INSERT (p2199023255712)-[:LIKES {creationDate: 1289546926226}]->(m10169) +MATCH (m10174:Message {id: 10174}), (p2199023255712:Person {id: 2199023255712}) INSERT (p2199023255712)-[:LIKES {creationDate: 1289488596321}]->(m10174) +MATCH (m68719486854:Message {id: 68719486854}), (p2199023255712:Person {id: 2199023255712}) INSERT (p2199023255712)-[:LIKES {creationDate: 1287721130777}]->(m68719486854) +MATCH (m274877909514:Message {id: 274877909514}), (p6597069766775:Person {id: 6597069766775}) INSERT (p6597069766775)-[:LIKES {creationDate: 1283615294371}]->(m274877909514) +MATCH (m274877909514:Message {id: 274877909514}), (p8796093022390:Person {id: 8796093022390}) INSERT (p8796093022390)-[:LIKES {creationDate: 1284618437114}]->(m274877909514) +MATCH (m343597392325:Message {id: 343597392325}), (p8796093022390:Person {id: 8796093022390}) INSERT (p8796093022390)-[:LIKES {creationDate: 1289061566907}]->(m343597392325) diff --git a/tests/spec/lpg/cypher/comprehensions_advanced.gtest b/tests/spec/lpg/cypher/comprehensions_advanced.gtest index 41fa7de91..2d5e5b963 100644 --- a/tests/spec/lpg/cypher/comprehensions_advanced.gtest +++ b/tests/spec/lpg/cypher/comprehensions_advanced.gtest @@ -38,7 +38,6 @@ tests: count: 7 - name: pattern_comprehension_size - skip: "pattern comprehension size() evaluation" query: | MATCH (a:Actor) RETURN a.name, size([(a)-[:ACTED_IN]->(:Movie) | 1]) AS movie_count @@ -46,13 +45,63 @@ tests: expect: ordered: true rows: - - [Beatrix, 3] + - [Alix, 2] + - [Beatrix, 2] + - [Butch, 2] - [Gus, 2] + - [Jules, 2] + - [Mia, 2] - [Vincent, 2] + + # A pattern comprehension inside an aggregate, grouped and not. + - name: pattern_comprehension_inside_an_aggregate + query: | + MATCH (a:Actor) + RETURN sum(size([(a)-[:ACTED_IN]->(:Movie) | 1])) AS roles + expect: + rows: + - [14] + + - name: pattern_comprehension_inside_a_grouped_aggregate + query: | + MATCH (a:Actor) + RETURN a.city AS city, sum(size([(a)-[:ACTED_IN]->(:Movie) | 1])) AS roles + ORDER BY city + expect: + ordered: true + rows: + - [Amsterdam, 4] + - [Barcelona, 2] + - [Berlin, 4] + - [Paris, 2] + - [Prague, 2] + + - name: pattern_comprehension_inside_with + query: | + MATCH (a:Actor) + WITH a, size([(a)-[:ACTED_IN]->(:Movie) | 1]) AS n + WHERE n = 2 + RETURN count(a) AS actors + expect: + rows: + - [7] + + # A pattern comprehension as the value of a map projection entry. + - name: pattern_comprehension_in_a_map_projection + query: | + MATCH (a:Actor) + RETURN a.name AS name, size(a{.name, roles: [(a)-[:ACTED_IN]->(:Movie) | 1]}.roles) AS roles + ORDER BY name + expect: + ordered: true + rows: - [Alix, 2] - - [Jules, 2] + - [Beatrix, 2] - [Butch, 2] + - [Gus, 2] + - [Jules, 2] - [Mia, 2] + - [Vincent, 2] - name: pattern_comprehension_with_property_extraction query: | diff --git a/tests/spec/lpg/cypher/expressions.gtest b/tests/spec/lpg/cypher/expressions.gtest index eac926ec2..7e1f3cd27 100644 --- a/tests/spec/lpg/cypher/expressions.gtest +++ b/tests/spec/lpg/cypher/expressions.gtest @@ -338,3 +338,12 @@ tests: expect: rows: - [true, false] + + # A map projection is a map: dotted access reads its keys + - name: dotted_access_on_a_map_projection + setup: + - "CREATE (:Person {name: 'Alix', age: 30})" + query: "MATCH (p:Person) RETURN p{.name}.name AS name, p{.name, next: p.age + 1}.next AS next" + expect: + rows: + - [Alix, 31] diff --git a/tests/spec/lpg/cypher/ldbc_snb_interactive.gtest b/tests/spec/lpg/cypher/ldbc_snb_interactive.gtest new file mode 100644 index 000000000..b3a8e9bdf --- /dev/null +++ b/tests/spec/lpg/cypher/ldbc_snb_interactive.gtest @@ -0,0 +1,1168 @@ +# LDBC SNB Interactive v1: the Cypher reference queries +# +# The queries of the LDBC Social Network Benchmark Interactive workload v1, as written in the Neo4j reference +# implementation (github.com/ldbc/ldbc_snb_interactive_v1_impls, cypher/queries). The read queries are unchanged +# apart from their parameter header comments; the updates replay events of the LDBC update stream, write their list +# parameters inline (gtest parameters are scalars) and are checked with a last statement. +# +# The expected rows follow Neo4j's semantics, quirks of the reference queries included: `CASE x WHEN null` never +# matches (null = null is null), so IS7 reports true for every reply. The GQL and SQL/PGQ translations +# (lpg/gql/ldbc_snb_interactive.gtest, lpg/sql_pgq/ldbc_snb_interactive.gtest) follow the LDBC specification. +# +# Dataset: ldbc_snb_mini, a small slice of the LDBC SNB Interactive v1 test data, every value copied unchanged (see +# its .setup file). The parameters are chosen from that slice; IC13 also uses the LDBC parameter pair (Gary Hill, +# Abdullah Koksal). The expected rows come from LDBC's own DuckDB (SQL) implementation run on the same slice, in the +# result shape of the Cypher reference queries. +# +# A skipped case names the bug it hits (and the issue, when there is one); remove the skip once it is fixed. + +meta: + language: cypher + model: lpg + section: ldbc + title: LDBC SNB Interactive v1 (Cypher reference queries) + dataset: ldbc_snb_mini + +tests: + + # --------------------------------------------------------------------------- + # Short reads + # --------------------------------------------------------------------------- + + # IS1. Profile of a person. + - name: is1_profile_of_a_person + params: + personId: 228 + query: | + MATCH (n:Person {id: $personId })-[:IS_LOCATED_IN]->(p:City) + RETURN + n.firstName AS firstName, + n.lastName AS lastName, + n.birthday AS birthday, + n.locationIP AS locationIP, + n.browserUsed AS browserUsed, + p.id AS cityId, + n.gender AS gender, + n.creationDate AS creationDate + expect: + ordered: true + rows: + - [Asher, Mamo, 524016000000, "213.55.93.153", Chrome, 1127, female, 1266720721912] + + # IS2. Recent messages of a person. + - name: is2_recent_messages_of_a_person + params: + personId: 228 + query: | + MATCH (:Person {id: $personId})<-[:HAS_CREATOR]-(message) + WITH + message, + message.id AS messageId, + message.creationDate AS messageCreationDate + ORDER BY messageCreationDate DESC, messageId ASC + LIMIT 10 + MATCH (message)-[:REPLY_OF*0..]->(post:Post), + (post)-[:HAS_CREATOR]->(person) + RETURN + messageId, + coalesce(message.imageFile,message.content) AS messageContent, + messageCreationDate, + post.id AS postId, + person.id AS personId, + person.firstName AS personFirstName, + person.lastName AS personLastName + ORDER BY messageCreationDate DESC, messageId ASC + expect: + ordered: true + rows: + - [343597393779, "photo343597393779.jpg", 1290590906265, 343597393779, 228, Asher, Mamo] + - [343597393778, "photo343597393778.jpg", 1290590905265, 343597393778, 228, Asher, Mamo] + - [343597393747, "About Luis Horna, as a strong serve for a relatively shoAbout Djibouti, uti National Army and its sub-branchesAbout Leonard Bernstein, hilharmonic, ", 1289066095205, 343597393747, 228, Asher, Mamo] + - [343597392325, "About Ne-Yo, November 22, 2010. InAbout Brian Wilson, elled for various reaAbout V", 1289049183991, 343597392324, 76, Jae-Jin, Park] + - [206158440292, "photo206158440292.jpg", 1279208419298, 206158440292, 228, Asher, Mamo] + - [206158431893, roflol, 1278531351427, 206158431892, 102, Philibert, Roindefo] + - [137438963511, "About William Lyon Mackenzie King, y. King worked to bring compromise andAbout Thomas Jefferson, erson, his wife", 1277076649771, 137438963511, 228, Asher, Mamo] + - [137438963744, "About Christopher Lee, or services to dramAbout Tina Turner, inning with a", 1273650346399, 137438963740, 150, Alfonso, Alvarez] + - [137438963499, "About William Lyon Mackenzie King, allowed his intense spirituality to distort About T", 1272740097725, 137438963499, 228, Asher, Mamo] + - [68719478457, "About Oscar Wilde, intellectuals. Their sAbout Anne, Queen of Great Britain, II became joint monarcAbout Lebanon, ld national infrastrucAbout Bad, Bad L", 1271850332021, 68719478453, 2199023255712, Aurora, Cruz] + + # IS3. Friends of a person. + - name: is3_friends_of_a_person + params: + personId: 228 + query: | + MATCH (n:Person {id: $personId })-[r:KNOWS]-(friend) + RETURN + friend.id AS personId, + friend.firstName AS firstName, + friend.lastName AS lastName, + r.creationDate AS friendshipCreationDate + ORDER BY + friendshipCreationDate DESC, + toInteger(personId) ASC + expect: + ordered: true + rows: + - [8796093022357, Gary, Hill, 1288580721183] + - [2199023255712, Aurora, Cruz, 1271536640884] + - [102, Philibert, Roindefo, 1268755084867] + - [76, Jae-Jin, Park, 1267889385714] + - [150, Alfonso, Alvarez, 1267126413921] + + # IS4. Content of a message. + - name: is4_content_of_a_message_post + params: + messageId: 10169 + query: | + MATCH (m:Message {id: $messageId }) + RETURN + m.creationDate as messageCreationDate, + coalesce(m.content, m.imageFile) as messageContent + expect: + ordered: true + rows: + - [1267290597839, "photo10169.jpg"] + + # IS4. Content of a message. + - name: is4_content_of_a_message_comment + params: + messageId: 343597392326 + query: | + MATCH (m:Message {id: $messageId }) + RETURN + m.creationDate as messageCreationDate, + coalesce(m.content, m.imageFile) as messageContent + expect: + ordered: true + rows: + - [1289092584789, "no way!"] + + # IS5. Creator of a message. + - name: is5_creator_of_a_message + params: + messageId: 343597392326 + query: | + MATCH (m:Message {id: $messageId })-[:HAS_CREATOR]->(p:Person) + RETURN + p.id AS personId, + p.firstName AS firstName, + p.lastName AS lastName + expect: + ordered: true + rows: + - [6597069766775, Jie, Yang] + + # IS6. Forum of a message. + - name: is6_forum_of_a_message + params: + messageId: 343597393752 + query: | + MATCH (m:Message {id: $messageId })-[:REPLY_OF*0..]->(p:Post)<-[:CONTAINER_OF]-(f:Forum)-[:HAS_MODERATOR]->(mod:Person) + RETURN + f.id AS forumId, + f.title AS forumTitle, + mod.id AS moderatorId, + mod.firstName AS moderatorFirstName, + mod.lastName AS moderatorLastName + expect: + ordered: true + rows: + - [872, "Wall of Asher Mamo", 228, Asher, Mamo] + + # IS7. Replies of a message. + - name: is7_replies_of_a_message + params: + messageId: 137438963499 + query: | + MATCH (m:Message {id: $messageId })<-[:REPLY_OF]-(c:Comment)-[:HAS_CREATOR]->(p:Person) + OPTIONAL MATCH (m)-[:HAS_CREATOR]->(a:Person)-[r:KNOWS]-(p) + RETURN c.id AS commentId, + c.content AS commentContent, + c.creationDate AS commentCreationDate, + p.id AS replyAuthorId, + p.firstName AS replyAuthorFirstName, + p.lastName AS replyAuthorLastName, + CASE r + WHEN null THEN false + ELSE true + END AS replyAuthorKnowsOriginalMessageAuthor + ORDER BY commentCreationDate DESC, replyAuthorId + expect: + ordered: true + rows: + - [137438963506, "About Thomas Hardy, y published as serials in magazines, weAbout Thomas Jeffe", 1272757138003, 150, Alfonso, Alvarez, true] + - [137438963501, "About Frank Zappa, ollages. His later albums shared this eclectic and experimental approach, irrespectiv", 1272755669544, 102, Philibert, Roindefo, true] + + + # --------------------------------------------------------------------------- + # Complex reads + # --------------------------------------------------------------------------- + + # IC1. Transitive friends with certain name. + - name: ic1_transitive_friends_with_a_name_maria + skip: "a list-valued grouping key next to collect() becomes 0 (friendUniversities)" + params: + personId: 228 + firstName: "Maria" + query: | + MATCH (p:Person {id: $personId}), (friend:Person {firstName: $firstName}) + WHERE NOT p=friend + WITH p, friend + MATCH path = shortestPath((p)-[:KNOWS*1..3]-(friend)) + WITH min(length(path)) AS distance, friend + ORDER BY + distance ASC, + friend.lastName ASC, + toInteger(friend.id) ASC + LIMIT 20 + + MATCH (friend)-[:IS_LOCATED_IN]->(friendCity:City) + OPTIONAL MATCH (friend)-[studyAt:STUDY_AT]->(uni:University)-[:IS_LOCATED_IN]->(uniCity:City) + WITH friend, collect( + CASE uni.name + WHEN null THEN null + ELSE [uni.name, studyAt.classYear, uniCity.name] + END ) AS unis, friendCity, distance + + OPTIONAL MATCH (friend)-[workAt:WORK_AT]->(company:Company)-[:IS_LOCATED_IN]->(companyCountry:Country) + WITH friend, collect( + CASE company.name + WHEN null THEN null + ELSE [company.name, workAt.workFrom, companyCountry.name] + END ) AS companies, unis, friendCity, distance + + RETURN + friend.id AS friendId, + friend.lastName AS friendLastName, + distance AS distanceFromPerson, + friend.birthday AS friendBirthday, + friend.creationDate AS friendCreationDate, + friend.gender AS friendGender, + friend.browserUsed AS friendBrowserUsed, + friend.locationIP AS friendLocationIp, + friend.email AS friendEmails, + friend.speaks AS friendLanguages, + friendCity.name AS friendCityName, + unis AS friendUniversities, + companies AS friendCompanies + ORDER BY + distanceFromPerson ASC, + friendLastName ASC, + toInteger(friendId) ASC + LIMIT 20 + expect: + ordered: true + rows: + - [143, Alkaios, 2, 410659200000, 1262456643976, female, Firefox, "62.217.119.183", "[Maria143@gmail.com]", "[fr, en]", Athens, "[[National_and_Kapodistrian_University_of_Athens, 2003, Athens]]", "[[Macedonian_Airlines, 2004, Greece]]"] + + # IC1. Transitive friends with certain name. + - name: ic1_transitive_friends_with_a_name_john + skip: "a list-valued grouping key next to collect() becomes 0 (friendUniversities)" + params: + personId: 228 + firstName: "John" + query: | + MATCH (p:Person {id: $personId}), (friend:Person {firstName: $firstName}) + WHERE NOT p=friend + WITH p, friend + MATCH path = shortestPath((p)-[:KNOWS*1..3]-(friend)) + WITH min(length(path)) AS distance, friend + ORDER BY + distance ASC, + friend.lastName ASC, + toInteger(friend.id) ASC + LIMIT 20 + + MATCH (friend)-[:IS_LOCATED_IN]->(friendCity:City) + OPTIONAL MATCH (friend)-[studyAt:STUDY_AT]->(uni:University)-[:IS_LOCATED_IN]->(uniCity:City) + WITH friend, collect( + CASE uni.name + WHEN null THEN null + ELSE [uni.name, studyAt.classYear, uniCity.name] + END ) AS unis, friendCity, distance + + OPTIONAL MATCH (friend)-[workAt:WORK_AT]->(company:Company)-[:IS_LOCATED_IN]->(companyCountry:Country) + WITH friend, collect( + CASE company.name + WHEN null THEN null + ELSE [company.name, workAt.workFrom, companyCountry.name] + END ) AS companies, unis, friendCity, distance + + RETURN + friend.id AS friendId, + friend.lastName AS friendLastName, + distance AS distanceFromPerson, + friend.birthday AS friendBirthday, + friend.creationDate AS friendCreationDate, + friend.gender AS friendGender, + friend.browserUsed AS friendBrowserUsed, + friend.locationIP AS friendLocationIp, + friend.email AS friendEmails, + friend.speaks AS friendLanguages, + friendCity.name AS friendCityName, + unis AS friendUniversities, + companies AS friendCompanies + ORDER BY + distanceFromPerson ASC, + friendLastName ASC, + toInteger(friendId) ASC + LIMIT 20 + expect: + ordered: true + rows: + - [41, Kumar, 3, 527731200000, 1266276257359, male, Safari, "27.116.33.147", "[John41@gmail.com, John41@jizan.cc, John41@yahoo.com, John41@zoho.com]", "[gu, mr, en]", Puttur, "[[The_Oxford_Educational_Institutions, 2004, Bangalore]]", "[[Jet_Airways, 2005, India], [Jagson_Airlines, 2005, India], [Deccan_360, 2006, India]]"] + + # IC2. Recent messages by your friends. + - name: ic2_recent_messages_by_friends + params: + personId: 228 + maxDate: 1285891200000 + query: | + MATCH (:Person {id: $personId })-[:KNOWS]-(friend:Person)<-[:HAS_CREATOR]-(message:Message) + WHERE message.creationDate <= $maxDate + RETURN + friend.id AS personId, + friend.firstName AS personFirstName, + friend.lastName AS personLastName, + message.id AS postOrCommentId, + coalesce(message.content,message.imageFile) AS postOrCommentContent, + message.creationDate AS postOrCommentCreationDate + ORDER BY + postOrCommentCreationDate DESC, + toInteger(postOrCommentId) ASC + LIMIT 20 + expect: + ordered: true + rows: + - [2199023255712, Aurora, Cruz, 274877909928, "no way!", 1285681173411] + - [76, Jae-Jin, Park, 274877909925, thx, 1285676860041] + - [102, Philibert, Roindefo, 274877912148, fine, 1284597772686] + - [76, Jae-Jin, Park, 206158432797, thx, 1280907556754] + - [102, Philibert, Roindefo, 206158431892, "About William Penn, – 16 September 1670) was an English admiral and politician who sat in the House of Commons from 1660 to 1670. He was the father of Wi", 1278511907643] + - [2199023255712, Aurora, Cruz, 206158432090, "photo206158432090.jpg", 1277597415173] + - [102, Philibert, Roindefo, 137438958564, cool, 1277084387523] + - [2199023255712, Aurora, Cruz, 137438963517, "I see", 1277076710526] + - [2199023255712, Aurora, Cruz, 137438955279, "photo137438955279.jpg", 1276553301898] + - [2199023255712, Aurora, Cruz, 137438955277, "photo137438955277.jpg", 1276553299898] + - [2199023255712, Aurora, Cruz, 137438955264, "photo137438955264.jpg", 1276274322999] + - [76, Jae-Jin, Park, 137438963765, right, 1273696787135] + - [150, Alfonso, Alvarez, 137438963740, "About Pope Paul VI, Church life during his pontificate excAbout Winston Churchill, United Kingdom during the Second WorldAbout Julia Gillard, ing Mitcham Demonstration School and UnAbout S", 1273649436061] + - [150, Alfonso, Alvarez, 137438963751, "About Julia Gillard, ister upon Labor's victory in the 2007 federal electionAbout Chile, Republic of Chile, is a country in South America occupyAbout South Korea, ith production focusing on electronics, automobiles, ", 1273617306061] + - [2199023255712, Aurora, Cruz, 137438963510, great, 1272813417041] + - [102, Philibert, Roindefo, 137438963508, "no way!", 1272758692818] + - [150, Alfonso, Alvarez, 137438963502, "About Stephen Sondheim, an composer and lyricist known for his contributions to musical th", 1272758679985] + - [150, Alfonso, Alvarez, 137438963506, "About Thomas Hardy, y published as serials in magazines, weAbout Thomas Jeffe", 1272757138003] + - [102, Philibert, Roindefo, 137438963501, "About Frank Zappa, ollages. His later albums shared this eclectic and experimental approach, irrespectiv", 1272755669544] + - [150, Alfonso, Alvarez, 68719487342, ok, 1271910621171] + + # IC3. Friends and friends of friends that have been to given countries. + - name: ic3_friends_in_countries_x_and_y + skip: "the chained comparison $endDate > message.creationDate >= $startDate returns null" + params: + personId: 228 + countryXName: "Uruguay" + countryYName: "Canada" + startDate: 1275350400000 + endDate: 1277942400000 + query: | + MATCH (countryX:Country {name: $countryXName }), + (countryY:Country {name: $countryYName }), + (person:Person {id: $personId }) + WITH person, countryX, countryY + LIMIT 1 + MATCH (city:City)-[:IS_PART_OF]->(country:Country) + WHERE country IN [countryX, countryY] + WITH person, countryX, countryY, collect(city) AS cities + MATCH (person)-[:KNOWS*1..2]-(friend)-[:IS_LOCATED_IN]->(city) + WHERE NOT person=friend AND NOT city IN cities + WITH DISTINCT friend, countryX, countryY + MATCH (friend)<-[:HAS_CREATOR]-(message), + (message)-[:IS_LOCATED_IN]->(country) + WHERE $endDate > message.creationDate >= $startDate AND + country IN [countryX, countryY] + WITH friend, + CASE WHEN country=countryX THEN 1 ELSE 0 END AS messageX, + CASE WHEN country=countryY THEN 1 ELSE 0 END AS messageY + WITH friend, sum(messageX) AS xCount, sum(messageY) AS yCount + WHERE xCount>0 AND yCount>0 + RETURN friend.id AS friendId, + friend.firstName AS friendFirstName, + friend.lastName AS friendLastName, + xCount, + yCount, + xCount + yCount AS xyCount + ORDER BY xyCount DESC, friendId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [2199023255712, Aurora, Cruz, 1, 2, 3] + + # IC4. New topics. + - name: ic4_new_topics + skip: "the chained comparison $startDate <= post.creationDate < $endDate returns null" + params: + personId: 228 + startDate: 1272672000000 + endDate: 1275264000000 + query: | + MATCH (person:Person {id: $personId })-[:KNOWS]-(friend:Person), + (friend)<-[:HAS_CREATOR]-(post:Post)-[:HAS_TAG]->(tag) + WITH DISTINCT tag, post + WITH tag, + CASE + WHEN $startDate <= post.creationDate < $endDate THEN 1 + ELSE 0 + END AS valid, + CASE + WHEN post.creationDate < $startDate THEN 1 + ELSE 0 + END AS inValid + WITH tag, sum(valid) AS postCount, sum(inValid) AS inValidPostCount + WHERE postCount>0 AND inValidPostCount=0 + RETURN tag.name AS tagName, postCount + ORDER BY postCount DESC, tagName ASC + LIMIT 10 + expect: + ordered: true + rows: + - [Julia_Gillard, 2] + - [Chile, 1] + - [Monaco, 1] + - [Nigeria, 1] + - [Pope_Paul_VI, 1] + - [South_Korea, 1] + - [South_Vietnam, 1] + - [Winston_Churchill, 1] + + # IC5. New groups. + - name: ic5_new_groups + skip: "OPTIONAL MATCH ... WHERE friend IN friends (a list from the WITH before it) drops the forums without a match instead of keeping them with a count of 0" + params: + personId: 228 + minDate: 1275350400000 + query: | + MATCH (person:Person { id: $personId })-[:KNOWS*1..2]-(friend) + WHERE + NOT person=friend + WITH DISTINCT friend + MATCH (friend)<-[membership:HAS_MEMBER]-(forum) + WHERE + membership.joinDate > $minDate + WITH + forum, + collect(friend) AS friends + OPTIONAL MATCH (friend)<-[:HAS_CREATOR]-(post)<-[:CONTAINER_OF]-(forum) + WHERE + friend IN friends + WITH + forum, + count(post) AS postCount + RETURN + forum.title AS forumName, + postCount + ORDER BY + postCount DESC, + forum.id ASC + LIMIT 20 + expect: + ordered: true + rows: + - ["Wall of Maria Alkaios", 0] + - ["Wall of Jae-Jin Park", 0] + - ["Wall of Asher Mamo", 0] + - ["Album 9 of Asher Mamo", 0] + - ["Wall of Alfonso Alvarez", 0] + - ["Wall of Abdala Ndiaye", 0] + - ["Wall of Aurora Cruz", 0] + - ["Album 3 of Asher Mamo", 0] + - ["Album 1 of Aurora Cruz", 0] + - ["Album 2 of Aurora Cruz", 0] + - ["Wall of Rafael Fernández", 0] + - ["Album 7 of Aurora Cruz", 0] + - ["Wall of Jie Yang", 0] + - ["Album 0 of Asher Mamo", 0] + - ["Album 6 of Alfonso Alvarez", 0] + - ["Album 8 of Abdala Ndiaye", 0] + + # IC6. Tag co-occurrence. + - name: ic6_tag_co_occurrence + skip: "a comma-separated MATCH whose second pattern uses a variable from an earlier WITH ({id: knownTagId}) returns no rows" + params: + personId: 228 + tagName: "Luis_Horna" + query: | + MATCH (knownTag:Tag { name: $tagName }) + WITH knownTag.id as knownTagId + + MATCH (person:Person { id: $personId })-[:KNOWS*1..2]-(friend) + WHERE NOT person=friend + WITH + knownTagId, + collect(distinct friend) as friends + UNWIND friends as f + MATCH (f)<-[:HAS_CREATOR]-(post:Post), + (post)-[:HAS_TAG]->(t:Tag{id: knownTagId}), + (post)-[:HAS_TAG]->(tag:Tag) + WHERE NOT t = tag + WITH + tag.name as tagName, + count(post) as postCount + RETURN + tagName, + postCount + ORDER BY + postCount DESC, + tagName ASC + LIMIT 10 + expect: + ordered: true + rows: + - [Brian_Wilson, 1] + - [Gibraltar, 1] + - ["Harry_S._Truman", 1] + - [Israel, 1] + - [Republic_of_the_Congo, 1] + - [Robert_Altman, 1] + - [Xiongnu, 1] + + # IC7. Recent likers. + - name: ic7_recent_likers + skip: "a node stored in a map loses its kind: latestLike.msg.id is null" + params: + personId: 228 + query: | + MATCH (person:Person {id: $personId})<-[:HAS_CREATOR]-(message:Message)<-[like:LIKES]-(liker:Person) + WITH liker, message, like.creationDate AS likeTime, person + ORDER BY likeTime DESC, toInteger(message.id) ASC + WITH liker, head(collect({msg: message, likeTime: likeTime})) AS latestLike, person + RETURN + liker.id AS personId, + liker.firstName AS personFirstName, + liker.lastName AS personLastName, + latestLike.likeTime AS likeCreationDate, + latestLike.msg.id AS commentOrPostId, + coalesce(latestLike.msg.content, latestLike.msg.imageFile) AS commentOrPostContent, + toInteger(floor(toFloat(latestLike.likeTime - latestLike.msg.creationDate)/1000.0)/60.0) AS minutesLatency, + not((liker)-[:KNOWS]-(person)) AS isNew + ORDER BY + likeCreationDate DESC, + toInteger(personId) ASC + LIMIT 20 + expect: + ordered: true + rows: + - [2199023255712, Aurora, Cruz, 1289546926226, 10169, "photo10169.jpg", 370938, false] + - [8796093022390, Abdullah, Koksal, 1289061566907, 343597392325, "About Ne-Yo, November 22, 2010. InAbout Brian Wilson, elled for various reaAbout V", 206, true] + - [150, Alfonso, Alvarez, 1282198715860, 206158440292, "photo206158440292.jpg", 49838, false] + - [76, Jae-Jin, Park, 1272582754260, 10174, "photo10174.jpg", 88202, false] + + # IC8. Recent replies. + - name: ic8_recent_replies + params: + personId: 228 + query: | + MATCH (start:Person {id: $personId})<-[:HAS_CREATOR]-(:Message)<-[:REPLY_OF]-(comment:Comment)-[:HAS_CREATOR]->(person:Person) + RETURN + person.id AS personId, + person.firstName AS personFirstName, + person.lastName AS personLastName, + comment.creationDate AS commentCreationDate, + comment.id AS commentId, + comment.content AS commentContent + ORDER BY + commentCreationDate DESC, + commentId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [76, Jae-Jin, Park, 1289135518830, 343597393750, "About Luis Horna, e surface is clay. He was the About Waiting for the End, ough it was "] + - [6597069766775, Jie, Yang, 1289092584789, 343597392326, "no way!"] + - [102, Philibert, Roindefo, 1289090250676, 343597393755, "About Al Capone, ing the money he made from his actAbout Djibouti, military forces of Djibouti and coAbout Leonard Bernstein, talent"] + - [2199023255712, Aurora, Cruz, 1277076710526, 137438963517, "I see"] + - [150, Alfonso, Alvarez, 1272757138003, 137438963506, "About Thomas Hardy, y published as serials in magazines, weAbout Thomas Jeffe"] + - [102, Philibert, Roindefo, 1272755669544, 137438963501, "About Frank Zappa, ollages. His later albums shared this eclectic and experimental approach, irrespectiv"] + - [143, Maria, Alkaios, 1271884315385, 68719478447, "About Kevin Rudd, ia's remaining About William Morris, traditional teAbout France, c"] + + # IC9. Recent messages by friends or friends of friends. + - name: ic9_recent_messages_by_friends_of_friends + params: + personId: 228 + maxDate: 1285891200000 + query: | + MATCH (root:Person {id: $personId })-[:KNOWS*1..2]-(friend:Person) + WHERE NOT friend = root + WITH collect(distinct friend) as friends + UNWIND friends as friend + MATCH (friend)<-[:HAS_CREATOR]-(message:Message) + WHERE message.creationDate < $maxDate + RETURN + friend.id AS personId, + friend.firstName AS personFirstName, + friend.lastName AS personLastName, + message.id AS commentOrPostId, + coalesce(message.content,message.imageFile) AS commentOrPostContent, + message.creationDate AS commentOrPostCreationDate + ORDER BY + commentOrPostCreationDate DESC, + message.id ASC + LIMIT 20 + expect: + ordered: true + rows: + - [143, Maria, Alkaios, 274877909920, roflol, 1285685102274] + - [6597069766775, Jie, Yang, 274877909919, "About Manuel Noriega, uest in April 2010. He arrived in Paris on April 27, 2010, and after a re-trial as a condition", 1285682584114] + - [2199023255712, Aurora, Cruz, 274877909928, "no way!", 1285681173411] + - [76, Jae-Jin, Park, 274877909925, thx, 1285676860041] + - [143, Maria, Alkaios, 274877909929, "I see", 1285665988168] + - [4398046511333, Rafael, "Fernández", 274877909927, "About Manuel Noriega, trafficking, racketeering, andAbout Mikhail Gorbachev, was the onl", 1285665306179] + - [6597069766775, Jie, Yang, 274877909924, "About Manuel Noriega, g, racketeering, and money laundering About Arthur Wellesley, 1st Duke of Wellington, m. He", 1285664359114] + - [4398046511333, Rafael, "Fernández", 274877912142, "no", 1284613964426] + - [102, Philibert, Roindefo, 274877912148, fine, 1284597772686] + - [143, Maria, Alkaios, 274877912139, "About Enrique Iglesias, hits on the various Billboard charts. Billboard has called him The King of Latin Pop and The King of DanAbout California King Bed, ive reviews from music critics, who praised Rihanna", 1284593996258] + - [4398046511333, Rafael, "Fernández", 274877909514, "About Guy Sebastian, 08 Australian tour. Like It Like That has three tracks with John Mayer o", 1283465660488] + - [143, Maria, Alkaios, 274877913531, right, 1283303511849] + - [153, Abdala, Ndiaye, 206158432792, duh, 1280914332302] + - [76, Jae-Jin, Park, 206158432797, thx, 1280907556754] + - [4398046511333, Rafael, "Fernández", 206158432782, "About Guy Sebastian, ur Asian countries and New Zealand. Sebastian had a second number one in New Zealand with Who's That Girl, two other top ten singles and a number three album, and gained four platinum and two gold certifications there. He ha", 1280894717004] + - [143, Maria, Alkaios, 206158431896, "yes", 1278512054636] + - [102, Philibert, Roindefo, 206158431892, "About William Penn, – 16 September 1670) was an English admiral and politician who sat in the House of Commons from 1660 to 1670. He was the father of Wi", 1278511907643] + - [2199023255712, Aurora, Cruz, 206158432090, "photo206158432090.jpg", 1277597415173] + - [102, Philibert, Roindefo, 137438958564, cool, 1277084387523] + - [2199023255712, Aurora, Cruz, 137438963517, "I see", 1277076710526] + + # IC10. Friend recommendation. + - name: ic10_friend_recommendation + skip: "#543: a pattern predicate in a list comprehension fails with Unsupported EXISTS subquery pattern; then datetime({epochMillis: ...}) and birthday.month return null" + params: + personId: 228 + month: 7 + query: | + MATCH (person:Person {id: $personId})-[:KNOWS*2..2]-(friend), + (friend)-[:IS_LOCATED_IN]->(city:City) + WHERE NOT friend=person AND + NOT (friend)-[:KNOWS]-(person) + WITH person, city, friend, datetime({epochMillis: friend.birthday}) as birthday + WHERE (birthday.month=$month AND birthday.day>=21) OR + (birthday.month=($month%12)+1 AND birthday.day<22) + WITH DISTINCT friend, city, person + OPTIONAL MATCH (friend)<-[:HAS_CREATOR]-(post:Post) + WITH friend, city, collect(post) AS posts, person + WITH friend, + city, + size(posts) AS postCount, + size([p IN posts WHERE (p)-[:HAS_TAG]->()<-[:HAS_INTEREST]-(person)]) AS commonPostCount + RETURN friend.id AS personId, + friend.firstName AS personFirstName, + friend.lastName AS personLastName, + commonPostCount - (postCount - commonPostCount) AS commonInterestScore, + friend.gender AS personGender, + city.name AS personCityName + ORDER BY commonInterestScore DESC, personId ASC + LIMIT 10 + expect: + ordered: true + rows: + - [10995116277918, Javed, Khan, 0, male, Major_Cities] + - [4398046511333, Rafael, "Fernández", -2, female, Barcelona] + + # IC11. Job referral. + - name: ic11_job_referral + params: + personId: 228 + countryName: "Philippines" + workFromYear: 2008 + query: | + MATCH (person:Person {id: $personId })-[:KNOWS*1..2]-(friend:Person) + WHERE not(person=friend) + WITH DISTINCT friend + MATCH (friend)-[workAt:WORK_AT]->(company:Company)-[:IS_LOCATED_IN]->(:Country {name: $countryName }) + WHERE workAt.workFrom < $workFromYear + RETURN + friend.id AS personId, + friend.firstName AS personFirstName, + friend.lastName AS personLastName, + company.name AS organizationName, + workAt.workFrom AS organizationWorkFromYear + ORDER BY + organizationWorkFromYear ASC, + toInteger(personId) ASC, + organizationName DESC + LIMIT 10 + expect: + ordered: true + rows: + - [2199023255712, Aurora, Cruz, South_East_Asian_Airlines, 2007] + - [2199023255712, Aurora, Cruz, Filipinas_Orient_Airways, 2007] + + # IC12. Expert search. + - name: ic12_expert_search_office_holder + params: + personId: 228 + tagClassName: "OfficeHolder" + query: | + MATCH (tag:Tag)-[:HAS_TYPE|IS_SUBCLASS_OF*0..]->(baseTagClass:TagClass) + WHERE tag.name = $tagClassName OR baseTagClass.name = $tagClassName + WITH collect(tag.id) as tags + MATCH (:Person {id: $personId })-[:KNOWS]-(friend:Person)<-[:HAS_CREATOR]-(comment:Comment)-[:REPLY_OF]->(:Post)-[:HAS_TAG]->(tag:Tag) + WHERE tag.id in tags + RETURN + friend.id AS personId, + friend.firstName AS personFirstName, + friend.lastName AS personLastName, + collect(DISTINCT tag.name) AS tagNames, + count(DISTINCT comment) AS replyCount + ORDER BY + replyCount DESC, + toInteger(personId) ASC + LIMIT 20 + expect: + ordered: true + rows: + - [76, Jae-Jin, Park, "[Julia_Gillard, Arthur_Wellesley,_1st_Duke_of_Wellington]", 2] + - [102, Philibert, Roindefo, "[Aung_San_Suu_Kyi, William_Lyon_Mackenzie_King, Thomas_Jefferson]", 2] + - [150, Alfonso, Alvarez, "[William_Lyon_Mackenzie_King, Thomas_Jefferson]", 1] + - [2199023255712, Aurora, Cruz, "[William_Lyon_Mackenzie_King, Thomas_Jefferson]", 1] + + # IC12. Expert search. + - name: ic12_expert_search_tennis_player + params: + personId: 228 + tagClassName: "TennisPlayer" + query: | + MATCH (tag:Tag)-[:HAS_TYPE|IS_SUBCLASS_OF*0..]->(baseTagClass:TagClass) + WHERE tag.name = $tagClassName OR baseTagClass.name = $tagClassName + WITH collect(tag.id) as tags + MATCH (:Person {id: $personId })-[:KNOWS]-(friend:Person)<-[:HAS_CREATOR]-(comment:Comment)-[:REPLY_OF]->(:Post)-[:HAS_TAG]->(tag:Tag) + WHERE tag.id in tags + RETURN + friend.id AS personId, + friend.firstName AS personFirstName, + friend.lastName AS personLastName, + collect(DISTINCT tag.name) AS tagNames, + count(DISTINCT comment) AS replyCount + ORDER BY + replyCount DESC, + toInteger(personId) ASC + LIMIT 20 + expect: + ordered: true + rows: + - [76, Jae-Jin, Park, "[Dudi_Sela, Luis_Horna]", 2] + - [150, Alfonso, Alvarez, "[Dudi_Sela, Venus_Williams]", 2] + - [102, Philibert, Roindefo, "[Luis_Horna]", 1] + - [2199023255712, Aurora, Cruz, "[Luis_Horna]", 1] + + # IC13. Single shortest path. + - name: ic13_single_shortest_path_ldbc_pair + skip: "#318: path IS NULL is true for a shortestPath path, so the query returns -1" + params: + person1Id: 8796093022357 + person2Id: 8796093022390 + query: | + MATCH + (person1:Person {id: $person1Id}), + (person2:Person {id: $person2Id}), + path = shortestPath((person1)-[:KNOWS*]-(person2)) + RETURN + CASE path IS NULL + WHEN true THEN -1 + ELSE length(path) + END AS shortestPathLength + expect: + ordered: true + rows: + - [2] + + # IC13. Single shortest path. + - name: ic13_single_shortest_path_three_hops + skip: "#318: path IS NULL is true for a shortestPath path, so the query returns -1" + params: + person1Id: 228 + person2Id: 41 + query: | + MATCH + (person1:Person {id: $person1Id}), + (person2:Person {id: $person2Id}), + path = shortestPath((person1)-[:KNOWS*]-(person2)) + RETURN + CASE path IS NULL + WHEN true THEN -1 + ELSE length(path) + END AS shortestPathLength + expect: + ordered: true + rows: + - [3] + + # IC14. Trusted connection paths. + - name: ic14_trusted_connection_paths + skip: "startNode(r).id is rejected (property access on a function result); #318: the path of allShortestPaths with *0.. is not bound" + params: + person1Id: 228 + person2Id: 41 + query: | + MATCH path = allShortestPaths((person1:Person { id: $person1Id })-[:KNOWS*0..]-(person2:Person { id: $person2Id })) + WITH collect(path) as paths + UNWIND paths as path + WITH path, relationships(path) as rels_in_path + WITH + [n in nodes(path) | n.id ] as personIdsInPath, + [r in rels_in_path | + reduce(w=0.0, v in [ + (a:Person)<-[:HAS_CREATOR]-(:Comment)-[:REPLY_OF]->(:Post)-[:HAS_CREATOR]->(b:Person) + WHERE + (a.id = startNode(r).id and b.id=endNode(r).id) OR (a.id=endNode(r).id and b.id=startNode(r).id) + | 1.0] | w+v) + ] as weight1, + [r in rels_in_path | + reduce(w=0.0,v in [ + (a:Person)<-[:HAS_CREATOR]-(:Comment)-[:REPLY_OF]->(:Comment)-[:HAS_CREATOR]->(b:Person) + WHERE + (a.id = startNode(r).id and b.id=endNode(r).id) OR (a.id=endNode(r).id and b.id=startNode(r).id) + | 0.5] | w+v) + ] as weight2 + WITH + personIdsInPath, + reduce(w=0.0,v in weight1| w+v) as w1, + reduce(w=0.0,v in weight2| w+v) as w2 + RETURN + personIdsInPath, + (w1+w2) as pathWeight + ORDER BY pathWeight desc + expect: + ordered: true + precision: 6 + rows: + - ["[228, 102, 143, 41]", 10.0] + - ["[228, 8796093022357, 143, 41]", 3.0] + + + # --------------------------------------------------------------------------- + # Updates + # --------------------------------------------------------------------------- + + # IU1. Add person: Isabel Garcia, from the LDBC update stream (lists written inline). + # IU1 stores the languages as `languages` (IC1 reads `speaks`). + - name: iu1_add_person + params: + cityId: 801 + personId: 10995116277976 + personFirstName: "Isabel" + personLastName: "Garcia" + gender: "female" + birthday: 421545600000 + creationDate: 1291607408799 + locationIP: "103.4.23.146" + browserUsed: "Firefox" + statements: + - | + MATCH (c:City {id: $cityId}) + CREATE (p:Person { + id: $personId, + firstName: $personFirstName, + lastName: $personLastName, + gender: $gender, + birthday: $birthday, + creationDate: $creationDate, + locationIP: $locationIP, + browserUsed: $browserUsed, + languages: ['en'], + email: ['Isabel10995116277976@gmx.com', 'Isabel10995116277976@yahoo.com', 'Isabel10995116277976@zoho.com'] + })-[:IS_LOCATED_IN]->(c) + WITH p, count(*) AS dummy1 + UNWIND [26] AS tagId + MATCH (t:Tag {id: tagId}) + CREATE (p)-[:HAS_INTEREST]->(t) + WITH p, count(*) AS dummy2 + UNWIND [[5522, 2003]] AS s + MATCH (u:Organisation {id: s[0]}) + CREATE (p)-[:STUDY_AT {classYear: s[1]}]->(u) + WITH p, count(*) AS dummy3 + UNWIND [[965, 2004], [955, 2005], [957, 2004]] AS w + MATCH (comp:Organisation {id: w[0]}) + CREATE (p)-[:WORKS_AT {workFrom: w[1]}]->(comp) + - | + MATCH (p:Person {id: 10995116277976})-[:IS_LOCATED_IN]->(c:City) + RETURN p.firstName, p.lastName, p.gender, p.birthday, p.browserUsed, p.languages, p.email, c.name + expect: + rows: + - [Isabel, Garcia, female, 421545600000, Firefox, "[en]", "[Isabel10995116277976@gmx.com, Isabel10995116277976@yahoo.com, Isabel10995116277976@zoho.com]", Cebu_City] + + # IU1. Add person: the edges (IU1 creates WORKS_AT, not WORK_AT). + - name: iu1_add_person_edges + params: + cityId: 801 + personId: 10995116277976 + personFirstName: "Isabel" + personLastName: "Garcia" + gender: "female" + birthday: 421545600000 + creationDate: 1291607408799 + locationIP: "103.4.23.146" + browserUsed: "Firefox" + statements: + - | + MATCH (c:City {id: $cityId}) + CREATE (p:Person { + id: $personId, + firstName: $personFirstName, + lastName: $personLastName, + gender: $gender, + birthday: $birthday, + creationDate: $creationDate, + locationIP: $locationIP, + browserUsed: $browserUsed, + languages: ['en'], + email: ['Isabel10995116277976@gmx.com', 'Isabel10995116277976@yahoo.com', 'Isabel10995116277976@zoho.com'] + })-[:IS_LOCATED_IN]->(c) + WITH p, count(*) AS dummy1 + UNWIND [26] AS tagId + MATCH (t:Tag {id: tagId}) + CREATE (p)-[:HAS_INTEREST]->(t) + WITH p, count(*) AS dummy2 + UNWIND [[5522, 2003]] AS s + MATCH (u:Organisation {id: s[0]}) + CREATE (p)-[:STUDY_AT {classYear: s[1]}]->(u) + WITH p, count(*) AS dummy3 + UNWIND [[965, 2004], [955, 2005], [957, 2004]] AS w + MATCH (comp:Organisation {id: w[0]}) + CREATE (p)-[:WORKS_AT {workFrom: w[1]}]->(comp) + - | + MATCH (p:Person {id: 10995116277976})-[r]->(x) + RETURN type(r), x.id, coalesce(r.classYear, r.workFrom) + expect: + rows: + - [HAS_INTEREST, 26, null] + - [IS_LOCATED_IN, 801, null] + - [STUDY_AT, 5522, 2003] + - [WORKS_AT, 955, 2005] + - [WORKS_AT, 957, 2004] + - [WORKS_AT, 965, 2004] + + # IU2. Add like to post: Asher likes a post of Aurora (LDBC update stream). + - name: iu2_add_like_to_post + params: + personId: 228 + postId: 137438955264 + creationDate: 1293549908486 + statements: + - | + MATCH (person:Person {id: $personId}), (post:Post {id: $postId}) + CREATE (person)-[:LIKES {creationDate: $creationDate}]->(post) + - | + MATCH (:Person {id: 228})-[l:LIKES]->(m:Post {id: 137438955264}) RETURN m.id, l.creationDate + expect: + rows: + - [137438955264, 1293549908486] + + # IU3. Add like to comment. No event of the LDBC update stream fits the dataset: + # Gary Hill likes a comment of Jie Yang. + - name: iu3_add_like_to_comment + params: + personId: 8796093022357 + commentId: 343597392326 + creationDate: 1290687902110 + statements: + - | + MATCH (person:Person {id: $personId}), (comment:Comment {id: $commentId}) + CREATE (person)-[:LIKES {creationDate: $creationDate}]->(comment) + - | + MATCH (:Person {id: 8796093022357})-[l:LIKES]->(c:Comment) RETURN c.id, l.creationDate + expect: + rows: + - [343597392326, 1290687902110] + + # IU4. Add forum: Album 2 of John Kumar, from the LDBC update stream (tag list inline). + - name: iu4_add_forum + params: + moderatorPersonId: 41 + forumId: 343597384372 + forumTitle: "Album 2 of John Kumar" + creationDate: 1291170270978 + statements: + - | + MATCH (p:Person {id: $moderatorPersonId}) + CREATE (f:Forum {id: $forumId, title: $forumTitle, creationDate: $creationDate})-[:HAS_MODERATOR]->(p) + WITH f + UNWIND [1410] AS tagId + MATCH (t:Tag {id: tagId}) + CREATE (f)-[:HAS_TAG]->(t) + - | + MATCH (x)-[r]->(y) WHERE x.id = 343597384372 OR y.id = 343597384372 RETURN x.id, type(r), y.id + expect: + rows: + - [343597384372, HAS_MODERATOR, 41] + - [343597384372, HAS_TAG, 1410] + + # IU5. Add forum membership: Jae-Jin joins a forum (LDBC update stream). + - name: iu5_add_forum_membership + params: + forumId: 206158431081 + personId: 76 + joinDate: 1291175534625 + statements: + - | + MATCH (f:Forum {id: $forumId}), (p:Person {id: $personId}) + CREATE (f)-[:HAS_MEMBER {joinDate: $joinDate}]->(p) + - | + MATCH (:Forum {id: 206158431081})-[h:HAS_MEMBER]->(p:Person {id: 76}) RETURN p.id, h.joinDate + expect: + rows: + - [76, 1291175534625] + + # IU6. Add post: Jae-Jin posts about Emilio Aguinaldo (LDBC update stream, tag list inline). + # The left arrows must store author <- post <- forum. + - name: iu6_add_post + skip: "CREATE (a)<-[:T]-(b) stores the edge as a -> b (HAS_CREATOR and CONTAINER_OF are reversed)" + params: + authorPersonId: 76 + countryId: 98 + forumId: 767 + postId: 412316868991 + creationDate: 1292734467726 + locationIP: "27.35.111.48" + browserUsed: "Chrome" + language: "uz" + content: "About Emilio Aguinaldo, nstrumental role during the Philippines' revolution against Spain, and t" + imageFile: "" + length: 96 + statements: + - | + MATCH (author:Person {id: $authorPersonId}), (country:Country {id: $countryId}), (forum:Forum {id: $forumId}) + CREATE (author)<-[:HAS_CREATOR]-(p:Post:Message { + id: $postId, + creationDate: $creationDate, + locationIP: $locationIP, + browserUsed: $browserUsed, + language: $language, + content: CASE $content WHEN '' THEN NULL ELSE $content END, + imageFile: CASE $imageFile WHEN '' THEN NULL ELSE $imageFile END, + length: $length + })<-[:CONTAINER_OF]-(forum), (p)-[:IS_LOCATED_IN]->(country) + WITH p + UNWIND [1538] AS tagId + MATCH (t:Tag {id: tagId}) + CREATE (p)-[:HAS_TAG]->(t) + - | + MATCH (x)-[r]->(y) WHERE x.id = 412316868991 OR y.id = 412316868991 RETURN x.id, type(r), y.id + expect: + rows: + - [767, CONTAINER_OF, 412316868991] + - [412316868991, HAS_CREATOR, 76] + - [412316868991, HAS_TAG, 1538] + - [412316868991, IS_LOCATED_IN, 98] + + # IU6. Add post: the properties (an empty imageFile is stored as null). + - name: iu6_add_post_properties + params: + authorPersonId: 76 + countryId: 98 + forumId: 767 + postId: 412316868991 + creationDate: 1292734467726 + locationIP: "27.35.111.48" + browserUsed: "Chrome" + language: "uz" + content: "About Emilio Aguinaldo, nstrumental role during the Philippines' revolution against Spain, and t" + imageFile: "" + length: 96 + statements: + - | + MATCH (author:Person {id: $authorPersonId}), (country:Country {id: $countryId}), (forum:Forum {id: $forumId}) + CREATE (author)<-[:HAS_CREATOR]-(p:Post:Message { + id: $postId, + creationDate: $creationDate, + locationIP: $locationIP, + browserUsed: $browserUsed, + language: $language, + content: CASE $content WHEN '' THEN NULL ELSE $content END, + imageFile: CASE $imageFile WHEN '' THEN NULL ELSE $imageFile END, + length: $length + })<-[:CONTAINER_OF]-(forum), (p)-[:IS_LOCATED_IN]->(country) + WITH p + UNWIND [1538] AS tagId + MATCH (t:Tag {id: tagId}) + CREATE (p)-[:HAS_TAG]->(t) + - | + MATCH (p:Post:Message {id: 412316868991}) RETURN p.content, p.imageFile, p.language, p.length, p.browserUsed + expect: + rows: + - ["About Emilio Aguinaldo, nstrumental role during the Philippines' revolution against Spain, and t", null, uz, 96, Chrome] + + # IU7. Add comment: Jie Yang replies to the post of IU6 (LDBC update stream, tag list + # inline). The message id is $replyToPostId + $replyToCommentId + 1. + - name: iu7_add_comment + skip: "CREATE (a)<-[:T]-(b) stores the edge as a -> b (HAS_CREATOR is reversed)" + setup: + - | + MATCH (author:Person {id: 76}), (country:Country {id: 98}), (forum:Forum {id: 767}) + CREATE (author)<-[:HAS_CREATOR]-(p:Post:Message { + id: 412316868991, + creationDate: 1292734467726, + locationIP: '27.35.111.48', + browserUsed: 'Chrome', + language: 'uz', + content: CASE 'About Emilio Aguinaldo, nstrumental role during the Philippines\' revolution against Spain, and t' WHEN '' THEN NULL ELSE 'About Emilio Aguinaldo, nstrumental role during the Philippines\' revolution against Spain, and t' END, + imageFile: CASE '' WHEN '' THEN NULL ELSE '' END, + length: 96 + })<-[:CONTAINER_OF]-(forum), (p)-[:IS_LOCATED_IN]->(country) + WITH p + UNWIND [1538] AS tagId + MATCH (t:Tag {id: tagId}) + CREATE (p)-[:HAS_TAG]->(t) + params: + authorPersonId: 6597069766775 + countryId: 1 + replyToPostId: 412316868991 + replyToCommentId: -1 + commentId: 412316868996 + creationDate: 1292750579384 + locationIP: "1.4.5.93" + browserUsed: "Internet Explorer" + content: "About Joss Whedon, film Dr. Horrible's Sing-Along Blog (2008). Whedon co-wrote and produced the horror film The Cabin in the Woods (2012), and wrote and directed the f" + length: 168 + statements: + - | + MATCH + (author:Person {id: $authorPersonId}), + (country:Country {id: $countryId}), + (message:Message {id: $replyToPostId + $replyToCommentId + 1}) // $replyToCommentId is -1 if the message is a reply to a post and vica versa (see spec) + CREATE (author)<-[:HAS_CREATOR]-(c:Comment:Message { + id: $commentId, + creationDate: $creationDate, + locationIP: $locationIP, + browserUsed: $browserUsed, + content: $content, + length: $length + })-[:REPLY_OF]->(message), + (c)-[:IS_LOCATED_IN]->(country) + WITH c + UNWIND [3075] AS tagId + MATCH (t:Tag {id: tagId}) + CREATE (c)-[:HAS_TAG]->(t) + - | + MATCH (x)-[r]->(y) WHERE x.id = 412316868996 OR y.id = 412316868996 RETURN x.id, type(r), y.id + expect: + rows: + - [412316868996, HAS_CREATOR, 6597069766775] + - [412316868996, HAS_TAG, 3075] + - [412316868996, IS_LOCATED_IN, 1] + - [412316868996, REPLY_OF, 412316868991] + + # IU8. Add friendship: Gary Hill and Javed Khan (LDBC update stream). + - name: iu8_add_friendship + params: + person1Id: 8796093022357 + person2Id: 10995116277918 + creationDate: 1290895862836 + statements: + - | + MATCH (p1:Person {id: $person1Id}), (p2:Person {id: $person2Id}) + CREATE (p1)-[:KNOWS {creationDate: $creationDate}]->(p2) + - | + MATCH (:Person {id: 8796093022357})-[k:KNOWS]->(p:Person {id: 10995116277918}) RETURN p.id, k.creationDate + expect: + rows: + - [10995116277918, 1290895862836] diff --git a/tests/spec/lpg/gql/14.1_next.gtest b/tests/spec/lpg/gql/14.1_next.gtest new file mode 100644 index 000000000..5865ffd15 --- /dev/null +++ b/tests/spec/lpg/gql/14.1_next.gtest @@ -0,0 +1,110 @@ +# GQL linear composition: NEXT +# +# The statement after NEXT reads the rows the one before it returns, as a +# query reads the rows of a WITH; only the last statement's RETURN is the +# result. NEXT used to run the second statement on its own, so it matched +# from every node and read the first statement's columns as null. +# +# Dataset `social_network`: Alix (30), Gus (25) and Vincent (35); KNOWS +# Alix->Gus and Gus->Vincent. + +meta: + language: gql + model: lpg + section: "14.1" + title: Linear composition (NEXT) + dataset: social_network + +tests: + + - name: next_passes_a_node_on + query: "MATCH (a:Person {name: 'Alix'}) RETURN a NEXT MATCH (a)-[:KNOWS]->(b) RETURN b.name AS b" + expect: + rows: + - [Gus] + + - name: next_passes_a_value_on + query: "MATCH (a:Person {name: 'Alix'}) RETURN a.age AS x NEXT RETURN x + 1 AS y" + expect: + rows: + - [31] + + - name: the_result_is_the_last_return + query: "MATCH (a:Person) RETURN a.name AS n NEXT RETURN count(n) AS c" + expect: + rows: + - [3] + + - name: next_after_an_aggregate + query: "MATCH (a:Person) RETURN count(a) AS c NEXT RETURN c * 2 AS d" + expect: + rows: + - [6] + + - name: a_chain_of_next + query: "MATCH (a:Person {name: 'Alix'}) RETURN a NEXT MATCH (a)-[:KNOWS]->(b) RETURN b NEXT MATCH (b)-[:KNOWS]->(c) RETURN c.name AS c" + expect: + rows: + - [Vincent] + + - name: next_after_order_by_and_limit + query: "MATCH (a:Person) RETURN a ORDER BY a.age DESC LIMIT 1 NEXT RETURN a.name AS n" + expect: + rows: + - [Vincent] + + - name: next_after_return_distinct + query: "MATCH (a:Person)-[:KNOWS]->(b) RETURN DISTINCT 1 AS one NEXT RETURN count(*) AS c" + expect: + rows: + - [1] + + # What the RETURN before NEXT leaves out is gone, as after a WITH. + - name: next_drops_what_the_return_leaves_out + query: "MATCH (a:Person {name: 'Alix'})-[:KNOWS]->(b) RETURN b NEXT RETURN a.name AS n" + expect: + error: "Undefined variable 'a'" + + # ORDER BY a value the RETURN leaves out (a.age) still orders the rows passed on. + - name: next_after_order_by_on_a_column_not_returned + query: "MATCH (a:Person) RETURN a.name AS n ORDER BY a.age DESC LIMIT 2 NEXT RETURN n" + expect: + ordered: true + rows: + - [Vincent] + - [Alix] + + # Descending, so the rows differ from the unsorted first two (Alix, Gus). + - name: next_after_order_by_on_an_alias + query: "MATCH (a:Person) RETURN a.name AS n ORDER BY n DESC LIMIT 2 NEXT RETURN n" + expect: + ordered: true + rows: + - [Vincent] + - [Gus] + + # ORDER BY a property of an alias: x is the node a. Descending, as above. + - name: next_after_order_by_on_a_property_of_an_alias + query: "MATCH (a:Person) RETURN a AS x ORDER BY x.name DESC LIMIT 2 NEXT RETURN x.name AS n" + expect: + ordered: true + rows: + - [Vincent] + - [Gus] + + # An alias inside a CASE key: Alix (30) and Vincent (35) first, then Gus (25). + - name: next_after_order_by_on_an_alias_in_an_expression + query: "MATCH (a:Person) RETURN a.name AS n, a.age AS age ORDER BY CASE WHEN age > 26 THEN 0 ELSE 1 END, n NEXT RETURN n" + expect: + ordered: true + rows: + - [Alix] + - [Vincent] + - [Gus] + + # Two KNOWS rows with the same company: RETURN DISTINCT * passes one on. + - name: next_after_return_distinct_star + query: "MATCH (a:Person)-[:KNOWS]->(b) MATCH (c:Company) WITH c RETURN DISTINCT * NEXT RETURN count(*) AS n" + expect: + rows: + - [1] diff --git a/tests/spec/lpg/gql/18.1_subqueries.gtest b/tests/spec/lpg/gql/18.1_subqueries.gtest index f684cbb5a..749d61926 100644 --- a/tests/spec/lpg/gql/18.1_subqueries.gtest +++ b/tests/spec/lpg/gql/18.1_subqueries.gtest @@ -110,3 +110,77 @@ tests: expect: rows: - [Alix, TechCorp, Amsterdam] + + # --------------------------------------------------------------------------- + # VALUE subqueries read the outer row (they gave every row the same answer, + # or failed outside a RETURN item). Dataset: Alix (30), Gus (25), Vincent + # (40), Jules (35), Mia (28); KNOWS Alix->Gus, Gus->Vincent, Vincent->Alix, + # Jules->Mia, Alix->Jules. + # --------------------------------------------------------------------------- + + - name: value_reads_the_outer_row + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + query: "MATCH (p:Person) RETURN p.name AS p, VALUE { MATCH (p)-[:KNOWS]->(f) RETURN f.name ORDER BY f.name LIMIT 1 } AS first" + expect: + rows: + - [Alix, Gus] + - [Gus, Vincent] + - [Vincent, Alix] + - [Jules, Mia] + - [Mia, null] + + - name: value_in_where + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + query: "MATCH (p:Person) WHERE VALUE { MATCH (p)-[:KNOWS]->(f) RETURN f.name ORDER BY f.name LIMIT 1 } = 'Gus' RETURN p.name AS p" + expect: + rows: + - [Alix] + + - name: value_in_with + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + query: "MATCH (p:Person) WITH p, VALUE { MATCH (p)-[:KNOWS]->(f) RETURN f.age ORDER BY f.age DESC LIMIT 1 } AS oldest RETURN p.name AS p, oldest" + expect: + rows: + - [Alix, 35] + - [Gus, 40] + - [Vincent, 30] + - [Jules, 28] + - [Mia, null] + + - name: value_inside_a_function + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + query: "MATCH (p:Person) RETURN p.name AS p, upper(VALUE { MATCH (p)-[:KNOWS]->(f) RETURN f.name ORDER BY f.name LIMIT 1 }) AS first" + expect: + rows: + - [Alix, GUS] + - [Gus, VINCENT] + - [Vincent, ALIX] + - [Jules, MIA] + - [Mia, null] + + - name: value_of_an_outer_variable_only + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + query: "MATCH (p:Person {name: 'Alix'}) RETURN VALUE { MATCH (p) RETURN p.age + 1 } AS next" + expect: + rows: + - [31] + + - name: value_without_an_outer_row + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + query: "RETURN VALUE { MATCH (f:Person) RETURN f.name ORDER BY f.name LIMIT 1 } AS first" + expect: + rows: + - [Alix] + + - name: value_with_two_columns_is_rejected + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + query: "MATCH (p:Person) RETURN VALUE { MATCH (p)-[:KNOWS]->(f) RETURN f.name, f.age } AS v" + expect: + error: "A VALUE subquery returns one column" diff --git a/tests/spec/lpg/gql/ldbc_snb_interactive.gtest b/tests/spec/lpg/gql/ldbc_snb_interactive.gtest new file mode 100644 index 000000000..7d3d39dcd --- /dev/null +++ b/tests/spec/lpg/gql/ldbc_snb_interactive.gtest @@ -0,0 +1,1134 @@ +# LDBC SNB Interactive v1: the queries in GQL +# +# GQL translations of the LDBC Social Network Benchmark Interactive workload v1 queries; the Cypher reference +# queries are in lpg/cypher/ldbc_snb_interactive.gtest. The translations keep the structure of the reference +# queries (Grafeo's GQL accepts WITH and UNWIND) and use GQL forms where they differ: quantifiers (`-{1,2}`), +# ANY SHORTEST and ALL SHORTEST, EXISTS subqueries for pattern predicates, CAST, `month()` and `day()`, `edges(p)`, +# no chained comparisons, `IS NULL` instead of `CASE x WHEN null`, and `lk` for the LIKES variable (LIKE is a +# reserved word in GQL). An update whose reference query continues after CREATE runs here as several statements, +# since GQL accepts no clause after INSERT yet (#483). +# +# Dataset: ldbc_snb_mini, a small slice of the LDBC SNB Interactive v1 test data, every value copied unchanged (see +# its .setup file). The parameters are chosen from that slice; IC13 also uses the LDBC parameter pair (Gary Hill, +# Abdullah Koksal). The expected rows come from LDBC's own DuckDB (SQL) implementation run on the same slice, in the +# result shape of the Cypher reference queries. +# +# A skipped case names the bug it hits (and the issue, when there is one); remove the skip once it is fixed. + +meta: + language: gql + model: lpg + section: ldbc + title: LDBC SNB Interactive v1 (GQL) + dataset: ldbc_snb_mini + +tests: + + # --------------------------------------------------------------------------- + # Short reads + # --------------------------------------------------------------------------- + + # IS1. Profile of a person. + - name: is1_profile_of_a_person + params: + personId: 228 + query: | + MATCH (n:Person {id: $personId})-[:IS_LOCATED_IN]->(p:City) + RETURN + n.firstName AS firstName, + n.lastName AS lastName, + n.birthday AS birthday, + n.locationIP AS locationIP, + n.browserUsed AS browserUsed, + p.id AS cityId, + n.gender AS gender, + n.creationDate AS creationDate + expect: + ordered: true + rows: + - [Asher, Mamo, 524016000000, "213.55.93.153", Chrome, 1127, female, 1266720721912] + + # IS2. Recent messages of a person. + - name: is2_recent_messages_of_a_person + skip: "GQL rejects ORDER BY and LIMIT after WITH" + params: + personId: 228 + query: | + MATCH (:Person {id: $personId})<-[:HAS_CREATOR]-(message) + WITH + message, + message.id AS messageId, + message.creationDate AS messageCreationDate + ORDER BY messageCreationDate DESC, messageId ASC + LIMIT 10 + MATCH (message)-[:REPLY_OF]->{0,}(post:Post), + (post)-[:HAS_CREATOR]->(person) + RETURN + messageId, + coalesce(message.imageFile, message.content) AS messageContent, + messageCreationDate, + post.id AS postId, + person.id AS personId, + person.firstName AS personFirstName, + person.lastName AS personLastName + ORDER BY messageCreationDate DESC, messageId ASC + expect: + ordered: true + rows: + - [343597393779, "photo343597393779.jpg", 1290590906265, 343597393779, 228, Asher, Mamo] + - [343597393778, "photo343597393778.jpg", 1290590905265, 343597393778, 228, Asher, Mamo] + - [343597393747, "About Luis Horna, as a strong serve for a relatively shoAbout Djibouti, uti National Army and its sub-branchesAbout Leonard Bernstein, hilharmonic, ", 1289066095205, 343597393747, 228, Asher, Mamo] + - [343597392325, "About Ne-Yo, November 22, 2010. InAbout Brian Wilson, elled for various reaAbout V", 1289049183991, 343597392324, 76, Jae-Jin, Park] + - [206158440292, "photo206158440292.jpg", 1279208419298, 206158440292, 228, Asher, Mamo] + - [206158431893, roflol, 1278531351427, 206158431892, 102, Philibert, Roindefo] + - [137438963511, "About William Lyon Mackenzie King, y. King worked to bring compromise andAbout Thomas Jefferson, erson, his wife", 1277076649771, 137438963511, 228, Asher, Mamo] + - [137438963744, "About Christopher Lee, or services to dramAbout Tina Turner, inning with a", 1273650346399, 137438963740, 150, Alfonso, Alvarez] + - [137438963499, "About William Lyon Mackenzie King, allowed his intense spirituality to distort About T", 1272740097725, 137438963499, 228, Asher, Mamo] + - [68719478457, "About Oscar Wilde, intellectuals. Their sAbout Anne, Queen of Great Britain, II became joint monarcAbout Lebanon, ld national infrastrucAbout Bad, Bad L", 1271850332021, 68719478453, 2199023255712, Aurora, Cruz] + + # IS3. Friends of a person. + - name: is3_friends_of_a_person + params: + personId: 228 + query: | + MATCH (n:Person {id: $personId})-[r:KNOWS]-(friend) + RETURN + friend.id AS personId, + friend.firstName AS firstName, + friend.lastName AS lastName, + r.creationDate AS friendshipCreationDate + ORDER BY + friendshipCreationDate DESC, + personId ASC + expect: + ordered: true + rows: + - [8796093022357, Gary, Hill, 1288580721183] + - [2199023255712, Aurora, Cruz, 1271536640884] + - [102, Philibert, Roindefo, 1268755084867] + - [76, Jae-Jin, Park, 1267889385714] + - [150, Alfonso, Alvarez, 1267126413921] + + # IS4. Content of a message. + - name: is4_content_of_a_message_post + params: + messageId: 10169 + query: | + MATCH (m:Message {id: $messageId}) + RETURN + m.creationDate AS messageCreationDate, + coalesce(m.content, m.imageFile) AS messageContent + expect: + ordered: true + rows: + - [1267290597839, "photo10169.jpg"] + + # IS4. Content of a message. + - name: is4_content_of_a_message_comment + params: + messageId: 343597392326 + query: | + MATCH (m:Message {id: $messageId}) + RETURN + m.creationDate AS messageCreationDate, + coalesce(m.content, m.imageFile) AS messageContent + expect: + ordered: true + rows: + - [1289092584789, "no way!"] + + # IS5. Creator of a message. + - name: is5_creator_of_a_message + params: + messageId: 343597392326 + query: | + MATCH (m:Message {id: $messageId})-[:HAS_CREATOR]->(p:Person) + RETURN + p.id AS personId, + p.firstName AS firstName, + p.lastName AS lastName + expect: + ordered: true + rows: + - [6597069766775, Jie, Yang] + + # IS6. Forum of a message. + - name: is6_forum_of_a_message + params: + messageId: 343597393752 + query: | + MATCH (m:Message {id: $messageId})-[:REPLY_OF]->{0,}(p:Post)<-[:CONTAINER_OF]-(f:Forum)-[:HAS_MODERATOR]->(mod:Person) + RETURN + f.id AS forumId, + f.title AS forumTitle, + mod.id AS moderatorId, + mod.firstName AS moderatorFirstName, + mod.lastName AS moderatorLastName + expect: + ordered: true + rows: + - [872, "Wall of Asher Mamo", 228, Asher, Mamo] + + # IS7. Replies of a message. + - name: is7_replies_of_a_message + params: + messageId: 137438963499 + query: | + MATCH (m:Message {id: $messageId})<-[:REPLY_OF]-(c:Comment)-[:HAS_CREATOR]->(p:Person) + OPTIONAL MATCH (m)-[:HAS_CREATOR]->(a:Person)-[r:KNOWS]-(p) + RETURN + c.id AS commentId, + c.content AS commentContent, + c.creationDate AS commentCreationDate, + p.id AS replyAuthorId, + p.firstName AS replyAuthorFirstName, + p.lastName AS replyAuthorLastName, + r IS NOT NULL AS replyAuthorKnowsOriginalMessageAuthor + ORDER BY commentCreationDate DESC, replyAuthorId + expect: + ordered: true + rows: + - [137438963506, "About Thomas Hardy, y published as serials in magazines, weAbout Thomas Jeffe", 1272757138003, 150, Alfonso, Alvarez, true] + - [137438963501, "About Frank Zappa, ollages. His later albums shared this eclectic and experimental approach, irrespectiv", 1272755669544, 102, Philibert, Roindefo, true] + + + # --------------------------------------------------------------------------- + # Complex reads + # --------------------------------------------------------------------------- + + # IC1. Transitive friends with certain name. + - name: ic1_transitive_friends_with_a_name_maria + skip: "GQL rejects ORDER BY and LIMIT after WITH; then a list-valued grouping key next to collect() becomes 0 (friendUniversities)" + params: + personId: 228 + firstName: "Maria" + query: | + MATCH (p:Person {id: $personId}), (friend:Person {firstName: $firstName}) + WHERE NOT p = friend + WITH p, friend + MATCH path = ANY SHORTEST (p)-[:KNOWS]-{1,3}(friend) + WITH min(length(path)) AS distance, friend + ORDER BY + distance ASC, + friend.lastName ASC, + friend.id ASC + LIMIT 20 + MATCH (friend)-[:IS_LOCATED_IN]->(friendCity:City) + OPTIONAL MATCH (friend)-[studyAt:STUDY_AT]->(uni:University)-[:IS_LOCATED_IN]->(uniCity:City) + WITH friend, collect( + CASE WHEN uni IS NULL THEN null + ELSE [uni.name, studyAt.classYear, uniCity.name] + END) AS unis, friendCity, distance + OPTIONAL MATCH (friend)-[workAt:WORK_AT]->(company:Company)-[:IS_LOCATED_IN]->(companyCountry:Country) + WITH friend, collect( + CASE WHEN company IS NULL THEN null + ELSE [company.name, workAt.workFrom, companyCountry.name] + END) AS companies, unis, friendCity, distance + RETURN + friend.id AS friendId, + friend.lastName AS friendLastName, + distance AS distanceFromPerson, + friend.birthday AS friendBirthday, + friend.creationDate AS friendCreationDate, + friend.gender AS friendGender, + friend.browserUsed AS friendBrowserUsed, + friend.locationIP AS friendLocationIp, + friend.email AS friendEmails, + friend.speaks AS friendLanguages, + friendCity.name AS friendCityName, + unis AS friendUniversities, + companies AS friendCompanies + ORDER BY + distanceFromPerson ASC, + friendLastName ASC, + friendId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [143, Alkaios, 2, 410659200000, 1262456643976, female, Firefox, "62.217.119.183", "[Maria143@gmail.com]", "[fr, en]", Athens, "[[National_and_Kapodistrian_University_of_Athens, 2003, Athens]]", "[[Macedonian_Airlines, 2004, Greece]]"] + + # IC1. Transitive friends with certain name. + - name: ic1_transitive_friends_with_a_name_john + skip: "GQL rejects ORDER BY and LIMIT after WITH; then a list-valued grouping key next to collect() becomes 0 (friendUniversities)" + params: + personId: 228 + firstName: "John" + query: | + MATCH (p:Person {id: $personId}), (friend:Person {firstName: $firstName}) + WHERE NOT p = friend + WITH p, friend + MATCH path = ANY SHORTEST (p)-[:KNOWS]-{1,3}(friend) + WITH min(length(path)) AS distance, friend + ORDER BY + distance ASC, + friend.lastName ASC, + friend.id ASC + LIMIT 20 + MATCH (friend)-[:IS_LOCATED_IN]->(friendCity:City) + OPTIONAL MATCH (friend)-[studyAt:STUDY_AT]->(uni:University)-[:IS_LOCATED_IN]->(uniCity:City) + WITH friend, collect( + CASE WHEN uni IS NULL THEN null + ELSE [uni.name, studyAt.classYear, uniCity.name] + END) AS unis, friendCity, distance + OPTIONAL MATCH (friend)-[workAt:WORK_AT]->(company:Company)-[:IS_LOCATED_IN]->(companyCountry:Country) + WITH friend, collect( + CASE WHEN company IS NULL THEN null + ELSE [company.name, workAt.workFrom, companyCountry.name] + END) AS companies, unis, friendCity, distance + RETURN + friend.id AS friendId, + friend.lastName AS friendLastName, + distance AS distanceFromPerson, + friend.birthday AS friendBirthday, + friend.creationDate AS friendCreationDate, + friend.gender AS friendGender, + friend.browserUsed AS friendBrowserUsed, + friend.locationIP AS friendLocationIp, + friend.email AS friendEmails, + friend.speaks AS friendLanguages, + friendCity.name AS friendCityName, + unis AS friendUniversities, + companies AS friendCompanies + ORDER BY + distanceFromPerson ASC, + friendLastName ASC, + friendId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [41, Kumar, 3, 527731200000, 1266276257359, male, Safari, "27.116.33.147", "[John41@gmail.com, John41@jizan.cc, John41@yahoo.com, John41@zoho.com]", "[gu, mr, en]", Puttur, "[[The_Oxford_Educational_Institutions, 2004, Bangalore]]", "[[Jet_Airways, 2005, India], [Jagson_Airlines, 2005, India], [Deccan_360, 2006, India]]"] + + # IC2. Recent messages by your friends. + - name: ic2_recent_messages_by_friends + params: + personId: 228 + maxDate: 1285891200000 + query: | + MATCH (:Person {id: $personId})-[:KNOWS]-(friend:Person)<-[:HAS_CREATOR]-(message:Message) + WHERE message.creationDate <= $maxDate + RETURN + friend.id AS personId, + friend.firstName AS personFirstName, + friend.lastName AS personLastName, + message.id AS postOrCommentId, + coalesce(message.content, message.imageFile) AS postOrCommentContent, + message.creationDate AS postOrCommentCreationDate + ORDER BY + postOrCommentCreationDate DESC, + postOrCommentId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [2199023255712, Aurora, Cruz, 274877909928, "no way!", 1285681173411] + - [76, Jae-Jin, Park, 274877909925, thx, 1285676860041] + - [102, Philibert, Roindefo, 274877912148, fine, 1284597772686] + - [76, Jae-Jin, Park, 206158432797, thx, 1280907556754] + - [102, Philibert, Roindefo, 206158431892, "About William Penn, – 16 September 1670) was an English admiral and politician who sat in the House of Commons from 1660 to 1670. He was the father of Wi", 1278511907643] + - [2199023255712, Aurora, Cruz, 206158432090, "photo206158432090.jpg", 1277597415173] + - [102, Philibert, Roindefo, 137438958564, cool, 1277084387523] + - [2199023255712, Aurora, Cruz, 137438963517, "I see", 1277076710526] + - [2199023255712, Aurora, Cruz, 137438955279, "photo137438955279.jpg", 1276553301898] + - [2199023255712, Aurora, Cruz, 137438955277, "photo137438955277.jpg", 1276553299898] + - [2199023255712, Aurora, Cruz, 137438955264, "photo137438955264.jpg", 1276274322999] + - [76, Jae-Jin, Park, 137438963765, right, 1273696787135] + - [150, Alfonso, Alvarez, 137438963740, "About Pope Paul VI, Church life during his pontificate excAbout Winston Churchill, United Kingdom during the Second WorldAbout Julia Gillard, ing Mitcham Demonstration School and UnAbout S", 1273649436061] + - [150, Alfonso, Alvarez, 137438963751, "About Julia Gillard, ister upon Labor's victory in the 2007 federal electionAbout Chile, Republic of Chile, is a country in South America occupyAbout South Korea, ith production focusing on electronics, automobiles, ", 1273617306061] + - [2199023255712, Aurora, Cruz, 137438963510, great, 1272813417041] + - [102, Philibert, Roindefo, 137438963508, "no way!", 1272758692818] + - [150, Alfonso, Alvarez, 137438963502, "About Stephen Sondheim, an composer and lyricist known for his contributions to musical th", 1272758679985] + - [150, Alfonso, Alvarez, 137438963506, "About Thomas Hardy, y published as serials in magazines, weAbout Thomas Jeffe", 1272757138003] + - [102, Philibert, Roindefo, 137438963501, "About Frank Zappa, ollages. His later albums shared this eclectic and experimental approach, irrespectiv", 1272755669544] + - [150, Alfonso, Alvarez, 68719487342, ok, 1271910621171] + + # IC3. Friends and friends of friends that have been to given countries. + - name: ic3_friends_in_countries_x_and_y + skip: "GQL rejects LIMIT after WITH, and a WHERE on a MATCH that follows WITH (#483); then the comma MATCH with country IN [countryX, countryY] returns no rows" + params: + personId: 228 + countryXName: "Uruguay" + countryYName: "Canada" + startDate: 1275350400000 + endDate: 1277942400000 + query: | + MATCH (countryX:Country {name: $countryXName}), + (countryY:Country {name: $countryYName}), + (person:Person {id: $personId}) + WITH person, countryX, countryY + LIMIT 1 + MATCH (city:City)-[:IS_PART_OF]->(country:Country) + WHERE country IN [countryX, countryY] + WITH person, countryX, countryY, collect(city) AS cities + MATCH (person)-[:KNOWS]-{1,2}(friend)-[:IS_LOCATED_IN]->(city) + WHERE NOT person = friend AND NOT city IN cities + WITH DISTINCT friend, countryX, countryY + MATCH (friend)<-[:HAS_CREATOR]-(message), + (message)-[:IS_LOCATED_IN]->(country) + WHERE message.creationDate >= $startDate AND message.creationDate < $endDate AND + country IN [countryX, countryY] + WITH friend, + CASE WHEN country = countryX THEN 1 ELSE 0 END AS messageX, + CASE WHEN country = countryY THEN 1 ELSE 0 END AS messageY + WITH friend, sum(messageX) AS xCount, sum(messageY) AS yCount + WHERE xCount > 0 AND yCount > 0 + RETURN friend.id AS friendId, + friend.firstName AS friendFirstName, + friend.lastName AS friendLastName, + xCount, + yCount, + xCount + yCount AS xyCount + ORDER BY xyCount DESC, friendId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [2199023255712, Aurora, Cruz, 1, 2, 3] + + # IC4. New topics. + - name: ic4_new_topics + params: + personId: 228 + startDate: 1272672000000 + endDate: 1275264000000 + query: | + MATCH (person:Person {id: $personId})-[:KNOWS]-(friend:Person), + (friend)<-[:HAS_CREATOR]-(post:Post)-[:HAS_TAG]->(tag) + WITH DISTINCT tag, post + WITH tag, + CASE + WHEN post.creationDate >= $startDate AND post.creationDate < $endDate THEN 1 + ELSE 0 + END AS valid, + CASE + WHEN post.creationDate < $startDate THEN 1 + ELSE 0 + END AS inValid + WITH tag, sum(valid) AS postCount, sum(inValid) AS inValidPostCount + WHERE postCount > 0 AND inValidPostCount = 0 + RETURN tag.name AS tagName, postCount + ORDER BY postCount DESC, tagName ASC + LIMIT 10 + expect: + ordered: true + rows: + - [Julia_Gillard, 2] + - [Chile, 1] + - [Monaco, 1] + - [Nigeria, 1] + - [Pope_Paul_VI, 1] + - [South_Korea, 1] + - [South_Vietnam, 1] + - [Winston_Churchill, 1] + + # IC5. New groups. + - name: ic5_new_groups + skip: "GQL rejects a WHERE on a MATCH that follows WITH (#483); then OPTIONAL MATCH ... WHERE friend IN friends loses the forums without a match" + params: + personId: 228 + minDate: 1275350400000 + query: | + MATCH (person:Person {id: $personId})-[:KNOWS]-{1,2}(friend) + WHERE NOT person = friend + WITH DISTINCT friend + MATCH (friend)<-[membership:HAS_MEMBER]-(forum) + WHERE membership.joinDate > $minDate + WITH forum, collect(friend) AS friends + OPTIONAL MATCH (friend)<-[:HAS_CREATOR]-(post)<-[:CONTAINER_OF]-(forum) + WHERE friend IN friends + WITH forum, count(post) AS postCount + RETURN + forum.title AS forumName, + postCount + ORDER BY + postCount DESC, + forum.id ASC + LIMIT 20 + expect: + ordered: true + rows: + - ["Wall of Maria Alkaios", 0] + - ["Wall of Jae-Jin Park", 0] + - ["Wall of Asher Mamo", 0] + - ["Album 9 of Asher Mamo", 0] + - ["Wall of Alfonso Alvarez", 0] + - ["Wall of Abdala Ndiaye", 0] + - ["Wall of Aurora Cruz", 0] + - ["Album 3 of Asher Mamo", 0] + - ["Album 1 of Aurora Cruz", 0] + - ["Album 2 of Aurora Cruz", 0] + - ["Wall of Rafael Fernández", 0] + - ["Album 7 of Aurora Cruz", 0] + - ["Wall of Jie Yang", 0] + - ["Album 0 of Asher Mamo", 0] + - ["Album 6 of Alfonso Alvarez", 0] + - ["Album 8 of Abdala Ndiaye", 0] + + # IC6. Tag co-occurrence. + - name: ic6_tag_co_occurrence + skip: "GQL rejects a WHERE on a MATCH that follows WITH (#483); then the comma MATCH with {id: knownTagId} returns no rows" + params: + personId: 228 + tagName: "Luis_Horna" + query: | + MATCH (knownTag:Tag {name: $tagName}) + WITH knownTag.id AS knownTagId + MATCH (person:Person {id: $personId})-[:KNOWS]-{1,2}(friend) + WHERE NOT person = friend + WITH knownTagId, collect(DISTINCT friend) AS friends + UNWIND friends AS f + MATCH (f)<-[:HAS_CREATOR]-(post:Post), + (post)-[:HAS_TAG]->(t:Tag {id: knownTagId}), + (post)-[:HAS_TAG]->(tag:Tag) + WHERE NOT t = tag + WITH tag.name AS tagName, count(post) AS postCount + RETURN tagName, postCount + ORDER BY postCount DESC, tagName ASC + LIMIT 10 + expect: + ordered: true + rows: + - [Brian_Wilson, 1] + - [Gibraltar, 1] + - ["Harry_S._Truman", 1] + - [Israel, 1] + - [Republic_of_the_Congo, 1] + - [Robert_Altman, 1] + - [Xiongnu, 1] + + # IC7. Recent likers. + - name: ic7_recent_likers + skip: "GQL rejects ORDER BY after WITH; then a node stored in a map loses its kind (latestLike.msg.id is null)" + params: + personId: 228 + query: | + MATCH (person:Person {id: $personId})<-[:HAS_CREATOR]-(message:Message)<-[lk:LIKES]-(liker:Person) + WITH liker, message, lk.creationDate AS likeTime, person + ORDER BY likeTime DESC, message.id ASC + WITH liker, head(collect({msg: message, likeTime: likeTime})) AS latestLike, person + RETURN + liker.id AS personId, + liker.firstName AS personFirstName, + liker.lastName AS personLastName, + latestLike.likeTime AS likeCreationDate, + latestLike.msg.id AS commentOrPostId, + coalesce(latestLike.msg.content, latestLike.msg.imageFile) AS commentOrPostContent, + CAST(floor(CAST(latestLike.likeTime - latestLike.msg.creationDate AS FLOAT) / 1000.0) / 60.0 AS INT) AS minutesLatency, + NOT EXISTS { MATCH (liker)-[:KNOWS]-(person) } AS isNew + ORDER BY + likeCreationDate DESC, + personId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [2199023255712, Aurora, Cruz, 1289546926226, 10169, "photo10169.jpg", 370938, false] + - [8796093022390, Abdullah, Koksal, 1289061566907, 343597392325, "About Ne-Yo, November 22, 2010. InAbout Brian Wilson, elled for various reaAbout V", 206, true] + - [150, Alfonso, Alvarez, 1282198715860, 206158440292, "photo206158440292.jpg", 49838, false] + - [76, Jae-Jin, Park, 1272582754260, 10174, "photo10174.jpg", 88202, false] + + # IC8. Recent replies. + - name: ic8_recent_replies + params: + personId: 228 + query: | + MATCH (start:Person {id: $personId})<-[:HAS_CREATOR]-(:Message)<-[:REPLY_OF]-(comment:Comment)-[:HAS_CREATOR]->(person:Person) + RETURN + person.id AS personId, + person.firstName AS personFirstName, + person.lastName AS personLastName, + comment.creationDate AS commentCreationDate, + comment.id AS commentId, + comment.content AS commentContent + ORDER BY + commentCreationDate DESC, + commentId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [76, Jae-Jin, Park, 1289135518830, 343597393750, "About Luis Horna, e surface is clay. He was the About Waiting for the End, ough it was "] + - [6597069766775, Jie, Yang, 1289092584789, 343597392326, "no way!"] + - [102, Philibert, Roindefo, 1289090250676, 343597393755, "About Al Capone, ing the money he made from his actAbout Djibouti, military forces of Djibouti and coAbout Leonard Bernstein, talent"] + - [2199023255712, Aurora, Cruz, 1277076710526, 137438963517, "I see"] + - [150, Alfonso, Alvarez, 1272757138003, 137438963506, "About Thomas Hardy, y published as serials in magazines, weAbout Thomas Jeffe"] + - [102, Philibert, Roindefo, 1272755669544, 137438963501, "About Frank Zappa, ollages. His later albums shared this eclectic and experimental approach, irrespectiv"] + - [143, Maria, Alkaios, 1271884315385, 68719478447, "About Kevin Rudd, ia's remaining About William Morris, traditional teAbout France, c"] + + # IC9. Recent messages by friends or friends of friends. + - name: ic9_recent_messages_by_friends_of_friends + skip: "GQL rejects a WHERE on a MATCH that follows WITH (#483)" + params: + personId: 228 + maxDate: 1285891200000 + query: | + MATCH (root:Person {id: $personId})-[:KNOWS]-{1,2}(friend:Person) + WHERE NOT friend = root + WITH collect(DISTINCT friend) AS friends + UNWIND friends AS friend + MATCH (friend)<-[:HAS_CREATOR]-(message:Message) + WHERE message.creationDate < $maxDate + RETURN + friend.id AS personId, + friend.firstName AS personFirstName, + friend.lastName AS personLastName, + message.id AS commentOrPostId, + coalesce(message.content, message.imageFile) AS commentOrPostContent, + message.creationDate AS commentOrPostCreationDate + ORDER BY + commentOrPostCreationDate DESC, + commentOrPostId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [143, Maria, Alkaios, 274877909920, roflol, 1285685102274] + - [6597069766775, Jie, Yang, 274877909919, "About Manuel Noriega, uest in April 2010. He arrived in Paris on April 27, 2010, and after a re-trial as a condition", 1285682584114] + - [2199023255712, Aurora, Cruz, 274877909928, "no way!", 1285681173411] + - [76, Jae-Jin, Park, 274877909925, thx, 1285676860041] + - [143, Maria, Alkaios, 274877909929, "I see", 1285665988168] + - [4398046511333, Rafael, "Fernández", 274877909927, "About Manuel Noriega, trafficking, racketeering, andAbout Mikhail Gorbachev, was the onl", 1285665306179] + - [6597069766775, Jie, Yang, 274877909924, "About Manuel Noriega, g, racketeering, and money laundering About Arthur Wellesley, 1st Duke of Wellington, m. He", 1285664359114] + - [4398046511333, Rafael, "Fernández", 274877912142, "no", 1284613964426] + - [102, Philibert, Roindefo, 274877912148, fine, 1284597772686] + - [143, Maria, Alkaios, 274877912139, "About Enrique Iglesias, hits on the various Billboard charts. Billboard has called him The King of Latin Pop and The King of DanAbout California King Bed, ive reviews from music critics, who praised Rihanna", 1284593996258] + - [4398046511333, Rafael, "Fernández", 274877909514, "About Guy Sebastian, 08 Australian tour. Like It Like That has three tracks with John Mayer o", 1283465660488] + - [143, Maria, Alkaios, 274877913531, right, 1283303511849] + - [153, Abdala, Ndiaye, 206158432792, duh, 1280914332302] + - [76, Jae-Jin, Park, 206158432797, thx, 1280907556754] + - [4398046511333, Rafael, "Fernández", 206158432782, "About Guy Sebastian, ur Asian countries and New Zealand. Sebastian had a second number one in New Zealand with Who's That Girl, two other top ten singles and a number three album, and gained four platinum and two gold certifications there. He ha", 1280894717004] + - [143, Maria, Alkaios, 206158431896, "yes", 1278512054636] + - [102, Philibert, Roindefo, 206158431892, "About William Penn, – 16 September 1670) was an English admiral and politician who sat in the House of Commons from 1660 to 1670. He was the father of Wi", 1278511907643] + - [2199023255712, Aurora, Cruz, 206158432090, "photo206158432090.jpg", 1277597415173] + - [102, Philibert, Roindefo, 137438958564, cool, 1277084387523] + - [2199023255712, Aurora, Cruz, 137438963517, "I see", 1277076710526] + + # IC10. Friend recommendation. + - name: ic10_friend_recommendation + skip: "#543: EXISTS with a two-hop pattern in a list comprehension fails with Unsupported EXISTS subquery pattern; then datetime({epochMillis: ...}) returns null" + params: + personId: 228 + month: 7 + query: | + MATCH (person:Person {id: $personId})-[:KNOWS]-{2,2}(friend), + (friend)-[:IS_LOCATED_IN]->(city:City) + WHERE NOT friend = person AND + NOT EXISTS { MATCH (friend)-[:KNOWS]-(person) } + WITH person, city, friend, datetime({epochMillis: friend.birthday}) AS birthday + WHERE (month(birthday) = $month AND day(birthday) >= 21) OR + (month(birthday) = ($month % 12) + 1 AND day(birthday) < 22) + WITH DISTINCT friend, city, person + OPTIONAL MATCH (friend)<-[:HAS_CREATOR]-(post:Post) + WITH friend, city, collect(post) AS posts, person + WITH friend, + city, + size(posts) AS postCount, + size([p IN posts WHERE EXISTS { MATCH (p)-[:HAS_TAG]->()<-[:HAS_INTEREST]-(person) }]) AS commonPostCount + RETURN friend.id AS personId, + friend.firstName AS personFirstName, + friend.lastName AS personLastName, + commonPostCount - (postCount - commonPostCount) AS commonInterestScore, + friend.gender AS personGender, + city.name AS personCityName + ORDER BY commonInterestScore DESC, personId ASC + LIMIT 10 + expect: + ordered: true + rows: + - [10995116277918, Javed, Khan, 0, male, Major_Cities] + - [4398046511333, Rafael, "Fernández", -2, female, Barcelona] + + # IC11. Job referral. + - name: ic11_job_referral + skip: "GQL rejects a WHERE on a MATCH that follows WITH (#483)" + params: + personId: 228 + countryName: "Philippines" + workFromYear: 2008 + query: | + MATCH (person:Person {id: $personId})-[:KNOWS]-{1,2}(friend:Person) + WHERE NOT person = friend + WITH DISTINCT friend + MATCH (friend)-[workAt:WORK_AT]->(company:Company)-[:IS_LOCATED_IN]->(:Country {name: $countryName}) + WHERE workAt.workFrom < $workFromYear + RETURN + friend.id AS personId, + friend.firstName AS personFirstName, + friend.lastName AS personLastName, + company.name AS organizationName, + workAt.workFrom AS organizationWorkFromYear + ORDER BY + organizationWorkFromYear ASC, + personId ASC, + organizationName DESC + LIMIT 10 + expect: + ordered: true + rows: + - [2199023255712, Aurora, Cruz, South_East_Asian_Airlines, 2007] + - [2199023255712, Aurora, Cruz, Filipinas_Orient_Airways, 2007] + + # IC12. Expert search. + - name: ic12_expert_search_office_holder + skip: "GQL rejects a WHERE on a MATCH that follows WITH (#483)" + params: + personId: 228 + tagClassName: "OfficeHolder" + query: | + MATCH (tag:Tag)-[:HAS_TYPE|IS_SUBCLASS_OF]->{0,}(baseTagClass:TagClass) + WHERE tag.name = $tagClassName OR baseTagClass.name = $tagClassName + WITH collect(tag.id) AS tags + MATCH (:Person {id: $personId})-[:KNOWS]-(friend:Person)<-[:HAS_CREATOR]-(comment:Comment)-[:REPLY_OF]->(:Post)-[:HAS_TAG]->(tag:Tag) + WHERE tag.id IN tags + RETURN + friend.id AS personId, + friend.firstName AS personFirstName, + friend.lastName AS personLastName, + collect(DISTINCT tag.name) AS tagNames, + count(DISTINCT comment) AS replyCount + ORDER BY + replyCount DESC, + personId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [76, Jae-Jin, Park, "[Julia_Gillard, Arthur_Wellesley,_1st_Duke_of_Wellington]", 2] + - [102, Philibert, Roindefo, "[Aung_San_Suu_Kyi, William_Lyon_Mackenzie_King, Thomas_Jefferson]", 2] + - [150, Alfonso, Alvarez, "[William_Lyon_Mackenzie_King, Thomas_Jefferson]", 1] + - [2199023255712, Aurora, Cruz, "[William_Lyon_Mackenzie_King, Thomas_Jefferson]", 1] + + # IC12. Expert search. + - name: ic12_expert_search_tennis_player + skip: "GQL rejects a WHERE on a MATCH that follows WITH (#483)" + params: + personId: 228 + tagClassName: "TennisPlayer" + query: | + MATCH (tag:Tag)-[:HAS_TYPE|IS_SUBCLASS_OF]->{0,}(baseTagClass:TagClass) + WHERE tag.name = $tagClassName OR baseTagClass.name = $tagClassName + WITH collect(tag.id) AS tags + MATCH (:Person {id: $personId})-[:KNOWS]-(friend:Person)<-[:HAS_CREATOR]-(comment:Comment)-[:REPLY_OF]->(:Post)-[:HAS_TAG]->(tag:Tag) + WHERE tag.id IN tags + RETURN + friend.id AS personId, + friend.firstName AS personFirstName, + friend.lastName AS personLastName, + collect(DISTINCT tag.name) AS tagNames, + count(DISTINCT comment) AS replyCount + ORDER BY + replyCount DESC, + personId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [76, Jae-Jin, Park, "[Dudi_Sela, Luis_Horna]", 2] + - [150, Alfonso, Alvarez, "[Dudi_Sela, Venus_Williams]", 2] + - [102, Philibert, Roindefo, "[Luis_Horna]", 1] + - [2199023255712, Aurora, Cruz, "[Luis_Horna]", 1] + + # IC13. Single shortest path. + - name: ic13_single_shortest_path_ldbc_pair + skip: "#318: path IS NULL is true for an ANY SHORTEST path, so the query returns -1" + params: + person1Id: 8796093022357 + person2Id: 8796093022390 + query: | + MATCH (person1:Person {id: $person1Id}), (person2:Person {id: $person2Id}) + MATCH path = ANY SHORTEST (person1)-[:KNOWS]-+(person2) + RETURN + CASE WHEN path IS NULL THEN -1 ELSE length(path) END AS shortestPathLength + expect: + ordered: true + rows: + - [2] + + # IC13. Single shortest path. + - name: ic13_single_shortest_path_three_hops + skip: "#318: path IS NULL is true for an ANY SHORTEST path, so the query returns -1" + params: + person1Id: 228 + person2Id: 41 + query: | + MATCH (person1:Person {id: $person1Id}), (person2:Person {id: $person2Id}) + MATCH path = ANY SHORTEST (person1)-[:KNOWS]-+(person2) + RETURN + CASE WHEN path IS NULL THEN -1 ELSE length(path) END AS shortestPathLength + expect: + ordered: true + rows: + - [3] + + # IC14. Trusted connection paths. + - name: ic14_trusted_connection_paths + skip: "startNode(r).id is rejected (property access on a function result); #318: the path of ALL SHORTEST with ->* is not bound" + params: + person1Id: 228 + person2Id: 41 + query: | + MATCH path = ALL SHORTEST (person1:Person {id: $person1Id})-[:KNOWS]-*(person2:Person {id: $person2Id}) + WITH path, edges(path) AS rels_in_path + WITH + [n IN nodes(path) | n.id] AS personIdsInPath, + [r IN rels_in_path | + COUNT { MATCH (a:Person)<-[:HAS_CREATOR]-(:Comment)-[:REPLY_OF]->(:Post)-[:HAS_CREATOR]->(b:Person) + WHERE (a.id = startNode(r).id AND b.id = endNode(r).id) OR (a.id = endNode(r).id AND b.id = startNode(r).id) } * 1.0 + ] AS weight1, + [r IN rels_in_path | + COUNT { MATCH (a:Person)<-[:HAS_CREATOR]-(:Comment)-[:REPLY_OF]->(:Comment)-[:HAS_CREATOR]->(b:Person) + WHERE (a.id = startNode(r).id AND b.id = endNode(r).id) OR (a.id = endNode(r).id AND b.id = startNode(r).id) } * 0.5 + ] AS weight2 + WITH + personIdsInPath, + reduce(w = 0.0, v IN weight1 | w + v) AS w1, + reduce(w = 0.0, v IN weight2 | w + v) AS w2 + RETURN + personIdsInPath, + w1 + w2 AS pathWeight + ORDER BY pathWeight DESC + expect: + ordered: true + precision: 6 + rows: + - ["[228, 102, 143, 41]", 10.0] + - ["[228, 8796093022357, 143, 41]", 3.0] + + + # --------------------------------------------------------------------------- + # Updates + # --------------------------------------------------------------------------- + + # IU1. Add person: Isabel Garcia, from the LDBC update stream (lists written inline). + # IU1 stores the languages as `languages` (IC1 reads `speaks`). + - name: iu1_add_person + params: + cityId: 801 + personId: 10995116277976 + personFirstName: "Isabel" + personLastName: "Garcia" + gender: "female" + birthday: 421545600000 + creationDate: 1291607408799 + locationIP: "103.4.23.146" + browserUsed: "Firefox" + statements: + - | + MATCH (c:City {id: $cityId}) + INSERT (p:Person { + id: $personId, + firstName: $personFirstName, + lastName: $personLastName, + gender: $gender, + birthday: $birthday, + creationDate: $creationDate, + locationIP: $locationIP, + browserUsed: $browserUsed, + languages: ['en'], + email: ['Isabel10995116277976@gmx.com', 'Isabel10995116277976@yahoo.com', 'Isabel10995116277976@zoho.com'] + })-[:IS_LOCATED_IN]->(c) + - | + MATCH (p:Person {id: $personId}) + FOR tagId IN [26] + MATCH (t:Tag {id: tagId}) + INSERT (p)-[:HAS_INTEREST]->(t) + - | + MATCH (p:Person {id: $personId}) + FOR s IN [[5522, 2003]] + MATCH (u:Organisation {id: s[0]}) + INSERT (p)-[:STUDY_AT {classYear: s[1]}]->(u) + - | + MATCH (p:Person {id: $personId}) + FOR w IN [[965, 2004], [955, 2005], [957, 2004]] + MATCH (comp:Organisation {id: w[0]}) + INSERT (p)-[:WORKS_AT {workFrom: w[1]}]->(comp) + - | + MATCH (p:Person {id: 10995116277976})-[:IS_LOCATED_IN]->(c:City) + RETURN p.firstName, p.lastName, p.gender, p.birthday, p.browserUsed, p.languages, p.email, c.name + expect: + rows: + - [Isabel, Garcia, female, 421545600000, Firefox, "[en]", "[Isabel10995116277976@gmx.com, Isabel10995116277976@yahoo.com, Isabel10995116277976@zoho.com]", Cebu_City] + + # IU1. Add person: the edges (IU1 creates WORKS_AT, not WORK_AT). + - name: iu1_add_person_edges + params: + cityId: 801 + personId: 10995116277976 + personFirstName: "Isabel" + personLastName: "Garcia" + gender: "female" + birthday: 421545600000 + creationDate: 1291607408799 + locationIP: "103.4.23.146" + browserUsed: "Firefox" + statements: + - | + MATCH (c:City {id: $cityId}) + INSERT (p:Person { + id: $personId, + firstName: $personFirstName, + lastName: $personLastName, + gender: $gender, + birthday: $birthday, + creationDate: $creationDate, + locationIP: $locationIP, + browserUsed: $browserUsed, + languages: ['en'], + email: ['Isabel10995116277976@gmx.com', 'Isabel10995116277976@yahoo.com', 'Isabel10995116277976@zoho.com'] + })-[:IS_LOCATED_IN]->(c) + - | + MATCH (p:Person {id: $personId}) + FOR tagId IN [26] + MATCH (t:Tag {id: tagId}) + INSERT (p)-[:HAS_INTEREST]->(t) + - | + MATCH (p:Person {id: $personId}) + FOR s IN [[5522, 2003]] + MATCH (u:Organisation {id: s[0]}) + INSERT (p)-[:STUDY_AT {classYear: s[1]}]->(u) + - | + MATCH (p:Person {id: $personId}) + FOR w IN [[965, 2004], [955, 2005], [957, 2004]] + MATCH (comp:Organisation {id: w[0]}) + INSERT (p)-[:WORKS_AT {workFrom: w[1]}]->(comp) + - | + MATCH (p:Person {id: 10995116277976})-[r]->(x) + RETURN type(r), x.id, coalesce(r.classYear, r.workFrom) + expect: + rows: + - [HAS_INTEREST, 26, null] + - [IS_LOCATED_IN, 801, null] + - [STUDY_AT, 5522, 2003] + - [WORKS_AT, 955, 2005] + - [WORKS_AT, 957, 2004] + - [WORKS_AT, 965, 2004] + + # IU2. Add like to post: Asher likes a post of Aurora (LDBC update stream). + - name: iu2_add_like_to_post + params: + personId: 228 + postId: 137438955264 + creationDate: 1293549908486 + statements: + - | + MATCH (person:Person {id: $personId}), (post:Post {id: $postId}) + INSERT (person)-[:LIKES {creationDate: $creationDate}]->(post) + - | + MATCH (:Person {id: 228})-[l:LIKES]->(m:Post {id: 137438955264}) RETURN m.id, l.creationDate + expect: + rows: + - [137438955264, 1293549908486] + + # IU3. Add like to comment. No event of the LDBC update stream fits the dataset: + # Gary Hill likes a comment of Jie Yang. + - name: iu3_add_like_to_comment + params: + personId: 8796093022357 + commentId: 343597392326 + creationDate: 1290687902110 + statements: + - | + MATCH (person:Person {id: $personId}), (comment:Comment {id: $commentId}) + INSERT (person)-[:LIKES {creationDate: $creationDate}]->(comment) + - | + MATCH (:Person {id: 8796093022357})-[l:LIKES]->(c:Comment) RETURN c.id, l.creationDate + expect: + rows: + - [343597392326, 1290687902110] + + # IU4. Add forum: Album 2 of John Kumar, from the LDBC update stream (tag list inline). + - name: iu4_add_forum + params: + moderatorPersonId: 41 + forumId: 343597384372 + forumTitle: "Album 2 of John Kumar" + creationDate: 1291170270978 + statements: + - | + MATCH (p:Person {id: $moderatorPersonId}) + INSERT (f:Forum {id: $forumId, title: $forumTitle, creationDate: $creationDate})-[:HAS_MODERATOR]->(p) + - | + MATCH (f:Forum {id: $forumId}) + FOR tagId IN [1410] + MATCH (t:Tag {id: tagId}) + INSERT (f)-[:HAS_TAG]->(t) + - | + MATCH (x)-[r]->(y) WHERE x.id = 343597384372 OR y.id = 343597384372 RETURN x.id, type(r), y.id + expect: + rows: + - [343597384372, HAS_MODERATOR, 41] + - [343597384372, HAS_TAG, 1410] + + # IU5. Add forum membership: Jae-Jin joins a forum (LDBC update stream). + - name: iu5_add_forum_membership + params: + forumId: 206158431081 + personId: 76 + joinDate: 1291175534625 + statements: + - | + MATCH (f:Forum {id: $forumId}), (p:Person {id: $personId}) + INSERT (f)-[:HAS_MEMBER {joinDate: $joinDate}]->(p) + - | + MATCH (:Forum {id: 206158431081})-[h:HAS_MEMBER]->(p:Person {id: 76}) RETURN p.id, h.joinDate + expect: + rows: + - [76, 1291175534625] + + # IU6. Add post: Jae-Jin posts about Emilio Aguinaldo (LDBC update stream, tag list inline). + # The left arrows must store author <- post <- forum. + - name: iu6_add_post + params: + authorPersonId: 76 + countryId: 98 + forumId: 767 + postId: 412316868991 + creationDate: 1292734467726 + locationIP: "27.35.111.48" + browserUsed: "Chrome" + language: "uz" + content: "About Emilio Aguinaldo, nstrumental role during the Philippines' revolution against Spain, and t" + imageFile: "" + length: 96 + statements: + - | + MATCH (author:Person {id: $authorPersonId}), (country:Country {id: $countryId}), (forum:Forum {id: $forumId}) + INSERT (author)<-[:HAS_CREATOR]-(p:Post:Message { + id: $postId, + creationDate: $creationDate, + locationIP: $locationIP, + browserUsed: $browserUsed, + language: $language, + content: CASE $content WHEN '' THEN NULL ELSE $content END, + imageFile: CASE $imageFile WHEN '' THEN NULL ELSE $imageFile END, + length: $length + })<-[:CONTAINER_OF]-(forum), (p)-[:IS_LOCATED_IN]->(country) + - | + MATCH (p:Post {id: $postId}) + FOR tagId IN [1538] + MATCH (t:Tag {id: tagId}) + INSERT (p)-[:HAS_TAG]->(t) + - | + MATCH (x)-[r]->(y) WHERE x.id = 412316868991 OR y.id = 412316868991 RETURN x.id, type(r), y.id + expect: + rows: + - [767, CONTAINER_OF, 412316868991] + - [412316868991, HAS_CREATOR, 76] + - [412316868991, HAS_TAG, 1538] + - [412316868991, IS_LOCATED_IN, 98] + + # IU6. Add post: the properties (an empty imageFile is stored as null). + - name: iu6_add_post_properties + params: + authorPersonId: 76 + countryId: 98 + forumId: 767 + postId: 412316868991 + creationDate: 1292734467726 + locationIP: "27.35.111.48" + browserUsed: "Chrome" + language: "uz" + content: "About Emilio Aguinaldo, nstrumental role during the Philippines' revolution against Spain, and t" + imageFile: "" + length: 96 + statements: + - | + MATCH (author:Person {id: $authorPersonId}), (country:Country {id: $countryId}), (forum:Forum {id: $forumId}) + INSERT (author)<-[:HAS_CREATOR]-(p:Post:Message { + id: $postId, + creationDate: $creationDate, + locationIP: $locationIP, + browserUsed: $browserUsed, + language: $language, + content: CASE $content WHEN '' THEN NULL ELSE $content END, + imageFile: CASE $imageFile WHEN '' THEN NULL ELSE $imageFile END, + length: $length + })<-[:CONTAINER_OF]-(forum), (p)-[:IS_LOCATED_IN]->(country) + - | + MATCH (p:Post {id: $postId}) + FOR tagId IN [1538] + MATCH (t:Tag {id: tagId}) + INSERT (p)-[:HAS_TAG]->(t) + - | + MATCH (p:Post:Message {id: 412316868991}) RETURN p.content, p.imageFile, p.language, p.length, p.browserUsed + expect: + rows: + - ["About Emilio Aguinaldo, nstrumental role during the Philippines' revolution against Spain, and t", null, uz, 96, Chrome] + + # IU7. Add comment: Jie Yang replies to the post of IU6 (LDBC update stream, tag list + # inline). The message id is $replyToPostId + $replyToCommentId + 1. + - name: iu7_add_comment + setup: + - | + MATCH (author:Person {id: 76}), (country:Country {id: 98}), (forum:Forum {id: 767}) + INSERT (author)<-[:HAS_CREATOR]-(p:Post:Message { + id: 412316868991, + creationDate: 1292734467726, + locationIP: '27.35.111.48', + browserUsed: 'Chrome', + language: 'uz', + content: CASE 'About Emilio Aguinaldo, nstrumental role during the Philippines\' revolution against Spain, and t' WHEN '' THEN NULL ELSE 'About Emilio Aguinaldo, nstrumental role during the Philippines\' revolution against Spain, and t' END, + imageFile: CASE '' WHEN '' THEN NULL ELSE '' END, + length: 96 + })<-[:CONTAINER_OF]-(forum), (p)-[:IS_LOCATED_IN]->(country) + - | + MATCH (p:Post {id: 412316868991}) + FOR tagId IN [1538] + MATCH (t:Tag {id: tagId}) + INSERT (p)-[:HAS_TAG]->(t) + params: + authorPersonId: 6597069766775 + countryId: 1 + replyToPostId: 412316868991 + replyToCommentId: -1 + commentId: 412316868996 + creationDate: 1292750579384 + locationIP: "1.4.5.93" + browserUsed: "Internet Explorer" + content: "About Joss Whedon, film Dr. Horrible's Sing-Along Blog (2008). Whedon co-wrote and produced the horror film The Cabin in the Woods (2012), and wrote and directed the f" + length: 168 + statements: + - | + MATCH + (author:Person {id: $authorPersonId}), + (country:Country {id: $countryId}), + (message:Message {id: $replyToPostId + $replyToCommentId + 1}) + INSERT (author)<-[:HAS_CREATOR]-(c:Comment:Message { + id: $commentId, + creationDate: $creationDate, + locationIP: $locationIP, + browserUsed: $browserUsed, + content: $content, + length: $length + })-[:REPLY_OF]->(message), + (c)-[:IS_LOCATED_IN]->(country) + - | + MATCH (c:Comment {id: $commentId}) + FOR tagId IN [3075] + MATCH (t:Tag {id: tagId}) + INSERT (c)-[:HAS_TAG]->(t) + - | + MATCH (x)-[r]->(y) WHERE x.id = 412316868996 OR y.id = 412316868996 RETURN x.id, type(r), y.id + expect: + rows: + - [412316868996, HAS_CREATOR, 6597069766775] + - [412316868996, HAS_TAG, 3075] + - [412316868996, IS_LOCATED_IN, 1] + - [412316868996, REPLY_OF, 412316868991] + + # IU8. Add friendship: Gary Hill and Javed Khan (LDBC update stream). + - name: iu8_add_friendship + params: + person1Id: 8796093022357 + person2Id: 10995116277918 + creationDate: 1290895862836 + statements: + - | + MATCH (p1:Person {id: $person1Id}), (p2:Person {id: $person2Id}) + INSERT (p1)-[:KNOWS {creationDate: $creationDate}]->(p2) + - | + MATCH (:Person {id: 8796093022357})-[k:KNOWS]->(p:Person {id: 10995116277918}) RETURN p.id, k.creationDate + expect: + rows: + - [10995116277918, 1290895862836] diff --git a/tests/spec/lpg/sql_pgq/ldbc_snb_interactive.gtest b/tests/spec/lpg/sql_pgq/ldbc_snb_interactive.gtest new file mode 100644 index 000000000..d7e247adc --- /dev/null +++ b/tests/spec/lpg/sql_pgq/ldbc_snb_interactive.gtest @@ -0,0 +1,565 @@ +# LDBC SNB Interactive v1: the read queries in SQL/PGQ +# +# SQL/PGQ (SQL:2023 GRAPH_TABLE) translations of the LDBC Social Network Benchmark Interactive workload v1 read +# queries that fit one SELECT over one GRAPH_TABLE. LDBC's own SQL implementations (PostgreSQL, DuckDB) query +# relational tables and cannot run on a graph; these translations follow the Cypher reference queries in +# lpg/cypher/ldbc_snb_interactive.gtest. A WHERE comes after the OPTIONAL MATCH clauses (Grafeo's GRAPH_TABLE takes +# one WHERE, last); walks, optional matches and DISTINCT aggregates replace the WITH stages of the Cypher queries. +# +# Not translated, since Grafeo's SQL/PGQ has no form for them yet: IC7 (the latest like per liker needs a subquery +# or a window function), IC10 (a negative pattern inside GRAPH_TABLE and a date from epoch milliseconds), IC13 (a +# shortest-path search inside GRAPH_TABLE), IC14 (ALL SHORTEST and a per-edge weight subquery) and the updates +# (no data modification in SQL/PGQ). +# +# Dataset: ldbc_snb_mini, a small slice of the LDBC SNB Interactive v1 test data, every value copied unchanged (see +# its .setup file). The parameters are chosen from that slice; IC13 also uses the LDBC parameter pair (Gary Hill, +# Abdullah Koksal). The expected rows come from LDBC's own DuckDB (SQL) implementation run on the same slice, in the +# result shape of the Cypher reference queries. +# +# A skipped case names the bug it hits (and the issue, when there is one); remove the skip once it is fixed. + +meta: + language: sql-pgq + model: lpg + section: ldbc + title: LDBC SNB Interactive v1 (SQL/PGQ) + dataset: ldbc_snb_mini + +tests: + + # --------------------------------------------------------------------------- + # Short reads + # --------------------------------------------------------------------------- + + # IS1. Profile of a person. + - name: is1_profile_of_a_person + params: + personId: 228 + query: | + SELECT firstName, lastName, birthday, locationIP, browserUsed, cityId, gender, creationDate + FROM GRAPH_TABLE ( + MATCH (n:Person {id: $personId})-[:IS_LOCATED_IN]->(p:City) + COLUMNS (n.firstName AS firstName, n.lastName AS lastName, n.birthday AS birthday, + n.locationIP AS locationIP, n.browserUsed AS browserUsed, p.id AS cityId, + n.gender AS gender, n.creationDate AS creationDate) + ) + expect: + ordered: true + rows: + - [Asher, Mamo, 524016000000, "213.55.93.153", Chrome, 1127, female, 1266720721912] + + # IS2. Recent messages of a person. + - name: is2_recent_messages_of_a_person + params: + personId: 228 + query: | + SELECT messageId, messageContent, messageCreationDate, postId, personId, personFirstName, personLastName + FROM GRAPH_TABLE ( + MATCH (:Person {id: $personId})<-[:HAS_CREATOR]-(message:Message)-[:REPLY_OF*0..]->(post:Post) + -[:HAS_CREATOR]->(person:Person) + COLUMNS (message.id AS messageId, COALESCE(message.imageFile, message.content) AS messageContent, + message.creationDate AS messageCreationDate, post.id AS postId, person.id AS personId, + person.firstName AS personFirstName, person.lastName AS personLastName) + ) + ORDER BY messageCreationDate DESC, messageId ASC + LIMIT 10 + expect: + ordered: true + rows: + - [343597393779, "photo343597393779.jpg", 1290590906265, 343597393779, 228, Asher, Mamo] + - [343597393778, "photo343597393778.jpg", 1290590905265, 343597393778, 228, Asher, Mamo] + - [343597393747, "About Luis Horna, as a strong serve for a relatively shoAbout Djibouti, uti National Army and its sub-branchesAbout Leonard Bernstein, hilharmonic, ", 1289066095205, 343597393747, 228, Asher, Mamo] + - [343597392325, "About Ne-Yo, November 22, 2010. InAbout Brian Wilson, elled for various reaAbout V", 1289049183991, 343597392324, 76, Jae-Jin, Park] + - [206158440292, "photo206158440292.jpg", 1279208419298, 206158440292, 228, Asher, Mamo] + - [206158431893, roflol, 1278531351427, 206158431892, 102, Philibert, Roindefo] + - [137438963511, "About William Lyon Mackenzie King, y. King worked to bring compromise andAbout Thomas Jefferson, erson, his wife", 1277076649771, 137438963511, 228, Asher, Mamo] + - [137438963744, "About Christopher Lee, or services to dramAbout Tina Turner, inning with a", 1273650346399, 137438963740, 150, Alfonso, Alvarez] + - [137438963499, "About William Lyon Mackenzie King, allowed his intense spirituality to distort About T", 1272740097725, 137438963499, 228, Asher, Mamo] + - [68719478457, "About Oscar Wilde, intellectuals. Their sAbout Anne, Queen of Great Britain, II became joint monarcAbout Lebanon, ld national infrastrucAbout Bad, Bad L", 1271850332021, 68719478453, 2199023255712, Aurora, Cruz] + + # IS3. Friends of a person. + - name: is3_friends_of_a_person + params: + personId: 228 + query: | + SELECT personId, firstName, lastName, friendshipCreationDate + FROM GRAPH_TABLE ( + MATCH (n:Person {id: $personId})-[r:KNOWS]-(friend:Person) + COLUMNS (friend.id AS personId, friend.firstName AS firstName, friend.lastName AS lastName, + r.creationDate AS friendshipCreationDate) + ) + ORDER BY friendshipCreationDate DESC, personId ASC + expect: + ordered: true + rows: + - [8796093022357, Gary, Hill, 1288580721183] + - [2199023255712, Aurora, Cruz, 1271536640884] + - [102, Philibert, Roindefo, 1268755084867] + - [76, Jae-Jin, Park, 1267889385714] + - [150, Alfonso, Alvarez, 1267126413921] + + # IS4. Content of a message. + - name: is4_content_of_a_message_post + params: + messageId: 10169 + query: | + SELECT messageCreationDate, messageContent + FROM GRAPH_TABLE ( + MATCH (m:Message {id: $messageId}) + COLUMNS (m.creationDate AS messageCreationDate, COALESCE(m.content, m.imageFile) AS messageContent) + ) + expect: + ordered: true + rows: + - [1267290597839, "photo10169.jpg"] + + # IS4. Content of a message. + - name: is4_content_of_a_message_comment + params: + messageId: 343597392326 + query: | + SELECT messageCreationDate, messageContent + FROM GRAPH_TABLE ( + MATCH (m:Message {id: $messageId}) + COLUMNS (m.creationDate AS messageCreationDate, COALESCE(m.content, m.imageFile) AS messageContent) + ) + expect: + ordered: true + rows: + - [1289092584789, "no way!"] + + # IS5. Creator of a message. + - name: is5_creator_of_a_message + params: + messageId: 343597392326 + query: | + SELECT personId, firstName, lastName + FROM GRAPH_TABLE ( + MATCH (m:Message {id: $messageId})-[:HAS_CREATOR]->(p:Person) + COLUMNS (p.id AS personId, p.firstName AS firstName, p.lastName AS lastName) + ) + expect: + ordered: true + rows: + - [6597069766775, Jie, Yang] + + # IS6. Forum of a message. + - name: is6_forum_of_a_message + params: + messageId: 343597393752 + query: | + SELECT forumId, forumTitle, moderatorId, moderatorFirstName, moderatorLastName + FROM GRAPH_TABLE ( + MATCH (m:Message {id: $messageId})-[:REPLY_OF*0..]->(p:Post)<-[:CONTAINER_OF]-(f:Forum) + -[:HAS_MODERATOR]->(mod:Person) + COLUMNS (f.id AS forumId, f.title AS forumTitle, mod.id AS moderatorId, + mod.firstName AS moderatorFirstName, mod.lastName AS moderatorLastName) + ) + expect: + ordered: true + rows: + - [872, "Wall of Asher Mamo", 228, Asher, Mamo] + + # IS7. Replies of a message. + - name: is7_replies_of_a_message + params: + messageId: 137438963499 + query: | + SELECT commentId, commentContent, commentCreationDate, replyAuthorId, replyAuthorFirstName, + replyAuthorLastName, replyAuthorKnowsOriginalMessageAuthor + FROM GRAPH_TABLE ( + MATCH (m:Message {id: $messageId})<-[:REPLY_OF]-(c:Comment)-[:HAS_CREATOR]->(p:Person) + OPTIONAL MATCH (m)-[:HAS_CREATOR]->(a:Person)-[r:KNOWS]-(p) + COLUMNS (c.id AS commentId, c.content AS commentContent, c.creationDate AS commentCreationDate, + p.id AS replyAuthorId, p.firstName AS replyAuthorFirstName, p.lastName AS replyAuthorLastName, + a.id IS NOT NULL AS replyAuthorKnowsOriginalMessageAuthor) + ) + ORDER BY commentCreationDate DESC, replyAuthorId ASC + expect: + ordered: true + rows: + - [137438963506, "About Thomas Hardy, y published as serials in magazines, weAbout Thomas Jeffe", 1272757138003, 150, Alfonso, Alvarez, true] + - [137438963501, "About Frank Zappa, ollages. His later albums shared this eclectic and experimental approach, irrespectiv", 1272755669544, 102, Philibert, Roindefo, true] + + + # --------------------------------------------------------------------------- + # Complex reads + # --------------------------------------------------------------------------- + + # IC1. Transitive friends with certain name. + - name: ic1_transitive_friends_with_a_name_maria + params: + personId: 228 + firstName: "Maria" + query: | + SELECT friendId, friendLastName, MIN(distance) AS distanceFromPerson, friendBirthday, friendCreationDate, + friendGender, friendBrowserUsed, friendLocationIp, friendEmails, friendLanguages, friendCityName, + COLLECT(DISTINCT university) AS friendUniversities, COLLECT(DISTINCT company) AS friendCompanies + FROM GRAPH_TABLE ( + MATCH (p:Person {id: $personId})-[k:KNOWS*1..3]-(friend:Person)-[:IS_LOCATED_IN]->(friendCity:City) + OPTIONAL MATCH (friend)-[studyAt:STUDY_AT]->(uni:University)-[:IS_LOCATED_IN]->(uniCity:City) + OPTIONAL MATCH (friend)-[workAt:WORK_AT]->(company:Company)-[:IS_LOCATED_IN]->(companyCountry:Country) + WHERE friend.firstName = $firstName AND friend.id <> p.id + COLUMNS (friend.id AS friendId, friend.lastName AS friendLastName, LENGTH(k) AS distance, + friend.birthday AS friendBirthday, friend.creationDate AS friendCreationDate, + friend.gender AS friendGender, friend.browserUsed AS friendBrowserUsed, + friend.locationIP AS friendLocationIp, friend.email AS friendEmails, friend.speaks AS friendLanguages, + friendCity.name AS friendCityName, + CASE WHEN uni IS NULL THEN NULL ELSE [uni.name, studyAt.classYear, uniCity.name] END AS university, + CASE WHEN company IS NULL THEN NULL + ELSE [company.name, workAt.workFrom, companyCountry.name] END AS company) + ) + GROUP BY friendId, friendLastName, friendBirthday, friendCreationDate, friendGender, friendBrowserUsed, + friendLocationIp, friendEmails, friendLanguages, friendCityName + ORDER BY distanceFromPerson ASC, friendLastName ASC, friendId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [143, Alkaios, 2, 410659200000, 1262456643976, female, Firefox, "62.217.119.183", "[Maria143@gmail.com]", "[fr, en]", Athens, "[[National_and_Kapodistrian_University_of_Athens, 2003, Athens]]", "[[Macedonian_Airlines, 2004, Greece]]"] + + # IC1. Transitive friends with certain name. + - name: ic1_transitive_friends_with_a_name_john + params: + personId: 228 + firstName: "John" + query: | + SELECT friendId, friendLastName, MIN(distance) AS distanceFromPerson, friendBirthday, friendCreationDate, + friendGender, friendBrowserUsed, friendLocationIp, friendEmails, friendLanguages, friendCityName, + COLLECT(DISTINCT university) AS friendUniversities, COLLECT(DISTINCT company) AS friendCompanies + FROM GRAPH_TABLE ( + MATCH (p:Person {id: $personId})-[k:KNOWS*1..3]-(friend:Person)-[:IS_LOCATED_IN]->(friendCity:City) + OPTIONAL MATCH (friend)-[studyAt:STUDY_AT]->(uni:University)-[:IS_LOCATED_IN]->(uniCity:City) + OPTIONAL MATCH (friend)-[workAt:WORK_AT]->(company:Company)-[:IS_LOCATED_IN]->(companyCountry:Country) + WHERE friend.firstName = $firstName AND friend.id <> p.id + COLUMNS (friend.id AS friendId, friend.lastName AS friendLastName, LENGTH(k) AS distance, + friend.birthday AS friendBirthday, friend.creationDate AS friendCreationDate, + friend.gender AS friendGender, friend.browserUsed AS friendBrowserUsed, + friend.locationIP AS friendLocationIp, friend.email AS friendEmails, friend.speaks AS friendLanguages, + friendCity.name AS friendCityName, + CASE WHEN uni IS NULL THEN NULL ELSE [uni.name, studyAt.classYear, uniCity.name] END AS university, + CASE WHEN company IS NULL THEN NULL + ELSE [company.name, workAt.workFrom, companyCountry.name] END AS company) + ) + GROUP BY friendId, friendLastName, friendBirthday, friendCreationDate, friendGender, friendBrowserUsed, + friendLocationIp, friendEmails, friendLanguages, friendCityName + ORDER BY distanceFromPerson ASC, friendLastName ASC, friendId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [41, Kumar, 3, 527731200000, 1266276257359, male, Safari, "27.116.33.147", "[John41@gmail.com, John41@jizan.cc, John41@yahoo.com, John41@zoho.com]", "[gu, mr, en]", Puttur, "[[The_Oxford_Educational_Institutions, 2004, Bangalore]]", "[[Jet_Airways, 2005, India], [Jagson_Airlines, 2005, India], [Deccan_360, 2006, India]]"] + + # IC2. Recent messages by your friends. + - name: ic2_recent_messages_by_friends + params: + personId: 228 + maxDate: 1285891200000 + query: | + SELECT personId, personFirstName, personLastName, postOrCommentId, postOrCommentContent, postOrCommentCreationDate + FROM GRAPH_TABLE ( + MATCH (:Person {id: $personId})-[:KNOWS]-(friend:Person)<-[:HAS_CREATOR]-(message:Message) + WHERE message.creationDate <= $maxDate + COLUMNS (friend.id AS personId, friend.firstName AS personFirstName, friend.lastName AS personLastName, + message.id AS postOrCommentId, COALESCE(message.content, message.imageFile) AS postOrCommentContent, + message.creationDate AS postOrCommentCreationDate) + ) + ORDER BY postOrCommentCreationDate DESC, postOrCommentId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [2199023255712, Aurora, Cruz, 274877909928, "no way!", 1285681173411] + - [76, Jae-Jin, Park, 274877909925, thx, 1285676860041] + - [102, Philibert, Roindefo, 274877912148, fine, 1284597772686] + - [76, Jae-Jin, Park, 206158432797, thx, 1280907556754] + - [102, Philibert, Roindefo, 206158431892, "About William Penn, – 16 September 1670) was an English admiral and politician who sat in the House of Commons from 1660 to 1670. He was the father of Wi", 1278511907643] + - [2199023255712, Aurora, Cruz, 206158432090, "photo206158432090.jpg", 1277597415173] + - [102, Philibert, Roindefo, 137438958564, cool, 1277084387523] + - [2199023255712, Aurora, Cruz, 137438963517, "I see", 1277076710526] + - [2199023255712, Aurora, Cruz, 137438955279, "photo137438955279.jpg", 1276553301898] + - [2199023255712, Aurora, Cruz, 137438955277, "photo137438955277.jpg", 1276553299898] + - [2199023255712, Aurora, Cruz, 137438955264, "photo137438955264.jpg", 1276274322999] + - [76, Jae-Jin, Park, 137438963765, right, 1273696787135] + - [150, Alfonso, Alvarez, 137438963740, "About Pope Paul VI, Church life during his pontificate excAbout Winston Churchill, United Kingdom during the Second WorldAbout Julia Gillard, ing Mitcham Demonstration School and UnAbout S", 1273649436061] + - [150, Alfonso, Alvarez, 137438963751, "About Julia Gillard, ister upon Labor's victory in the 2007 federal electionAbout Chile, Republic of Chile, is a country in South America occupyAbout South Korea, ith production focusing on electronics, automobiles, ", 1273617306061] + - [2199023255712, Aurora, Cruz, 137438963510, great, 1272813417041] + - [102, Philibert, Roindefo, 137438963508, "no way!", 1272758692818] + - [150, Alfonso, Alvarez, 137438963502, "About Stephen Sondheim, an composer and lyricist known for his contributions to musical th", 1272758679985] + - [150, Alfonso, Alvarez, 137438963506, "About Thomas Hardy, y published as serials in magazines, weAbout Thomas Jeffe", 1272757138003] + - [102, Philibert, Roindefo, 137438963501, "About Frank Zappa, ollages. His later albums shared this eclectic and experimental approach, irrespectiv", 1272755669544] + - [150, Alfonso, Alvarez, 68719487342, ok, 1271910621171] + + # IC3. Friends and friends of friends that have been to given countries. + - name: ic3_friends_in_countries_x_and_y + skip: "an aggregate over a CASE expression breaks GROUP BY: the grouping columns come out as 0 (friendFirstName, friendLastName)" + params: + personId: 228 + countryXName: "Uruguay" + countryYName: "Canada" + startDate: 1275350400000 + endDate: 1277942400000 + query: | + SELECT friendId, friendFirstName, friendLastName, + COUNT(DISTINCT CASE WHEN country = $countryXName THEN messageId END) AS xCount, + COUNT(DISTINCT CASE WHEN country = $countryYName THEN messageId END) AS yCount, + COUNT(DISTINCT messageId) AS xyCount + FROM GRAPH_TABLE ( + MATCH (person:Person {id: $personId})-[:KNOWS*1..2]-(friend:Person)-[:IS_LOCATED_IN]->(:City) + -[:IS_PART_OF]->(home:Country), + (friend)<-[:HAS_CREATOR]-(message:Message)-[:IS_LOCATED_IN]->(c:Country) + WHERE friend.id <> person.id AND home.name <> $countryXName AND home.name <> $countryYName + AND c.name IN [$countryXName, $countryYName] + AND message.creationDate >= $startDate AND message.creationDate < $endDate + COLUMNS (friend.id AS friendId, friend.firstName AS friendFirstName, friend.lastName AS friendLastName, + message.id AS messageId, c.name AS country) + ) + GROUP BY friendId, friendFirstName, friendLastName + HAVING xCount > 0 AND yCount > 0 + ORDER BY xyCount DESC, friendId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [2199023255712, Aurora, Cruz, 1, 2, 3] + + # IC4. New topics. + - name: ic4_new_topics + skip: "an aggregate over a CASE expression breaks GROUP BY: no rows" + params: + personId: 228 + startDate: 1272672000000 + endDate: 1275264000000 + query: | + SELECT tagName, + COUNT(DISTINCT CASE WHEN creationDate >= $startDate AND creationDate < $endDate THEN postId END) AS postCount, + COUNT(DISTINCT CASE WHEN creationDate < $startDate THEN postId END) AS oldPostCount + FROM GRAPH_TABLE ( + MATCH (person:Person {id: $personId})-[:KNOWS]-(friend:Person)<-[:HAS_CREATOR]-(post:Post)-[:HAS_TAG]->(tag:Tag) + COLUMNS (tag.name AS tagName, post.id AS postId, post.creationDate AS creationDate) + ) + GROUP BY tagName + HAVING postCount > 0 AND oldPostCount = 0 + ORDER BY postCount DESC, tagName ASC + LIMIT 10 + expect: + ordered: true + rows: + - [Julia_Gillard, 2, 0] + - [Chile, 1, 0] + - [Monaco, 1, 0] + - [Nigeria, 1, 0] + - [Pope_Paul_VI, 1, 0] + - [South_Korea, 1, 0] + - [South_Vietnam, 1, 0] + - [Winston_Churchill, 1, 0] + + # IC5. New groups. + - name: ic5_new_groups + skip: "LIMIT is applied before GROUP BY: 3 of the 16 forums are missing" + params: + personId: 228 + minDate: 1275350400000 + query: | + SELECT forumName, COUNT(DISTINCT postId) AS postCount + FROM GRAPH_TABLE ( + MATCH (person:Person {id: $personId})-[:KNOWS*1..2]-(friend:Person)<-[membership:HAS_MEMBER]-(forum:Forum) + OPTIONAL MATCH (friend)<-[:HAS_CREATOR]-(post:Post)<-[:CONTAINER_OF]-(forum) + WHERE friend.id <> person.id AND membership.joinDate > $minDate + COLUMNS (forum.id AS forumId, forum.title AS forumName, post.id AS postId) + ) + GROUP BY forumId, forumName + ORDER BY postCount DESC, forumId ASC + LIMIT 20 + expect: + ordered: true + rows: + - ["Wall of Maria Alkaios", 0] + - ["Wall of Jae-Jin Park", 0] + - ["Wall of Asher Mamo", 0] + - ["Album 9 of Asher Mamo", 0] + - ["Wall of Alfonso Alvarez", 0] + - ["Wall of Abdala Ndiaye", 0] + - ["Wall of Aurora Cruz", 0] + - ["Album 3 of Asher Mamo", 0] + - ["Album 1 of Aurora Cruz", 0] + - ["Album 2 of Aurora Cruz", 0] + - ["Wall of Rafael Fernández", 0] + - ["Album 7 of Aurora Cruz", 0] + - ["Wall of Jie Yang", 0] + - ["Album 0 of Asher Mamo", 0] + - ["Album 6 of Alfonso Alvarez", 0] + - ["Album 8 of Abdala Ndiaye", 0] + + # IC6. Tag co-occurrence. + - name: ic6_tag_co_occurrence + skip: "a property map on a later node of the pattern is ignored ((:Tag {name: $tagName})), so every post of a friend counts" + params: + personId: 228 + tagName: "Luis_Horna" + query: | + SELECT tagName, COUNT(DISTINCT postId) AS postCount + FROM GRAPH_TABLE ( + MATCH (person:Person {id: $personId})-[:KNOWS*1..2]-(friend:Person)<-[:HAS_CREATOR]-(post:Post) + -[:HAS_TAG]->(:Tag {name: $tagName}), + (post)-[:HAS_TAG]->(tag:Tag) + WHERE friend.id <> person.id AND tag.name <> $tagName + COLUMNS (tag.name AS tagName, post.id AS postId) + ) + GROUP BY tagName + ORDER BY postCount DESC, tagName ASC + LIMIT 10 + expect: + ordered: true + rows: + - [Brian_Wilson, 1] + - [Gibraltar, 1] + - ["Harry_S._Truman", 1] + - [Israel, 1] + - [Republic_of_the_Congo, 1] + - [Robert_Altman, 1] + - [Xiongnu, 1] + + # IC8. Recent replies. + - name: ic8_recent_replies + params: + personId: 228 + query: | + SELECT personId, personFirstName, personLastName, commentCreationDate, commentId, commentContent + FROM GRAPH_TABLE ( + MATCH (start:Person {id: $personId})<-[:HAS_CREATOR]-(:Message)<-[:REPLY_OF]-(comment:Comment) + -[:HAS_CREATOR]->(person:Person) + COLUMNS (person.id AS personId, person.firstName AS personFirstName, person.lastName AS personLastName, + comment.creationDate AS commentCreationDate, comment.id AS commentId, + comment.content AS commentContent) + ) + ORDER BY commentCreationDate DESC, commentId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [76, Jae-Jin, Park, 1289135518830, 343597393750, "About Luis Horna, e surface is clay. He was the About Waiting for the End, ough it was "] + - [6597069766775, Jie, Yang, 1289092584789, 343597392326, "no way!"] + - [102, Philibert, Roindefo, 1289090250676, 343597393755, "About Al Capone, ing the money he made from his actAbout Djibouti, military forces of Djibouti and coAbout Leonard Bernstein, talent"] + - [2199023255712, Aurora, Cruz, 1277076710526, 137438963517, "I see"] + - [150, Alfonso, Alvarez, 1272757138003, 137438963506, "About Thomas Hardy, y published as serials in magazines, weAbout Thomas Jeffe"] + - [102, Philibert, Roindefo, 1272755669544, 137438963501, "About Frank Zappa, ollages. His later albums shared this eclectic and experimental approach, irrespectiv"] + - [143, Maria, Alkaios, 1271884315385, 68719478447, "About Kevin Rudd, ia's remaining About William Morris, traditional teAbout France, c"] + + # IC9. Recent messages by friends or friends of friends. + - name: ic9_recent_messages_by_friends_of_friends + skip: "LIMIT is applied before DISTINCT: 10 rows instead of 20" + params: + personId: 228 + maxDate: 1285891200000 + query: | + SELECT DISTINCT personId, personFirstName, personLastName, commentOrPostId, commentOrPostContent, + commentOrPostCreationDate + FROM GRAPH_TABLE ( + MATCH (root:Person {id: $personId})-[:KNOWS*1..2]-(friend:Person)<-[:HAS_CREATOR]-(message:Message) + WHERE friend.id <> root.id AND message.creationDate < $maxDate + COLUMNS (friend.id AS personId, friend.firstName AS personFirstName, friend.lastName AS personLastName, + message.id AS commentOrPostId, COALESCE(message.content, message.imageFile) AS commentOrPostContent, + message.creationDate AS commentOrPostCreationDate) + ) + ORDER BY commentOrPostCreationDate DESC, commentOrPostId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [143, Maria, Alkaios, 274877909920, roflol, 1285685102274] + - [6597069766775, Jie, Yang, 274877909919, "About Manuel Noriega, uest in April 2010. He arrived in Paris on April 27, 2010, and after a re-trial as a condition", 1285682584114] + - [2199023255712, Aurora, Cruz, 274877909928, "no way!", 1285681173411] + - [76, Jae-Jin, Park, 274877909925, thx, 1285676860041] + - [143, Maria, Alkaios, 274877909929, "I see", 1285665988168] + - [4398046511333, Rafael, "Fernández", 274877909927, "About Manuel Noriega, trafficking, racketeering, andAbout Mikhail Gorbachev, was the onl", 1285665306179] + - [6597069766775, Jie, Yang, 274877909924, "About Manuel Noriega, g, racketeering, and money laundering About Arthur Wellesley, 1st Duke of Wellington, m. He", 1285664359114] + - [4398046511333, Rafael, "Fernández", 274877912142, "no", 1284613964426] + - [102, Philibert, Roindefo, 274877912148, fine, 1284597772686] + - [143, Maria, Alkaios, 274877912139, "About Enrique Iglesias, hits on the various Billboard charts. Billboard has called him The King of Latin Pop and The King of DanAbout California King Bed, ive reviews from music critics, who praised Rihanna", 1284593996258] + - [4398046511333, Rafael, "Fernández", 274877909514, "About Guy Sebastian, 08 Australian tour. Like It Like That has three tracks with John Mayer o", 1283465660488] + - [143, Maria, Alkaios, 274877913531, right, 1283303511849] + - [153, Abdala, Ndiaye, 206158432792, duh, 1280914332302] + - [76, Jae-Jin, Park, 206158432797, thx, 1280907556754] + - [4398046511333, Rafael, "Fernández", 206158432782, "About Guy Sebastian, ur Asian countries and New Zealand. Sebastian had a second number one in New Zealand with Who's That Girl, two other top ten singles and a number three album, and gained four platinum and two gold certifications there. He ha", 1280894717004] + - [143, Maria, Alkaios, 206158431896, "yes", 1278512054636] + - [102, Philibert, Roindefo, 206158431892, "About William Penn, – 16 September 1670) was an English admiral and politician who sat in the House of Commons from 1660 to 1670. He was the father of Wi", 1278511907643] + - [2199023255712, Aurora, Cruz, 206158432090, "photo206158432090.jpg", 1277597415173] + - [102, Philibert, Roindefo, 137438958564, cool, 1277084387523] + - [2199023255712, Aurora, Cruz, 137438963517, "I see", 1277076710526] + + # IC11. Job referral. + - name: ic11_job_referral + skip: "a property map on a later node of the pattern is ignored ((:Country {name: $countryName})), so every company counts" + params: + personId: 228 + countryName: "Philippines" + workFromYear: 2008 + query: | + SELECT DISTINCT personId, personFirstName, personLastName, organizationName, organizationWorkFromYear + FROM GRAPH_TABLE ( + MATCH (person:Person {id: $personId})-[:KNOWS*1..2]-(friend:Person)-[workAt:WORK_AT]->(company:Company) + -[:IS_LOCATED_IN]->(:Country {name: $countryName}) + WHERE friend.id <> person.id AND workAt.workFrom < $workFromYear + COLUMNS (friend.id AS personId, friend.firstName AS personFirstName, friend.lastName AS personLastName, + company.name AS organizationName, workAt.workFrom AS organizationWorkFromYear) + ) + ORDER BY organizationWorkFromYear ASC, personId ASC, organizationName DESC + LIMIT 10 + expect: + ordered: true + rows: + - [2199023255712, Aurora, Cruz, South_East_Asian_Airlines, 2007] + - [2199023255712, Aurora, Cruz, Filipinas_Orient_Airways, 2007] + + # IC12. Expert search. + - name: ic12_expert_search_office_holder + params: + personId: 228 + tagClassName: "OfficeHolder" + query: | + SELECT personId, personFirstName, personLastName, COLLECT(DISTINCT tagName) AS tagNames, + COUNT(DISTINCT commentId) AS replyCount + FROM GRAPH_TABLE ( + MATCH (:Person {id: $personId})-[:KNOWS]-(friend:Person)<-[:HAS_CREATOR]-(comment:Comment)-[:REPLY_OF]->(:Post) + -[:HAS_TAG]->(tag:Tag)-[:HAS_TYPE]->(:TagClass)-[:IS_SUBCLASS_OF*0..]->(baseTagClass:TagClass) + WHERE baseTagClass.name = $tagClassName + COLUMNS (friend.id AS personId, friend.firstName AS personFirstName, friend.lastName AS personLastName, + tag.name AS tagName, comment.id AS commentId) + ) + GROUP BY personId, personFirstName, personLastName + ORDER BY replyCount DESC, personId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [76, Jae-Jin, Park, "[Julia_Gillard, Arthur_Wellesley,_1st_Duke_of_Wellington]", 2] + - [102, Philibert, Roindefo, "[Aung_San_Suu_Kyi, William_Lyon_Mackenzie_King, Thomas_Jefferson]", 2] + - [150, Alfonso, Alvarez, "[William_Lyon_Mackenzie_King, Thomas_Jefferson]", 1] + - [2199023255712, Aurora, Cruz, "[William_Lyon_Mackenzie_King, Thomas_Jefferson]", 1] + + # IC12. Expert search. + - name: ic12_expert_search_tennis_player + params: + personId: 228 + tagClassName: "TennisPlayer" + query: | + SELECT personId, personFirstName, personLastName, COLLECT(DISTINCT tagName) AS tagNames, + COUNT(DISTINCT commentId) AS replyCount + FROM GRAPH_TABLE ( + MATCH (:Person {id: $personId})-[:KNOWS]-(friend:Person)<-[:HAS_CREATOR]-(comment:Comment)-[:REPLY_OF]->(:Post) + -[:HAS_TAG]->(tag:Tag)-[:HAS_TYPE]->(:TagClass)-[:IS_SUBCLASS_OF*0..]->(baseTagClass:TagClass) + WHERE baseTagClass.name = $tagClassName + COLUMNS (friend.id AS personId, friend.firstName AS personFirstName, friend.lastName AS personLastName, + tag.name AS tagName, comment.id AS commentId) + ) + GROUP BY personId, personFirstName, personLastName + ORDER BY replyCount DESC, personId ASC + LIMIT 20 + expect: + ordered: true + rows: + - [76, Jae-Jin, Park, "[Dudi_Sela, Luis_Horna]", 2] + - [150, Alfonso, Alvarez, "[Dudi_Sela, Venus_Williams]", 2] + - [102, Philibert, Roindefo, "[Luis_Horna]", 1] + - [2199023255712, Aurora, Cruz, "[Luis_Horna]", 1] diff --git a/tests/spec/lpg/sql_pgq/paths_and_optional.gtest b/tests/spec/lpg/sql_pgq/paths_and_optional.gtest index e9a08f8ca..b6dba12dd 100644 --- a/tests/spec/lpg/sql_pgq/paths_and_optional.gtest +++ b/tests/spec/lpg/sql_pgq/paths_and_optional.gtest @@ -40,6 +40,19 @@ tests: - [Alix, Gus] - [Gus, Vincent] + # The edge variable of a quantified pattern is the list of the path's + # edges, also for one hop. + - name: variable_length_1_to_1_binds_a_list + query: | + SELECT * FROM GRAPH_TABLE ( + MATCH (src:Person)-[p:KNOWS*1..1]->(dst:Person) + COLUMNS (src.name AS source, size(p) AS edges, LENGTH(p) AS hops) + ) + expect: + rows: + - [Alix, 1, 1] + - [Gus, 1, 1] + - name: variable_length_exactly_2 query: | SELECT * FROM GRAPH_TABLE ( diff --git a/tests/spec/rosetta/bound_before_a_later_match.gtest b/tests/spec/rosetta/bound_before_a_later_match.gtest new file mode 100644 index 000000000..fc8f51b13 --- /dev/null +++ b/tests/spec/rosetta/bound_before_a_later_match.gtest @@ -0,0 +1,331 @@ +# Rosetta: nodes and edges bound before a later MATCH +# +# A pattern that names an edge bound before matches that edge only: in a +# later MATCH, through WITH, in a CALL subquery that imports it, and in a +# path back to nodes a subquery imports. The first setup has two parallel +# KNOWS edges Alix->Gus (w 1 and 4) and a LIKES loop on Mia (w 6), so a +# check on the endpoints alone shows up as extra rows. +# +# Setup 1 (GQL, one statement): Alix, Gus, Vincent, Mia and Amsterdam. KNOWS +# Alix->Gus (w 1), Gus->Vincent (w 2), Vincent->Alix (w 3), Alix->Gus (w 4); +# LIVES_IN Alix->Amsterdam (w 5); LIKES Mia->Mia (w 6). +# +# The last section checks values carried through a later MATCH. Its setup +# gives nodes and edges a property with the same name (`w`) and overlapping +# IDs (nodes have 100 and up, edges 1 to 8; Jules and Mia have none), so a +# node or edge read as the other entity shows in the value. + +meta: + model: lpg + section: rosetta + title: Nodes and edges bound before a later MATCH + dataset: empty + requires: [cypher] + +tests: + + # --------------------------------------------------------------------------- + # A later pattern through a bound edge matches that edge + # --------------------------------------------------------------------------- + + - name: later_match_through_a_bound_edge + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH ()-[r]->() MATCH (x)-[r]->(y) RETURN r.w AS w, x.name AS x, y.name AS y" + cypher: "MATCH ()-[r]->() MATCH (x)-[r]->(y) RETURN r.w AS w, x.name AS x, y.name AS y" + expect: + rows: + - [1, Alix, Gus] + - [2, Gus, Vincent] + - [3, Vincent, Alix] + - [4, Alix, Gus] + - [5, Alix, Amsterdam] + - [6, Mia, Mia] + + - name: bound_edge_between_bound_nodes_matches_once + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH (a)-[r]->(b) MATCH (a)-[r]->(b) RETURN r.w AS w" + cypher: "MATCH (a)-[r]->(b) MATCH (a)-[r]->(b) RETURN r.w AS w" + expect: + rows: + - [1] + - [2] + - [3] + - [4] + - [5] + - [6] + + - name: bound_edge_read_backwards + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH (a)-[r]->(b) MATCH (x)<-[r]-(y) RETURN r.w AS w, x.name AS x, y.name AS y" + cypher: "MATCH (a)-[r]->(b) MATCH (x)<-[r]-(y) RETURN r.w AS w, x.name AS x, y.name AS y" + expect: + rows: + - [1, Gus, Alix] + - [2, Vincent, Gus] + - [3, Alix, Vincent] + - [4, Gus, Alix] + - [5, Amsterdam, Alix] + - [6, Mia, Mia] + + - name: bound_edge_backwards_between_its_own_ends_is_a_loop + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH (a)-[r]->(b) MATCH (a)<-[r]-(b) RETURN r.w AS w" + cypher: "MATCH (a)-[r]->(b) MATCH (a)<-[r]-(b) RETURN r.w AS w" + expect: + rows: + - [6] + + - name: bound_edge_undirected + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH (a)-[r:KNOWS {w: 2}]->(b) MATCH (x)-[r]-(y) RETURN x.name AS x, y.name AS y" + cypher: "MATCH (a)-[r:KNOWS {w: 2}]->(b) MATCH (x)-[r]-(y) RETURN x.name AS x, y.name AS y" + expect: + rows: + - [Gus, Vincent] + - [Vincent, Gus] + + - name: bound_edge_of_another_type_matches_nothing + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH ()-[r:KNOWS]->() MATCH ()-[r:LIVES_IN]->() RETURN count(*) AS n" + cypher: "MATCH ()-[r:KNOWS]->() MATCH ()-[r:LIVES_IN]->() RETURN count(*) AS n" + expect: + rows: + - [0] + + - name: bound_edge_with_a_property_map + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH ()-[r]->() MATCH (x)-[r {w: 3}]->(y) RETURN x.name AS x, y.name AS y" + cypher: "MATCH ()-[r]->() MATCH (x)-[r {w: 3}]->(y) RETURN x.name AS x, y.name AS y" + expect: + rows: + - [Vincent, Alix] + + - name: bound_edge_in_a_named_path + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH ()-[r:KNOWS]->() MATCH p = (x)-[r]->(y) RETURN r.w AS w, length(p) AS len" + cypher: "MATCH ()-[r:KNOWS]->() MATCH p = (x)-[r]->(y) RETURN r.w AS w, length(p) AS len" + expect: + rows: + - [1, 1] + - [2, 1] + - [3, 1] + - [4, 1] + + - name: bound_edge_list_of_a_variable_length_pattern + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH ()-[r:KNOWS*2]->() MATCH (x)-[r:KNOWS*2]->(y) RETURN count(*) AS n" + cypher: "MATCH ()-[r:KNOWS*2]->() MATCH (x)-[r:KNOWS*2]->(y) RETURN count(*) AS n" + expect: + rows: + - [5] + + - name: bound_edge_after_an_ordered_cut + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + cypher: "MATCH ()-[r]->() WITH r ORDER BY r.w LIMIT 2 MATCH (x)-[r]->(y) RETURN r.w AS w, x.name AS x, y.name AS y" + expect: + rows: + - [1, Alix, Gus] + - [2, Gus, Vincent] + + # GQL only: openCypher matches each relationship at most once per pattern, + # so in Cypher this pattern has no match (Grafeo's Cypher does not apply that + # rule yet). + - name: same_edge_twice_in_a_path_is_a_loop + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + query: "MATCH (a)-[r]->(b)-[r]->(c) RETURN a.name AS a, b.name AS b, c.name AS c" + expect: + rows: + - [Mia, Mia, Mia] + + - name: optional_match_through_a_bound_edge + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH ()-[r]->() OPTIONAL MATCH (x)-[r:KNOWS]->(y) RETURN r.w AS w, y.name AS y" + cypher: "MATCH ()-[r]->() OPTIONAL MATCH (x)-[r:KNOWS]->(y) RETURN r.w AS w, y.name AS y" + expect: + rows: + - [1, Gus] + - [2, Vincent] + - [3, Alix] + - [4, Gus] + - [5, null] + - [6, null] + + # --------------------------------------------------------------------------- + # CALL subqueries: imported edges and nodes stay bound + # --------------------------------------------------------------------------- + + - name: subquery_matches_the_edge_it_imports + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH ()-[r]->() CALL { WITH r MATCH (x)-[r]->(y) RETURN y.name AS target } RETURN r.w AS w, target" + cypher: "MATCH ()-[r]->() CALL { WITH r MATCH (x)-[r]->(y) RETURN y.name AS target } RETURN r.w AS w, target" + expect: + rows: + - [1, Gus] + - [2, Vincent] + - [3, Alix] + - [4, Gus] + - [5, Amsterdam] + - [6, Mia] + + - name: subquery_importing_everything_matches_the_edge + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH ()-[r]->() CALL { WITH * MATCH (x)-[r]->(y) RETURN y.name AS target } RETURN r.w AS w, target" + cypher: "MATCH ()-[r]->() CALL { WITH * MATCH (x)-[r]->(y) RETURN y.name AS target } RETURN r.w AS w, target" + expect: + rows: + - [1, Gus] + - [2, Vincent] + - [3, Alix] + - [4, Gus] + - [5, Amsterdam] + - [6, Mia] + + - name: bound_edge_after_a_subquery + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH ()-[r]->() CALL { RETURN 1 AS one } MATCH (x)-[r]->(y) RETURN count(*) AS n" + cypher: "MATCH ()-[r]->() CALL { RETURN 1 AS one } MATCH (x)-[r]->(y) RETURN count(*) AS n" + expect: + rows: + - [6] + + - name: subquery_closes_a_cycle_on_the_nodes_it_imports + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS {w: 1}]->(gus), (gus)-[:KNOWS {w: 2}]->(vincent), (vincent)-[:KNOWS {w: 3}]->(alix), (alix)-[:KNOWS {w: 4}]->(gus), (alix)-[:LIVES_IN {w: 5}]->(ams), (mia)-[:LIKES {w: 6}]->(mia)" + variants: + gql: "MATCH (a)-->(b) CALL { WITH a, b MATCH (b)-->(a) RETURN count(*) AS back } RETURN a.name AS a, b.name AS b, back" + cypher: "MATCH (a)-->(b) CALL { WITH a, b MATCH (b)-->(a) RETURN count(*) AS back } RETURN a.name AS a, b.name AS b, back" + expect: + rows: + - [Alix, Gus, 0] + - [Alix, Gus, 0] + - [Gus, Vincent, 0] + - [Vincent, Alix, 0] + - [Alix, Amsterdam, 0] + - [Mia, Mia, 1] + + # --------------------------------------------------------------------------- + # Values carried through a later MATCH stay the node or edge they are + # --------------------------------------------------------------------------- + + - name: node_property_after_a_later_match + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH (a:Person) MATCH (c:City {name: 'Paris'}) RETURN a.name AS n, a.w AS w" + cypher: "MATCH (a:Person) MATCH (c:City {name: 'Paris'}) RETURN a.name AS n, a.w AS w" + expect: + rows: + - [Alix, 100] + - [Gus, 101] + - [Vincent, 103] + - [Jules, null] + - [Mia, null] + + - name: node_property_after_with_and_a_later_match + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH (a:Person) WITH a MATCH (c:City {name: 'Paris'}) RETURN a.name AS n, a.w AS w" + cypher: "MATCH (a:Person) WITH a MATCH (c:City {name: 'Paris'}) RETURN a.name AS n, a.w AS w" + expect: + rows: + - [Alix, 100] + - [Gus, 101] + - [Vincent, 103] + - [Jules, null] + - [Mia, null] + + - name: edge_property_after_a_later_match + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH ()-[r:KNOWS]->() MATCH (c:City {name: 'Paris'}) RETURN r.since AS since, r.w AS w" + cypher: "MATCH ()-[r:KNOWS]->() MATCH (c:City {name: 'Paris'}) RETURN r.since AS since, r.w AS w" + expect: + rows: + - [2010, 1] + - [2012, 2] + - [2015, 3] + - [2020, 4] + - [2018, 5] + + - name: edge_filter_after_a_later_match + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH ()-[r:KNOWS]->() MATCH (c:City {name: 'Paris'}) WHERE r.w > 2 RETURN r.w AS w" + cypher: "MATCH ()-[r:KNOWS]->() MATCH (c:City {name: 'Paris'}) WHERE r.w > 2 RETURN r.w AS w" + expect: + rows: + - [3] + - [4] + - [5] + + - name: edge_property_after_a_later_path_match + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH ()-[r:KNOWS]->() MATCH p = (x:Person {name: 'Mia'})-[]->(y) RETURN r.w AS w, y.name AS y" + cypher: "MATCH ()-[r:KNOWS]->() MATCH p = (x:Person {name: 'Mia'})-[]->(y) RETURN r.w AS w, y.name AS y" + expect: + rows: + - [1, Paris] + - [2, Paris] + - [3, Paris] + - [4, Paris] + - [5, Paris] + + # --------------------------------------------------------------------------- + # A later pattern after a shortest path comes back to the nodes it bound + # --------------------------------------------------------------------------- + + # Nobody Alix knows knows Gus, so no two KNOWS edges lead from Alix back to + # the path's Gus (the last hop used to bind a new `a`: Gus and Jules). + - name: a_later_pattern_back_to_a_shortest_path_end + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH p = ANY SHORTEST (a:Person {name: 'Gus'})-[:KNOWS]->+(b:Person {name: 'Alix'}) MATCH (b)-[:KNOWS]->(c)-[:KNOWS]->(a) RETURN c.name AS c" + cypher: "MATCH p = shortestPath((a:Person {name: 'Gus'})-[:KNOWS*]->(b:Person {name: 'Alix'})) MATCH (b)-[:KNOWS]->(c)-[:KNOWS]->(a) RETURN c.name AS c" + expect: + empty: true + + # One edge back to the path's start does exist: Alix knows Gus. + - name: a_later_edge_back_to_a_shortest_path_start + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH p = ANY SHORTEST (a:Person {name: 'Gus'})-[:KNOWS]->+(b:Person {name: 'Alix'}) MATCH (b)-[r:KNOWS]->(a) RETURN r.w AS w" + cypher: "MATCH p = shortestPath((a:Person {name: 'Gus'})-[:KNOWS*]->(b:Person {name: 'Alix'})) MATCH (b)-[r:KNOWS]->(a) RETURN r.w AS w" + expect: + rows: + - [1] diff --git a/tests/spec/rosetta/call_subqueries.gtest b/tests/spec/rosetta/call_subqueries.gtest new file mode 100644 index 000000000..28496e2d1 --- /dev/null +++ b/tests/spec/rosetta/call_subqueries.gtest @@ -0,0 +1,694 @@ +# Rosetta: CALL subqueries +# +# A Cypher CALL subquery sees the outer variables its importing WITH names. +# As in openCypher, that WITH may only list them: a WHERE, DISTINCT, an alias or an +# expression in it is an error (a second WITH can filter or rename), and so is +# a name the outer query does not have. A first WITH that names no variable is +# an ordinary WITH. +# +# Dataset `entity_kinds` (tests/spec/datasets/entity_kinds.setup): Alix (30), +# Gus (25), Vincent (40), Jules (35) and Mia (28); KNOWS Alix->Gus, +# Gus->Vincent, Vincent->Alix, Jules->Mia and Alix->Jules. + +meta: + model: lpg + section: rosetta + title: CALL subqueries + dataset: entity_kinds + requires: [cypher] + +tests: + + # --------------------------------------------------------------------------- + # The importing WITH (Cypher) + # --------------------------------------------------------------------------- + + - name: importing_with_where_is_rejected + variants: + cypher: "MATCH (a:Person) CALL { WITH a WHERE a.age > 30 RETURN a.name AS m } RETURN m" + expect: + error: "Importing WITH should consist only of simple references to outside variables. WHERE is not allowed" + + - name: importing_with_star_and_where_is_rejected + variants: + cypher: "MATCH (a:Person) CALL { WITH * WHERE a.age > 30 RETURN a.name AS m } RETURN m" + expect: + error: "Importing WITH should consist only of simple references to outside variables. WHERE is not allowed" + + - name: importing_with_an_alias_is_rejected + variants: + cypher: "MATCH (a:Person) CALL { WITH a AS b RETURN b.name AS m } RETURN m" + expect: + error: "Importing WITH should consist only of simple references to outside variables. Aliasing or expressions are not supported" + + - name: importing_with_an_expression_is_rejected + variants: + cypher: "MATCH (a:Person) CALL { WITH a, a.age + 1 AS next RETURN next } RETURN next" + expect: + error: "Importing WITH should consist only of simple references to outside variables. Aliasing or expressions are not supported" + + - name: importing_with_distinct_is_rejected + variants: + cypher: "MATCH (a:Person) CALL { WITH DISTINCT a RETURN a.name AS m } RETURN m" + expect: + error: "Importing WITH should consist only of simple references to outside variables. DISTINCT is not allowed" + + - name: importing_a_name_the_outer_query_lacks_is_rejected + variants: + cypher: "MATCH (a:Person) CALL { WITH b RETURN 1 AS one } RETURN one" + expect: + error: "Undefined variable 'b'" + + - name: a_second_with_filters_the_imported_rows + variants: + cypher: "MATCH (a:Person) CALL { WITH a WITH a WHERE a.age > 30 RETURN a.name AS m } RETURN m" + expect: + rows: + - [Vincent] + - [Jules] + + - name: a_second_with_renames_an_imported_variable + variants: + cypher: "MATCH (a:Person) WHERE a.age > 30 CALL { WITH a WITH a AS b RETURN b.name AS m } RETURN a.name AS a, m" + expect: + rows: + - [Vincent, Vincent] + - [Jules, Jules] + + # The variable a list comprehension binds is its own, not an outer one: the + # WITH names no outer variable and is an ordinary WITH. + - name: a_first_with_of_a_comprehension_imports_nothing + variants: + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH [x IN [1, 2] | x * 10] AS xs RETURN xs } RETURN a.name AS a, xs" + expect: + rows: + - [Alix, "[10, 20]"] + + - name: a_first_with_without_variables_imports_nothing + variants: + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH 1 AS x RETURN x } RETURN a.name AS a, x" + expect: + rows: + - [Alix, 1] + + # --------------------------------------------------------------------------- + # What a CALL subquery returns: its nodes and edges stay nodes and edges, so + # a later MATCH, id() and equality see the same entity + # --------------------------------------------------------------------------- + + - name: a_returned_node_starts_a_later_match + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b } MATCH (b)-[:KNOWS]->(y) RETURN b.name AS b, y.name AS y" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b } MATCH (b)-[:KNOWS]->(y) RETURN b.name AS b, y.name AS y" + expect: + rows: + - [Gus, Vincent] + - [Jules, Mia] + + - name: a_returned_node_ends_a_later_match + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b } MATCH (x:Person)-[:KNOWS]->(b) RETURN b.name AS b, x.name AS x" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b } MATCH (x:Person)-[:KNOWS]->(b) RETURN b.name AS b, x.name AS x" + expect: + rows: + - [Gus, Alix] + - [Jules, Alix] + + - name: a_returned_node_equals_a_matched_node + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b } MATCH (y:Person) WHERE y = b RETURN y.name AS y" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b } MATCH (y:Person) WHERE y = b RETURN y.name AS y" + expect: + rows: + - [Gus] + - [Jules] + + - name: a_returned_node_keeps_its_id + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b, id(b) AS inner } RETURN b.name AS b, id(b) = inner AS same" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b, id(b) AS inner } RETURN b.name AS b, id(b) = inner AS same" + expect: + rows: + - [Gus, true] + - [Jules, true] + + - name: a_returned_node_through_a_with + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b } WITH b MATCH (b)-[:KNOWS]->(y) RETURN b.name AS b, y.name AS y" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b } WITH b MATCH (b)-[:KNOWS]->(y) RETURN b.name AS b, y.name AS y" + expect: + rows: + - [Gus, Vincent] + - [Jules, Mia] + + - name: a_distinct_returned_node_starts_a_later_match + variants: + gql: "MATCH (a:Person) CALL { WITH a MATCH (a)-[:LIVES_IN]->(c) RETURN DISTINCT c } MATCH (p:Person)-[:LIVES_IN]->(c) RETURN a.name AS a, p.name AS p" + cypher: "MATCH (a:Person) CALL { WITH a MATCH (a)-[:LIVES_IN]->(c) RETURN DISTINCT c } MATCH (p:Person)-[:LIVES_IN]->(c) RETURN a.name AS a, p.name AS p" + expect: + rows: + - [Alix, Alix] + - [Gus, Gus] + - [Mia, Mia] + + - name: a_returned_edge_in_a_later_match + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[r:KNOWS]->() RETURN r } MATCH (x)-[r]->(z) RETURN x.name AS x, z.name AS z, r.w AS w" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[r:KNOWS]->() RETURN r } MATCH (x)-[r]->(z) RETURN x.name AS x, z.name AS z, r.w AS w" + expect: + rows: + - [Alix, Gus, 1] + - [Alix, Jules, 5] + + - name: a_returned_edge_is_an_edge + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[r:KNOWS]->() RETURN r } RETURN type(r) AS t, r.w AS w" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[r:KNOWS]->() RETURN r } RETURN type(r) AS t, r.w AS w" + expect: + rows: + - [KNOWS, 1] + - [KNOWS, 5] + + - name: collected_returned_nodes_start_a_later_match + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN collect(b) AS bs } UNWIND bs AS b MATCH (b)-[:KNOWS]->(y) RETURN b.name AS b, y.name AS y" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN collect(b) AS bs } UNWIND bs AS b MATCH (b)-[:KNOWS]->(y) RETURN b.name AS b, y.name AS y" + expect: + rows: + - [Gus, Vincent] + - [Jules, Mia] + + - name: an_uncorrelated_call_returns_nodes + variants: + gql: "CALL { MATCH (c:City) RETURN c } MATCH (p:Person)-[:LIVES_IN]->(c) RETURN p.name AS p, c.name AS c" + cypher: "CALL { MATCH (c:City) RETURN c } MATCH (p:Person)-[:LIVES_IN]->(c) RETURN p.name AS p, c.name AS c" + expect: + rows: + - [Alix, Amsterdam] + - [Gus, Berlin] + - [Mia, Paris] + + - name: an_uncorrelated_call_after_a_match_returns_nodes + variants: + gql: "MATCH (p:Person {name: 'Gus'}) CALL { MATCH (c:City) RETURN c } MATCH (p)-[:LIVES_IN]->(c) RETURN p.name AS p, c.name AS c" + cypher: "MATCH (p:Person {name: 'Gus'}) CALL { MATCH (c:City) RETURN c } MATCH (p)-[:LIVES_IN]->(c) RETURN p.name AS p, c.name AS c" + expect: + rows: + - [Gus, Berlin] + + # --------------------------------------------------------------------------- + # GQL: a CALL subquery sees every variable of the outer row. In GQL and + # Cypher, CALL (a, b) limits that to a and b and CALL () to none (Cypher's + # CALL (*) imports all). A WITH in it is an ordinary WITH. + # --------------------------------------------------------------------------- + + - name: gql_call_sees_an_outer_node + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN bn" + expect: + rows: + - [Gus] + - [Jules] + + - name: gql_call_sees_an_outer_edge + variants: + gql: "MATCH ()-[r:KNOWS]->() CALL { MATCH (x)-[r]->(y) RETURN x.name AS xn } RETURN count(*) AS c" + expect: + rows: + - [5] + + - name: gql_call_nested_sees_the_outermost_row + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { CALL { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN bn } RETURN bn" + expect: + rows: + - [Gus] + - [Jules] + + - name: gql_call_with_where_filters + variants: + gql: "MATCH (a:Person) CALL { WITH a WHERE a.age > 30 RETURN a.name AS m } RETURN m" + expect: + rows: + - [Vincent] + - [Jules] + + - name: gql_call_with_renames + variants: + gql: "MATCH (a:Person WHERE a.age > 30) CALL { WITH a AS b RETURN b.name AS m } RETURN a.name AS a, m" + expect: + rows: + - [Vincent, Vincent] + - [Jules, Jules] + + - name: gql_call_with_after_a_match + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { MATCH (a)-[:KNOWS]->(b) WITH b RETURN b.name AS bn } RETURN bn" + expect: + rows: + - [Gus] + - [Jules] + + - name: call_scope_clause + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN bn" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN bn" + expect: + rows: + - [Gus] + - [Jules] + + # c is not imported, so the subquery's c is any city. + - name: call_scope_clause_limits_what_the_subquery_sees + variants: + gql: "MATCH (a:Person {name: 'Alix'}), (c:City {name: 'Paris'}) CALL (a) { MATCH (c:City) RETURN count(c) AS n } RETURN n" + cypher: "MATCH (a:Person {name: 'Alix'}), (c:City {name: 'Paris'}) CALL (a) { MATCH (c:City) RETURN count(c) AS n } RETURN n" + expect: + rows: + - [3] + + # Nothing is imported: the subquery's a is any person who knows someone. + - name: call_empty_scope_clause + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL () { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN count(*) AS c" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL () { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN count(*) AS c" + expect: + rows: + - [5] + + - name: gql_optional_call_scope_clause + variants: + gql: "MATCH (a:Person) OPTIONAL CALL (a) { MATCH (a)-[:LIVES_IN]->(c) RETURN c.name AS c } RETURN a.name AS a, c" + expect: + rows: + - [Alix, Amsterdam] + - [Gus, Berlin] + - [Vincent, null] + - [Jules, null] + - [Mia, Paris] + + - name: call_scope_clause_naming_a_missing_variable + variants: + gql: "MATCH (a:Person) CALL (b) { RETURN 1 AS one } RETURN one" + cypher: "MATCH (a:Person) CALL (b) { RETURN 1 AS one } RETURN one" + expect: + error: "Undefined variable 'b'" + + - name: cypher_call_scope_clause_imports_all + variants: + cypher: "MATCH (a:Person {name: 'Alix'}) CALL (*) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS bn } RETURN bn" + expect: + rows: + - [Gus] + - [Jules] + + # After a scope clause a WITH is an ordinary WITH: its WHERE filters. + - name: cypher_call_scope_clause_then_an_ordinary_with + variants: + cypher: "MATCH (a:Person) CALL (a) { WITH a WHERE a.age > 30 RETURN a.name AS m } RETURN m" + expect: + rows: + - [Vincent] + - [Jules] + + - name: cypher_call_with_an_empty_scope_clause_comes_first + variants: + cypher: "CALL () { MATCH (c:City) RETURN c } MATCH (p:Person)-[:LIVES_IN]->(c) RETURN p.name AS p, c.name AS c" + expect: + rows: + - [Alix, Amsterdam] + - [Gus, Berlin] + - [Mia, Paris] + + # --------------------------------------------------------------------------- + # RETURN * in a CALL subquery returns the variables the subquery binds + # itself; the outer row's variables stay as they are + # --------------------------------------------------------------------------- + + - name: return_star_returns_the_variables_of_the_subquery + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN * } RETURN a.name AS a, b.name AS b" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN * } RETURN a.name AS a, b.name AS b" + expect: + rows: + - [Alix, Gus] + - [Alix, Jules] + + - name: return_star_with_an_implicit_scope + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { MATCH (a)-[:KNOWS]->(b) RETURN * } RETURN a.name AS a, b.name AS b" + expect: + rows: + - [Alix, Gus] + - [Alix, Jules] + + - name: return_star_with_a_scope_clause + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL (a) { MATCH (a)-[r:KNOWS]->(b) RETURN * } RETURN type(r) AS t, b.name AS b" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL (a) { MATCH (a)-[r:KNOWS]->(b) RETURN * } RETURN type(r) AS t, b.name AS b" + expect: + rows: + - [KNOWS, Gus] + - [KNOWS, Jules] + + # The later MATCH ends at the b the subquery returned, not at any node. + - name: a_node_return_star_returns_starts_a_later_match + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN * } MATCH (x:Person)-[:KNOWS]->(b) RETURN b.name AS b, x.name AS x" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN * } MATCH (x:Person)-[:KNOWS]->(b) RETURN b.name AS b, x.name AS x" + expect: + rows: + - [Gus, Alix] + - [Jules, Alix] + + - name: return_star_from_a_call_that_comes_first + variants: + gql: "CALL { MATCH (c:City) RETURN * } RETURN c.name AS c" + cypher: "CALL { MATCH (c:City) RETURN * } RETURN c.name AS c" + expect: + rows: + - [Amsterdam] + - [Berlin] + - [Paris] + + - name: return_star_from_a_nested_call + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { CALL { MATCH (a)-[:KNOWS]->(b) RETURN * } RETURN b.name AS bn } RETURN bn" + expect: + rows: + - [Gus] + - [Jules] + + # A subquery that binds nothing new returns no column: each outer row once. + - name: return_star_of_nothing_new + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a RETURN * } RETURN a.name AS a" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a RETURN * } RETURN a.name AS a" + expect: + rows: + - [Alix] + + # After a procedure call the outer row holds what it yields (label), so + # RETURN * returns the subquery's own p: two labels times five people. + - name: return_star_after_a_procedure_call + variants: + cypher: "CALL db.labels() YIELD label CALL { WITH label MATCH (p:Person) RETURN * } RETURN count(*) AS c" + expect: + rows: + - [10] + + # The RETURN under an ORDER BY and LIMIT also passes on the node itself. + - name: a_returned_node_after_order_by_and_limit + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { MATCH (a)-[:KNOWS]->(b) RETURN b ORDER BY b.name DESC LIMIT 1 } MATCH (b)-[:KNOWS]->(y) RETURN b.name AS b, y.name AS y" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b ORDER BY b.name DESC LIMIT 1 } MATCH (b)-[:KNOWS]->(y) RETURN b.name AS b, y.name AS y" + expect: + rows: + - [Jules, Mia] + + # The sort key reads a variable the RETURN leaves out (b.age). + - name: a_sort_key_the_return_leaves_out + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS n ORDER BY b.age LIMIT 1 } RETURN n" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) RETURN b.name AS n ORDER BY b.age LIMIT 1 } RETURN n" + expect: + rows: + - [Gus] + + # The scope clause names a only: the outer b is not visible in the subquery. + - name: call_scope_clause_hides_other_outer_variables + variants: + gql: "MATCH (a:Person {name: 'Alix'}), (b:City {name: 'Paris'}) CALL (a) { RETURN b.name AS bn } RETURN bn" + cypher: "MATCH (a:Person {name: 'Alix'}), (b:City {name: 'Paris'}) CALL (a) { RETURN b.name AS bn } RETURN bn" + expect: + error: "Undefined variable 'b'" + + # An OPTIONAL MATCH that starts a scoped subquery keeps each outer row's own + # match (or none). + - name: an_optional_match_first_in_a_scoped_call + variants: + gql: "MATCH (a:Person) CALL (a) { OPTIONAL MATCH (a)-[:LIVES_IN]->(c) RETURN c.name AS c } RETURN a.name AS a, c" + expect: + rows: + - [Alix, Amsterdam] + - [Gus, Berlin] + - [Vincent, null] + - [Jules, null] + - [Mia, Paris] + + # --------------------------------------------------------------------------- + # Scope rules (openCypher): a subquery returns new names only, and the variables + # its scope clause imports stay visible after a WITH that leaves them out + # --------------------------------------------------------------------------- + + - name: returning_an_outer_name_is_rejected + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL { RETURN 1 AS a } RETURN a" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { RETURN 1 AS a } RETURN a" + expect: + error: "already declared outside the CALL subquery" + + - name: returning_an_imported_variable_is_rejected + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL (a) { RETURN a } RETURN a.name AS n" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL (a) { RETURN a } RETURN a.name AS n" + expect: + error: "already declared outside the CALL subquery" + + - name: an_imported_variable_stays_visible_after_a_with + skip: "#545: a WITH in the subquery still ends the scope of an import (Undefined variable 'a')" + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL (a) { WITH 1 AS x RETURN a.name AS n, x } RETURN n, x" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL (a) { WITH 1 AS x RETURN a.name AS n, x } RETURN n, x" + expect: + rows: + - [Alix, 1] + + # The count over no match is still one row, with the imported a. + - name: an_imported_variable_stays_visible_after_an_aggregating_with + skip: "#545: a WITH in the subquery still ends the scope of an import (Undefined variable 'a')" + variants: + gql: "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) WITH count(b) AS k RETURN a.name AS n, k } RETURN n, k" + cypher: "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) WITH count(b) AS k RETURN a.name AS n, k } RETURN n, k" + expect: + rows: + - [Alix, 2] + - [Gus, 1] + - [Vincent, 1] + - [Jules, 1] + - [Mia, 0] + + # A variable of the subquery's own is out of scope after a WITH that leaves + # it out, as anywhere else. + - name: a_with_ends_the_scope_of_a_subquery_variable + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL (a) { MATCH (a)-[:KNOWS]->(b) WITH count(b) AS k RETURN b.name AS n, k } RETURN n, k" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL (a) { MATCH (a)-[:KNOWS]->(b) WITH count(b) AS k RETURN b.name AS n, k } RETURN n, k" + expect: + error: "Undefined variable 'b'" + + # Carried through the WITH, the import stays readable. + - name: an_imported_variable_carried_through_a_with + variants: + gql: "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) WITH a, count(b) AS k RETURN a.name AS n, k } RETURN n, k" + cypher: "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) WITH a, count(b) AS k RETURN a.name AS n, k } RETURN n, k" + expect: + rows: + - [Alix, 2] + - [Gus, 1] + - [Vincent, 1] + - [Jules, 1] + + - name: a_with_ends_the_scope_outside_a_subquery + variants: + gql: "MATCH (a:Person {name: 'Alix'}) WITH 1 AS x RETURN a.name AS n" + cypher: "MATCH (a:Person {name: 'Alix'}) WITH 1 AS x RETURN a.name AS n" + expect: + error: "Undefined variable 'a'" + + # As in openCypher, an expression a WITH passes on needs a name: `n.name` + # does not keep `n`, so nothing later could read it. + - name: an_unaliased_expression_in_with_is_rejected + variants: + gql: "MATCH (n:Person {name: 'Alix'}) WITH n.name RETURN n.name" + cypher: "MATCH (n:Person {name: 'Alix'}) WITH n.name RETURN n.name" + expect: + error: "Expression in WITH must be aliased (use AS)" + + - name: an_aliased_expression_in_with + variants: + gql: "MATCH (n:Person {name: 'Alix'}) WITH n.name AS name, n RETURN name, n.age AS age" + cypher: "MATCH (n:Person {name: 'Alix'}) WITH n.name AS name, n RETURN name, n.age AS age" + expect: + rows: + - [Alix, 30] + + # --------------------------------------------------------------------------- + # ORDER BY, SKIP, LIMIT and UNION in the subquery body + # --------------------------------------------------------------------------- + + # The oldest person each one knows: a top 1 per outer row. Mia knows nobody. + - name: order_by_and_limit_per_outer_row + variants: + gql: "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS f ORDER BY b.age DESC LIMIT 1 } RETURN a.name AS a, f" + cypher: "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS f ORDER BY b.age DESC LIMIT 1 } RETURN a.name AS a, f" + expect: + rows: + - [Alix, Jules] + - [Gus, Vincent] + - [Vincent, Alix] + - [Jules, Mia] + + # Only Alix knows two people (Gus and Jules); skipping the first leaves Jules. + - name: skip_per_outer_row + variants: + gql: "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS f ORDER BY f SKIP 1 } RETURN a.name AS a, f" + cypher: "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS f ORDER BY f SKIP 1 } RETURN a.name AS a, f" + expect: + rows: + - [Alix, Jules] + + # Whom Alix knows and who knows Alix. + - name: union_in_the_body + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS x UNION MATCH (b)-[:KNOWS]->(a) RETURN b.name AS x } RETURN x" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS x UNION MATCH (b)-[:KNOWS]->(a) RETURN b.name AS x } RETURN x" + expect: + rows: + - [Gus] + - [Jules] + - [Vincent] + + # A CALL nested in the first part: the second part still starts from the + # outer row. + - name: union_after_a_nested_call + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL (a) { CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS y } RETURN y AS x UNION MATCH (b)-[:KNOWS]->(a) RETURN b.name AS x } RETURN x" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL (a) { CALL (a) { MATCH (a)-[:KNOWS]->(b) RETURN b.name AS y } RETURN y AS x UNION MATCH (b)-[:KNOWS]->(a) RETURN b.name AS x } RETURN x" + expect: + rows: + - [Gus] + - [Jules] + - [Vincent] + + # Each part sees only what its own importing WITH imports: the first part + # imports a, not b. + - name: a_union_part_sees_only_its_own_imports + variants: + cypher: "MATCH (a:Person {name: 'Alix'}), (b:City {name: 'Paris'}) CALL { WITH a RETURN b.name AS x UNION WITH b RETURN b.name AS x } RETURN x" + expect: + error: "Undefined variable 'b'" + + - name: union_parts_with_imports_of_their_own + variants: + cypher: "MATCH (a:Person {name: 'Alix'}), (b:City {name: 'Paris'}) CALL { WITH a RETURN a.name AS x UNION WITH b RETURN b.name AS x } RETURN x" + expect: + rows: + - [Alix] + - [Paris] + + # UNION ALL keeps both rows; UNION keeps one. + - name: union_all_in_the_body + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL (a) { RETURN a.name AS x UNION ALL RETURN a.name AS x } RETURN x" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL (a) { RETURN a.name AS x UNION ALL RETURN a.name AS x } RETURN x" + expect: + rows: + - [Alix] + - [Alix] + + - name: union_in_the_body_removes_duplicates + variants: + gql: "MATCH (a:Person {name: 'Alix'}) CALL (a) { RETURN a.name AS x UNION RETURN a.name AS x } RETURN x" + cypher: "MATCH (a:Person {name: 'Alix'}) CALL (a) { RETURN a.name AS x UNION RETURN a.name AS x } RETURN x" + expect: + rows: + - [Alix] + + # An importing WITH only lists outer variables: ORDER BY, SKIP and LIMIT + # after it are errors too (a second WITH can order and cut). + - name: importing_with_order_by_is_rejected + variants: + cypher: "MATCH (a:Person) CALL { WITH a ORDER BY a.age MATCH (a)-[:KNOWS]->(b) RETURN b.name AS f } RETURN f" + expect: + error: "Importing WITH should consist only of simple references to outside variables. ORDER BY is not allowed" + + - name: importing_with_skip_is_rejected + variants: + cypher: "MATCH (a:Person) CALL { WITH a SKIP 1 MATCH (a)-[:KNOWS]->(b) RETURN b.name AS f } RETURN f" + expect: + error: "Importing WITH should consist only of simple references to outside variables. SKIP is not allowed" + + - name: importing_with_limit_is_rejected + variants: + cypher: "MATCH (a:Person) CALL { WITH a LIMIT 1 MATCH (a)-[:KNOWS]->(b) RETURN b.name AS f } RETURN f" + expect: + error: "Importing WITH should consist only of simple references to outside variables. LIMIT is not allowed" + + - name: a_second_with_orders_and_cuts + variants: + cypher: "MATCH (a:Person {name: 'Alix'}) CALL { WITH a MATCH (a)-[:KNOWS]->(b) WITH b ORDER BY b.age LIMIT 1 RETURN b.name AS f } RETURN f" + expect: + rows: + - [Gus] + + # --------------------------------------------------------------------------- + # A subquery that imports nothing sees no outer variable + # --------------------------------------------------------------------------- + + # `CALL ()` imports nothing: `a` is not a variable inside it (it used to + # read null, so every count was 0). + - name: an_empty_scope_clause_sees_no_outer_variable + variants: + gql: "MATCH (a:Person) CALL () { MATCH (b:Person) WHERE b.age > a.age RETURN count(b) AS c } RETURN a.name AS a, c" + cypher: "MATCH (a:Person) CALL () { MATCH (b:Person) WHERE b.age > a.age RETURN count(b) AS c } RETURN a.name AS a, c" + expect: + error: "Undefined variable 'a'" + + # A Cypher CALL without an importing WITH imports nothing either. + - name: a_call_without_an_importing_with_sees_no_outer_variable + variants: + cypher: "MATCH (a:Person) CALL { MATCH (b:Person) WHERE b.age > a.age RETURN count(b) AS c } RETURN a.name AS a, c" + expect: + error: "Undefined variable 'a'" + + # A name the outer query has is a new variable inside such a subquery: here + # a node pattern that matches every node (5 people and 3 cities). + - name: an_outer_name_is_a_new_variable_inside + variants: + gql: "UNWIND [1] AS a CALL () { MATCH (a) RETURN a AS x } RETURN count(x) AS n" + cypher: "UNWIND [1] AS a CALL () { MATCH (a) RETURN a AS x } RETURN count(x) AS n" + expect: + rows: + - [8] + + # --------------------------------------------------------------------------- + # Writes in the subquery body: DELETE, REMOVE and MERGE (CREATE and SET + # already ran); the later MATCH reads what the subquery wrote + # --------------------------------------------------------------------------- + + # Alix and Gus live outside Paris: two edges go, Mia's to Paris stays. + - name: delete_in_the_body + variants: + gql: "MATCH (p:Person) CALL (p) { MATCH (p)-[r:LIVES_IN]->(c:City) WHERE c.name <> 'Paris' DELETE r RETURN count(*) AS deleted } WITH sum(deleted) AS total MATCH ()-[r:LIVES_IN]->() RETURN total, count(r) AS remaining" + cypher: "MATCH (p:Person) CALL { WITH p MATCH (p)-[r:LIVES_IN]->(c:City) WHERE c.name <> 'Paris' DELETE r RETURN count(*) AS deleted } WITH sum(deleted) AS total MATCH ()-[r:LIVES_IN]->() RETURN total, count(r) AS remaining" + expect: + rows: + - [2, 1] + + # Alix, Gus and Vincent have a w; afterwards nobody has. + - name: remove_in_the_body + variants: + gql: "MATCH (p:Person) CALL (p) { REMOVE p.w RETURN count(*) AS done } WITH sum(done) AS total MATCH (q:Person) RETURN total, count(q.w) AS left" + cypher: "MATCH (p:Person) CALL { WITH p REMOVE p.w RETURN count(*) AS done } WITH sum(done) AS total MATCH (q:Person) RETURN total, count(q.w) AS left" + expect: + rows: + - [5, 0] + + # Alix already lives in Amsterdam: the MERGE matches that edge, adds none. + - name: merge_in_the_body + variants: + cypher: "MATCH (p:Person {name: 'Alix'}) CALL { WITH p MERGE (p)-[:LIVES_IN]->(c:City {name: 'Amsterdam'}) RETURN c.name AS city } WITH city MATCH (:Person {name: 'Alix'})-[r:LIVES_IN]->() RETURN city, count(r) AS edges" + expect: + rows: + - [Amsterdam, 1] diff --git a/tests/spec/rosetta/clauses_after_with.gtest b/tests/spec/rosetta/clauses_after_with.gtest new file mode 100644 index 000000000..5679e3ab0 --- /dev/null +++ b/tests/spec/rosetta/clauses_after_with.gtest @@ -0,0 +1,101 @@ +# Rosetta: clauses after WITH +# +# A WITH ends one part of a statement and starts the next: a MATCH after it +# reads from the rows WITH passes on, sees only the variables WITH keeps (a +# dropped name is free to bind again), and sees the writes made before it in +# the same statement. The WHERE of the first part filters before the WITH. +# +# Setup (GQL): P n:1 -[K]-> P n:2 -[K]-> P n:3. + +meta: + model: lpg + section: rosetta + title: Clauses after WITH + dataset: empty + requires: [cypher] + +tests: + + - name: match_after_with + setup: + - "INSERT (:P {n: 1})-[:K]->(:P {n: 2})-[:K]->(:P {n: 3})" + variants: + gql: "MATCH (a:P)-[:K]->(b) WITH b MATCH (b)-[:K]->(c) RETURN b.n, c.n" + cypher: "MATCH (a:P)-[:K]->(b) WITH b MATCH (b)-[:K]->(c) RETURN b.n, c.n" + expect: + rows: + - [2, 3] + + - name: a_dropped_variable_binds_again + setup: + - "INSERT (:P {n: 1})-[:K]->(:P {n: 2})-[:K]->(:P {n: 3})" + variants: + gql: "MATCH (a:P)-[:K]->(b) WITH b MATCH (b)-[:K]->(a) RETURN b.n, a.n" + cypher: "MATCH (a:P)-[:K]->(b) WITH b MATCH (b)-[:K]->(a) RETURN b.n, a.n" + expect: + rows: + - [2, 3] + + - name: where_filters_before_with + setup: + - "INSERT (:P {n: 1})-[:K]->(:P {n: 2})-[:K]->(:P {n: 3})" + variants: + gql: "MATCH (a:P) WHERE a.n = 1 WITH a MATCH (a)-[:K]->(b) RETURN b.n" + cypher: "MATCH (a:P) WHERE a.n = 1 WITH a MATCH (a)-[:K]->(b) RETURN b.n" + expect: + rows: + - [2] + + - name: match_after_an_aggregating_with + setup: + - "INSERT (:P {n: 1})-[:K]->(:P {n: 2})-[:K]->(:P {n: 3})" + variants: + gql: "MATCH (a:P) WITH count(a) AS c MATCH (b:P) RETURN c, b.n ORDER BY b.n" + cypher: "MATCH (a:P) WITH count(a) AS c MATCH (b:P) RETURN c, b.n ORDER BY b.n" + expect: + ordered: true + rows: + - [3, 1] + - [3, 2] + - [3, 3] + + # #480: a MATCH after SET ... WITH sees the property the SET wrote. + - name: match_after_set_with + setup: + - "INSERT (:P {n: 1})-[:K]->(:P {n: 2})-[:K]->(:P {n: 3})" + variants: + gql: "MATCH (a:P {n: 1}) SET a.w = 7 WITH a MATCH (b:P {n: 1}) RETURN b.w" + cypher: "MATCH (a:P {n: 1}) SET a.w = 7 WITH a MATCH (b:P {n: 1}) RETURN b.w" + expect: + rows: + - [7] + + # #479: a MATCH after an insert in the same statement finds the new node. + - name: match_after_insert_with + variants: + gql: "INSERT (:N {id: 'c'}) WITH 1 AS x MATCH (n:N) RETURN n.id" + cypher: "CREATE (:N {id: 'c'}) WITH 1 AS x MATCH (n:N) RETURN n.id" + expect: + rows: + - [c] + + # A REMOVE before WITH runs before it, also when WITH drops its variable. + - name: remove_before_a_with_that_drops_its_variable + setup: + - "INSERT (:P {n: 1, w: 7})" + variants: + gql: "MATCH (a:P) REMOVE a.w WITH a.n AS n MATCH (b:P) RETURN n, b.w" + cypher: "MATCH (a:P) REMOVE a.w WITH a.n AS n MATCH (b:P) RETURN n, b.w" + expect: + rows: + - [1, null] + + - name: match_after_remove_with + setup: + - "INSERT (:P {n: 1, w: 7})" + variants: + gql: "MATCH (a:P {n: 1}) REMOVE a.w WITH a MATCH (b:P {n: 1}) RETURN b.w" + cypher: "MATCH (a:P {n: 1}) REMOVE a.w WITH a MATCH (b:P {n: 1}) RETURN b.w" + expect: + rows: + - [null] diff --git a/tests/spec/rosetta/correlated_subqueries.gtest b/tests/spec/rosetta/correlated_subqueries.gtest new file mode 100644 index 000000000..b99b2400d --- /dev/null +++ b/tests/spec/rosetta/correlated_subqueries.gtest @@ -0,0 +1,451 @@ +# Rosetta: EXISTS and COUNT subqueries tied to the outer row +# +# A subquery may refer to the row it is evaluated for through a value (a +# property map like {id: s.id}, an inner WHERE, a variable from UNWIND), not +# only through a shared node or edge. Each row gets its own answer, in RETURN, +# in WITH and in WHERE, also when identical rows come in. +# +# Most setups: (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'}). + +meta: + model: lpg + section: rosetta + title: EXISTS and COUNT subqueries tied to the outer row + dataset: empty + requires: [cypher] + +tests: + + # --------------------------------------------------------------------------- + # In RETURN and WITH + # --------------------------------------------------------------------------- + + - name: return_exists_tied_by_a_property_map + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) RETURN s.id AS id, EXISTS { MATCH (x:Other {id: s.id}) } AS e" + cypher: "MATCH (s:File) RETURN s.id AS id, EXISTS { MATCH (x:Other {id: s.id}) } AS e" + expect: + rows: + - [f1, false] + - [f2, true] + + - name: return_count_tied_by_a_property_map + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) RETURN s.id AS id, COUNT { MATCH (x {id: s.id}) } AS c" + cypher: "MATCH (s:File) RETURN s.id AS id, COUNT { MATCH (x {id: s.id}) } AS c" + expect: + rows: + - [f1, 1] + - [f2, 2] + + - name: return_count_tied_by_an_inner_where + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) RETURN s.id AS id, COUNT { MATCH (x:Other) WHERE x.id = s.id } AS c" + cypher: "MATCH (s:File) RETURN s.id AS id, COUNT { MATCH (x:Other) WHERE x.id = s.id } AS c" + expect: + rows: + - [f1, 0] + - [f2, 1] + + - name: return_count_tied_to_an_unwound_value + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "UNWIND ['f1', 'f2'] AS k RETURN k, COUNT { MATCH (x {id: k}) } AS c" + cypher: "UNWIND ['f1', 'f2'] AS k RETURN k, COUNT { MATCH (x {id: k}) } AS c" + expect: + rows: + - [f1, 1] + - [f2, 2] + + - name: with_count_tied_by_a_property_map + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) WITH s, COUNT { MATCH (x {id: s.id}) } AS c RETURN s.id AS id, c" + cypher: "MATCH (s:File) WITH s, COUNT { MATCH (x {id: s.id}) } AS c RETURN s.id AS id, c" + expect: + rows: + - [f1, 1] + - [f2, 2] + + - name: return_exists_inside_case + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) RETURN s.id AS id, CASE WHEN EXISTS { MATCH (x:Other {id: s.id}) } THEN 'shared' ELSE 'own' END AS k" + cypher: "MATCH (s:File) RETURN s.id AS id, CASE WHEN EXISTS { MATCH (x:Other {id: s.id}) } THEN 'shared' ELSE 'own' END AS k" + expect: + rows: + - [f1, own] + - [f2, shared] + + # --------------------------------------------------------------------------- + # COUNT in WHERE + # --------------------------------------------------------------------------- + + - name: where_count_tied_by_a_property_map + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) WHERE COUNT { MATCH (x {id: s.id}) } = 1 RETURN s.id AS id" + cypher: "MATCH (s:File) WHERE COUNT { MATCH (x {id: s.id}) } = 1 RETURN s.id AS id" + expect: + rows: + - [f1] + + - name: where_count_tied_by_an_inner_where + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) WHERE COUNT { MATCH (x:Other) WHERE x.id = s.id } = 1 RETURN s.id AS id" + cypher: "MATCH (s:File) WHERE COUNT { MATCH (x:Other) WHERE x.id = s.id } = 1 RETURN s.id AS id" + expect: + rows: + - [f2] + + - name: where_count_inside_or + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) WHERE s.id = 'none' OR COUNT { MATCH (x {id: s.id}) } > 1 RETURN s.id AS id" + cypher: "MATCH (s:File) WHERE s.id = 'none' OR COUNT { MATCH (x {id: s.id}) } > 1 RETURN s.id AS id" + expect: + rows: + - [f2] + + # Gus comes in twice (once per KNOWS edge); each row has its own count, so + # both pass: two KNOWS edges, and two edges times three people. + - name: where_count_keeps_identical_rows_apart + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus'}), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(:Person {name: 'Vincent'})" + variants: + gql: "MATCH (a:Person)-[:KNOWS]->(b) WITH a WHERE COUNT { MATCH (a)-[:KNOWS]->(x) } = 2 AND COUNT { MATCH (a)-[:KNOWS]->(x), (y:Person) } = 6 RETURN a.name AS n" + cypher: "MATCH (a:Person)-[:KNOWS]->(b) WITH a WHERE COUNT { MATCH (a)-[:KNOWS]->(x) } = 2 AND COUNT { MATCH (a)-[:KNOWS]->(x), (y:Person) } = 6 RETURN a.name AS n" + expect: + rows: + - [Gus] + - [Gus] + + # --------------------------------------------------------------------------- + # A node or edge of the row, matched again inside the subquery + # --------------------------------------------------------------------------- + + # Each row's own edge: once, and once per city. + - name: count_through_the_edge_of_the_row + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus'}), (gus)-[:KNOWS]->(:Person {name: 'Vincent'}), (:City {name: 'Paris'}), (:City {name: 'Berlin'})" + variants: + gql: "MATCH (a)-[r:KNOWS]->(b) RETURN a.name AS a, COUNT { MATCH ()-[r]->() } AS one, COUNT { MATCH (x)-[r]->(y), (c:City) } AS two" + cypher: "MATCH (a)-[r:KNOWS]->(b) RETURN a.name AS a, COUNT { MATCH ()-[r]->() } AS one, COUNT { MATCH (x)-[r]->(y), (c:City) } AS two" + expect: + rows: + - [Alix, 1, 2] + - [Gus, 1, 2] + + # Alix knows Gus only, so with both ends from the row the pattern holds for + # Gus alone (once per city). + - name: count_through_both_nodes_of_the_row + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus'}), (gus)-[:KNOWS]->(:Person {name: 'Vincent'}), (:City {name: 'Paris'}), (:City {name: 'Berlin'})" + variants: + gql: "MATCH (a:Person {name: 'Alix'}), (b:Person) RETURN b.name AS b, COUNT { MATCH (a)-[:KNOWS]->(b), (c:City) } AS n" + cypher: "MATCH (a:Person {name: 'Alix'}), (b:Person) RETURN b.name AS b, COUNT { MATCH (a)-[:KNOWS]->(b), (c:City) } AS n" + expect: + rows: + - [Alix, 0] + - [Gus, 2] + - [Vincent, 0] + + # --------------------------------------------------------------------------- + # EXISTS in WHERE + # --------------------------------------------------------------------------- + + - name: where_exists_tied_by_a_property_map + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) WHERE EXISTS { MATCH (x:Other {id: s.id}) } RETURN s.id AS id" + cypher: "MATCH (s:File) WHERE EXISTS { MATCH (x:Other {id: s.id}) } RETURN s.id AS id" + expect: + rows: + - [f2] + + - name: where_not_exists_tied_by_a_property_map + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) WHERE NOT EXISTS { MATCH (x:Other {id: s.id}) } RETURN s.id AS id" + cypher: "MATCH (s:File) WHERE NOT EXISTS { MATCH (x:Other {id: s.id}) } RETURN s.id AS id" + expect: + rows: + - [f1] + + - name: where_exists_tied_by_an_inner_where + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) WHERE EXISTS { MATCH (x:Other) WHERE x.id = s.id } RETURN s.id AS id" + cypher: "MATCH (s:File) WHERE EXISTS { MATCH (x:Other) WHERE x.id = s.id } RETURN s.id AS id" + expect: + rows: + - [f2] + + - name: where_exists_tied_by_a_value_next_to_another_condition + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) WHERE s.id <> 'none' AND EXISTS { MATCH (x:Other {id: s.id}) } RETURN s.id AS id" + cypher: "MATCH (s:File) WHERE s.id <> 'none' AND EXISTS { MATCH (x:Other {id: s.id}) } RETURN s.id AS id" + expect: + rows: + - [f2] + + - name: where_exists_tied_by_a_value_inside_or + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:File {id: 'f3'}), (:Other {id: 'f2'})" + variants: + gql: "MATCH (s:File) WHERE s.id = 'f1' OR EXISTS { MATCH (x:Other {id: s.id}) } RETURN s.id AS id" + cypher: "MATCH (s:File) WHERE s.id = 'f1' OR EXISTS { MATCH (x:Other {id: s.id}) } RETURN s.id AS id" + expect: + rows: + - [f1] + - [f2] + + - name: where_exists_tied_to_an_unwound_value + setup: + - "INSERT (:File {id: 'f1'}), (:File {id: 'f2'}), (:Other {id: 'f2'})" + variants: + gql: "UNWIND ['f1', 'f2', 'f3'] AS k WITH k WHERE EXISTS { MATCH (x {id: k}) } RETURN k" + cypher: "UNWIND ['f1', 'f2', 'f3'] AS k WITH k WHERE EXISTS { MATCH (x {id: k}) } RETURN k" + expect: + rows: + - [f1] + - [f2] + + # Gus comes in twice; an OR keeps both rows (it filters, it does not merge). + - name: where_exists_inside_or_keeps_identical_rows + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus'}), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(:Person {name: 'Vincent'}), (:City {name: 'Paris'})" + variants: + gql: "MATCH (a:Person)-[:KNOWS]->(b) WITH a WHERE a.name = 'none' OR EXISTS { MATCH (a)-[:KNOWS]->(x), (c:City) } RETURN a.name AS n" + cypher: "MATCH (a:Person)-[:KNOWS]->(b) WITH a WHERE a.name = 'none' OR EXISTS { MATCH (a)-[:KNOWS]->(x), (c:City) } RETURN a.name AS n" + expect: + rows: + - [Alix] + - [Gus] + - [Gus] + + # An EXISTS with an inner WHERE, inside an AND within one OR branch. + - name: where_exists_inside_and_within_or + setup: + - "INSERT (:A {id: 'a', x: 1}), (b:B {id: 'b'})-[:R]->(:C {id: 'c'})" + variants: + gql: "MATCH (n) WHERE (n:A AND n.x = 1) OR (n:B AND EXISTS { MATCH (n)-[:R]->(m) WHERE m.id = 'c' }) RETURN n.id AS id" + cypher: "MATCH (n) WHERE (n:A AND n.x = 1) OR (n:B AND EXISTS { MATCH (n)-[:R]->(m) WHERE m.id = 'c' }) RETURN n.id AS id" + expect: + rows: + - [a] + - [b] + + # The inner WHERE compares with the row's edge: Alix has another KNOWS edge + # for each of hers, Gus has only the one. + - name: where_exists_compares_with_the_edge_of_the_row + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(:Person {name: 'Gus'}), (alix)-[:KNOWS]->(:Person {name: 'Vincent'}), (:Person {name: 'Mia'})<-[:KNOWS]-(:Person {name: 'Jules'})" + variants: + gql: "MATCH (a)-[r:KNOWS]->(b) WHERE EXISTS { MATCH (a)-[s:KNOWS]->(c) WHERE s <> r } RETURN a.name AS a, b.name AS b" + cypher: "MATCH (a)-[r:KNOWS]->(b) WHERE EXISTS { MATCH (a)-[s:KNOWS]->(c) WHERE s <> r } RETURN a.name AS a, b.name AS b" + expect: + rows: + - [Alix, Gus] + - [Alix, Vincent] + + # --------------------------------------------------------------------------- + # A subquery over two edges in a row, run once per row + # --------------------------------------------------------------------------- + + - name: return_exists_over_two_edges + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus'}), (gus)-[:LIVES_IN]->(:City {name: 'Berlin'}), (:Person {name: 'Harm'})" + variants: + gql: "MATCH (n:Person) RETURN n.name AS n, EXISTS { MATCH (n)-[:KNOWS]->(m)-[:LIVES_IN]->(c:City) } AS e" + cypher: "MATCH (n:Person) RETURN n.name AS n, EXISTS { MATCH (n)-[:KNOWS]->(m)-[:LIVES_IN]->(c:City) } AS e" + expect: + rows: + - [Alix, true] + - [Gus, false] + - [Harm, false] + + - name: where_exists_over_two_edges_inside_or + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus'}), (gus)-[:LIVES_IN]->(:City {name: 'Berlin'}), (:Person {name: 'Harm'})" + variants: + gql: "MATCH (n:Person) WHERE n.name = 'Harm' OR EXISTS { MATCH (n)-[:KNOWS]->(m)-[:LIVES_IN]->(c:City) } RETURN n.name AS n" + cypher: "MATCH (n:Person) WHERE n.name = 'Harm' OR EXISTS { MATCH (n)-[:KNOWS]->(m)-[:LIVES_IN]->(c:City) } RETURN n.name AS n" + expect: + rows: + - [Alix] + - [Harm] + + - name: return_count_over_two_edges + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus'}), (gus)-[:LIVES_IN]->(:City {name: 'Berlin'}), (:Person {name: 'Harm'})" + variants: + gql: "MATCH (n:Person) RETURN n.name AS n, COUNT { MATCH (n)-[:KNOWS]->(m)-[:LIVES_IN]->(c) } AS c" + cypher: "MATCH (n:Person) RETURN n.name AS n, COUNT { MATCH (n)-[:KNOWS]->(m)-[:LIVES_IN]->(c) } AS c" + expect: + rows: + - [Alix, 1] + - [Gus, 0] + - [Harm, 0] + + # Cypher only: GQL's CALL does not import outer variables yet. + - name: call_over_two_edges + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus'}), (gus)-[:LIVES_IN]->(:City {name: 'Berlin'}), (vincent:Person {name: 'Vincent'})-[:KNOWS]->(alix)" + variants: + cypher: "MATCH (a:Person) CALL { WITH a MATCH (a)-[:KNOWS]->(m)-[:KNOWS]->(x) RETURN x.name AS x } RETURN a.name AS a, x" + expect: + rows: + - [Vincent, Gus] + + # --------------------------------------------------------------------------- + # GQL VALUE subqueries that count (GQL only: VALUE is GQL) + # --------------------------------------------------------------------------- + + # Alix knows Gus (age 25) and Vincent (no age), and likes Gus twice. + - name: value_count_of_a_property_skips_nulls + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus', age: 25}), (alix)-[:KNOWS]->(:Person {name: 'Vincent'}), (alix)-[:LIKES]->(gus), (alix)-[:LIKES]->(gus)" + variants: + gql: "MATCH (a:Person {name: 'Alix'}) RETURN VALUE { MATCH (a)-[:KNOWS]->(b) RETURN count(b.age) } AS c" + expect: + rows: + - [1] + + - name: value_count_distinct + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus', age: 25}), (alix)-[:KNOWS]->(:Person {name: 'Vincent'}), (alix)-[:LIKES]->(gus), (alix)-[:LIKES]->(gus)" + variants: + gql: "MATCH (a:Person {name: 'Alix'}) RETURN VALUE { MATCH (a)-[:LIKES]->(b) RETURN count(DISTINCT b) } AS c" + expect: + rows: + - [1] + + - name: value_count_distinct_property_skips_nulls_and_duplicates + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(:Person {name: 'Gus', age: 25}), (alix)-[:KNOWS]->(:Person {name: 'Vincent'}), (alix)-[:KNOWS]->(:Person {name: 'Jules', age: 25})" + variants: + gql: "MATCH (a:Person) RETURN a.name AS n, VALUE { MATCH (a)-[:KNOWS]->(b) RETURN count(DISTINCT b.age) } AS c ORDER BY n" + expect: + ordered: true + rows: + - [Alix, 1] + - [Gus, 0] + - [Jules, 0] + - [Vincent, 0] + + - name: value_count_of_all_matches + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus', age: 25}), (alix)-[:KNOWS]->(:Person {name: 'Vincent'}), (alix)-[:LIKES]->(gus), (alix)-[:LIKES]->(gus)" + variants: + gql: "MATCH (a:Person {name: 'Alix'}) RETURN VALUE { MATCH (a)-[:LIKES]->(b) RETURN count(*) } AS c" + expect: + rows: + - [2] + + - name: value_count_of_a_property_in_where + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus', age: 25}), (alix)-[:KNOWS]->(:Person {name: 'Vincent'}), (alix)-[:LIKES]->(gus), (alix)-[:LIKES]->(gus)" + variants: + gql: "MATCH (a:Person) WHERE VALUE { MATCH (a)-[:KNOWS]->(b) RETURN count(b.age) } = 1 RETURN a.name AS n" + expect: + rows: + - [Alix] + + # A variable named like the planner's own subquery columns keeps its value: + # the planner picks a name the row does not have. + - name: a_variable_named_like_a_subquery_column + setup: + - "INSERT (alix:Person {name: 'Alix'})-[:KNOWS]->(gus:Person {name: 'Gus'}), (:City {name: 'Paris'})" + variants: + gql: "MATCH (a:Person {name: 'Alix'}) LET __subquery_0 = 7 RETURN __subquery_0 AS mine, EXISTS { MATCH (a)-[:KNOWS]->(b), (c:City) } AS e" + cypher: "MATCH (a:Person {name: 'Alix'}) WITH a, 7 AS __subquery_0 RETURN __subquery_0 AS mine, EXISTS { MATCH (a)-[:KNOWS]->(b), (c:City) } AS e" + expect: + rows: + - [7, true] + + # --------------------------------------------------------------------------- + # A subquery that shares nothing with the row has one answer for all rows + # --------------------------------------------------------------------------- + + # Five KNOWS edges, the same for every city. + - name: a_count_that_shares_nothing_with_the_row + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH (c:City) RETURN c.name AS c, COUNT { MATCH (x)-[:KNOWS]->(y) } AS n" + cypher: "MATCH (c:City) RETURN c.name AS c, COUNT { MATCH (x)-[:KNOWS]->(y) } AS n" + expect: + rows: + - [Amsterdam, 5] + - [Berlin, 5] + - [Paris, 5] + + - name: an_exists_that_shares_nothing_with_the_row + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH (c:City) RETURN c.name AS c, EXISTS { MATCH (x)-[:LIVES_IN]->(y:City {name: 'Paris'}) } AS e, EXISTS { MATCH (x)-[:LIVES_IN]->(y:City {name: 'Prague'}) } AS f" + cypher: "MATCH (c:City) RETURN c.name AS c, EXISTS { MATCH (x)-[:LIVES_IN]->(y:City {name: 'Paris'}) } AS e, EXISTS { MATCH (x)-[:LIVES_IN]->(y:City {name: 'Prague'}) } AS f" + expect: + rows: + - [Amsterdam, true, false] + - [Berlin, true, false] + - [Paris, true, false] + + # The same in WITH, and on an empty graph part: no edge of the type. + - name: a_count_that_shares_nothing_in_with + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH (c:City) WITH c, COUNT { MATCH (x)-[:LIKES]->(y) } AS n RETURN c.name AS c, n" + cypher: "MATCH (c:City) WITH c, COUNT { MATCH (x)-[:LIKES]->(y) } AS n RETURN c.name AS c, n" + expect: + rows: + - [Amsterdam, 0] + - [Berlin, 0] + - [Paris, 0] + + # In WHERE too: one count for every city, combined with a row condition. + - name: a_count_that_shares_nothing_in_where + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH (c:City) WHERE COUNT { MATCH (x)-[:KNOWS]->(y) } = 5 AND c.name <> 'Berlin' RETURN c.name AS c" + cypher: "MATCH (c:City) WHERE COUNT { MATCH (x)-[:KNOWS]->(y) } = 5 AND c.name <> 'Berlin' RETURN c.name AS c" + expect: + rows: + - [Amsterdam] + - [Paris] + + # After a write in the same statement the subquery sees every row's write: + # three tags, one per city, counted on each row. + - name: a_count_that_shares_nothing_after_a_write + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH (c:City) INSERT (:Tag) RETURN c.name AS c, COUNT { MATCH (t:Tag) } AS n" + cypher: "MATCH (c:City) CREATE (:Tag) RETURN c.name AS c, COUNT { MATCH (t:Tag) } AS n" + expect: + rows: + - [Amsterdam, 3] + - [Berlin, 3] + - [Paris, 3] diff --git a/tests/spec/rosetta/edge_lists.gtest b/tests/spec/rosetta/edge_lists.gtest new file mode 100644 index 000000000..1bbe57977 --- /dev/null +++ b/tests/spec/rosetta/edge_lists.gtest @@ -0,0 +1,84 @@ +# Rosetta: the edge list of a variable-length pattern +# +# A quantified edge pattern binds its variable to the list of the path's +# edges, also when the quantifier allows one hop only (Cypher `*1`, `*1..1`, +# GQL `{1,1}`): it used to bind a single edge there, so size(), head() and +# list functions read null. An item taken from such a list stays an edge, +# also through collect(). +# +# Dataset `entity_kinds` (tests/spec/datasets/entity_kinds.setup): KNOWS +# Alix->Gus (w 1), Gus->Vincent (w 2), Vincent->Alix (w 3), Jules->Mia (w 4) +# and Alix->Jules (w 5). + +meta: + model: lpg + section: rosetta + title: The edge list of a variable-length pattern + dataset: entity_kinds + requires: [cypher] + +tests: + + - name: a_one_hop_quantifier_binds_a_list + variants: + gql: "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS]->{1,1}(b) RETURN b.name AS b, size(rs) AS n" + cypher: "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS*1..1]->(b) RETURN b.name AS b, size(rs) AS n" + expect: + rows: + - [Gus, 1] + - [Jules, 1] + + - name: a_single_hop_count_binds_a_list + variants: + cypher: "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS*1]->(b) RETURN b.name AS b, size(rs) AS n" + expect: + rows: + - [Gus, 1] + - [Jules, 1] + + - name: the_list_of_a_one_hop_quantifier_in_where + variants: + gql: "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS]->{1,1}(b) WHERE size(rs) = 1 RETURN b.name AS b" + cypher: "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS*1..1]->(b) WHERE size(rs) = 1 RETURN b.name AS b" + expect: + rows: + - [Jules] + - [Gus] + + - name: the_list_of_a_one_hop_quantifier_through_with + variants: + gql: "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS]->{1,1}(b) WITH b, size(rs) AS n RETURN b.name AS b, n" + cypher: "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS*1..1]->(b) WITH b, size(rs) AS n RETURN b.name AS b, n" + expect: + rows: + - [Gus, 1] + - [Jules, 1] + + - name: collected_first_edges_stay_edges + variants: + gql: "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS]->{1,2}(b) WITH collect(head(rs)) AS es UNWIND es AS e RETURN type(e) AS t, e.w AS w" + cypher: "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS*1..2]->(b) WITH collect(head(rs)) AS es UNWIND es AS e RETURN type(e) AS t, e.w AS w" + expect: + rows: + - [KNOWS, 1] + - [KNOWS, 1] + - [KNOWS, 5] + - [KNOWS, 5] + + - name: collected_last_relationships_stay_edges + variants: + gql: "MATCH p = (a:Person {name: 'Alix'})-[:KNOWS]->(b) WITH collect(last(relationships(p))) AS es UNWIND es AS e RETURN type(e) AS t, e.w AS w" + cypher: "MATCH p = (a:Person {name: 'Alix'})-[:KNOWS]->(b) WITH collect(last(relationships(p))) AS es UNWIND es AS e RETURN type(e) AS t, e.w AS w" + expect: + rows: + - [KNOWS, 1] + - [KNOWS, 5] + + - name: a_plain_edge_is_still_an_edge + variants: + gql: "MATCH (a:Person {name: 'Alix'})-[r:KNOWS]->(b) RETURN type(r) AS t, r.w AS w" + cypher: "MATCH (a:Person {name: 'Alix'})-[r:KNOWS]->(b) RETURN type(r) AS t, r.w AS w" + expect: + rows: + - [KNOWS, 1] + - [KNOWS, 5] diff --git a/tests/spec/rosetta/exists_count_shared_variables.gtest b/tests/spec/rosetta/exists_count_shared_variables.gtest new file mode 100644 index 000000000..e82a653b9 --- /dev/null +++ b/tests/spec/rosetta/exists_count_shared_variables.gtest @@ -0,0 +1,243 @@ +# Rosetta: EXISTS and COUNT subqueries that share nodes and edges with the row +# +# A subquery variable the row already binds is that node or edge: both ends +# of an edge, the edge itself, or only the far end. A named variable the row +# does not bind is new. A node the row holds as null matches nothing, so +# EXISTS is false and COUNT is 0. +# +# Setup (GQL, one statement): Alix, Gus, Vincent and Mia, and Amsterdam. +# KNOWS Alix->Gus, Gus->Alix, Gus->Vincent and Vincent->Mia (only Alix and Gus +# know each other both ways); LIVES_IN Alix->Amsterdam and Mia->Amsterdam. + +meta: + model: lpg + section: rosetta + title: EXISTS and COUNT subqueries that share nodes and edges with the row + dataset: empty + requires: [cypher] + +tests: + + # --------------------------------------------------------------------------- + # In WHERE + # --------------------------------------------------------------------------- + + - name: exists_an_edge_between_two_bound_nodes + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a:Person)-[:KNOWS]->(b:Person) WHERE EXISTS { MATCH (b)-[:KNOWS]->(a) } RETURN a.name AS a, b.name AS b" + cypher: "MATCH (a:Person)-[:KNOWS]->(b:Person) WHERE EXISTS { MATCH (b)-[:KNOWS]->(a) } RETURN a.name AS a, b.name AS b" + expect: + rows: + - [Alix, Gus] + - [Gus, Alix] + + - name: not_exists_an_edge_between_two_bound_nodes + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a:Person)-[:KNOWS]->(b:Person) WHERE NOT EXISTS { MATCH (b)-[:KNOWS]->(a) } RETURN a.name AS a, b.name AS b" + cypher: "MATCH (a:Person)-[:KNOWS]->(b:Person) WHERE NOT EXISTS { MATCH (b)-[:KNOWS]->(a) } RETURN a.name AS a, b.name AS b" + expect: + rows: + - [Gus, Vincent] + - [Vincent, Mia] + + - name: exists_an_edge_between_two_bound_nodes_or_a_condition + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a:Person)-[:KNOWS]->(b:Person) WHERE EXISTS { MATCH (b)-[:KNOWS]->(a) } OR a.name = 'Vincent' RETURN a.name AS a, b.name AS b" + cypher: "MATCH (a:Person)-[:KNOWS]->(b:Person) WHERE EXISTS { MATCH (b)-[:KNOWS]->(a) } OR a.name = 'Vincent' RETURN a.name AS a, b.name AS b" + expect: + rows: + - [Alix, Gus] + - [Gus, Alix] + - [Vincent, Mia] + + - name: count_edges_between_two_bound_nodes + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a:Person)-[:KNOWS]->(b:Person) WHERE COUNT { MATCH (b)-[:KNOWS]->(a) } = 0 RETURN a.name AS a, b.name AS b" + cypher: "MATCH (a:Person)-[:KNOWS]->(b:Person) WHERE COUNT { MATCH (b)-[:KNOWS]->(a) } = 0 RETURN a.name AS a, b.name AS b" + expect: + rows: + - [Gus, Vincent] + - [Vincent, Mia] + + - name: exists_through_a_bound_edge + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a)-[r]->(b) WHERE EXISTS { MATCH (x)-[r]->(:City) } RETURN a.name AS a" + cypher: "MATCH (a)-[r]->(b) WHERE EXISTS { MATCH (x)-[r]->(:City) } RETURN a.name AS a" + expect: + rows: + - [Alix] + - [Mia] + + - name: exists_with_a_new_named_start + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (c:City) WHERE EXISTS { MATCH (x)-[:LIVES_IN]->(:City) } RETURN c.name AS c" + cypher: "MATCH (c:City) WHERE EXISTS { MATCH (x)-[:LIVES_IN]->(:City) } RETURN c.name AS c" + expect: + rows: + - [Amsterdam] + + - name: not_exists_with_a_new_named_start + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (c:City) WHERE NOT EXISTS { MATCH (x)-[:WORKS_AT]->() } RETURN c.name AS c" + cypher: "MATCH (c:City) WHERE NOT EXISTS { MATCH (x)-[:WORKS_AT]->() } RETURN c.name AS c" + expect: + rows: + - [Amsterdam] + + - name: exists_a_path_to_a_bound_node + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a:Person {name: 'Alix'}), (b:Person) WHERE EXISTS { MATCH (a)-[:KNOWS*1..2]->(b) } RETURN b.name AS b" + cypher: "MATCH (a:Person {name: 'Alix'}), (b:Person) WHERE EXISTS { MATCH (a)-[:KNOWS*1..2]->(b) } RETURN b.name AS b" + expect: + rows: + - [Alix] + - [Gus] + - [Vincent] + + - name: not_exists_from_a_null_node + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a:Person) OPTIONAL MATCH (a)-[:LIVES_IN]->(c) WITH a, c WHERE NOT EXISTS { MATCH (c)<-[:LIVES_IN]-() } RETURN a.name AS a" + cypher: "MATCH (a:Person) OPTIONAL MATCH (a)-[:LIVES_IN]->(c) WITH a, c WHERE NOT EXISTS { MATCH (c)<-[:LIVES_IN]-() } RETURN a.name AS a" + expect: + rows: + - [Gus] + - [Vincent] + + # --------------------------------------------------------------------------- + # In RETURN + # --------------------------------------------------------------------------- + + - name: return_exists_an_edge_between_two_bound_nodes + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a:Person)-[:KNOWS]->(b:Person) RETURN a.name AS a, b.name AS b, EXISTS { MATCH (b)-[:KNOWS]->(a) } AS back" + cypher: "MATCH (a:Person)-[:KNOWS]->(b:Person) RETURN a.name AS a, b.name AS b, EXISTS { MATCH (b)-[:KNOWS]->(a) } AS back" + expect: + rows: + - [Alix, Gus, true] + - [Gus, Alix, true] + - [Gus, Vincent, false] + - [Vincent, Mia, false] + + - name: return_count_edges_between_two_bound_nodes + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a:Person)-[:KNOWS]->(b:Person) RETURN a.name AS a, b.name AS b, COUNT { MATCH (b)-[:KNOWS]->(a) } AS back" + cypher: "MATCH (a:Person)-[:KNOWS]->(b:Person) RETURN a.name AS a, b.name AS b, COUNT { MATCH (b)-[:KNOWS]->(a) } AS back" + expect: + rows: + - [Alix, Gus, 1] + - [Gus, Alix, 1] + - [Gus, Vincent, 0] + - [Vincent, Mia, 0] + + - name: return_count_with_a_new_named_end + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a:Person) RETURN a.name AS a, COUNT { MATCH (a)-[:KNOWS]->(f) } AS friends" + cypher: "MATCH (a:Person) RETURN a.name AS a, COUNT { MATCH (a)-[:KNOWS]->(f) } AS friends" + expect: + rows: + - [Alix, 1] + - [Gus, 2] + - [Vincent, 1] + - [Mia, 0] + + - name: return_exists_through_a_bound_edge + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a)-[r]->(b) RETURN a.name AS a, b.name AS b, EXISTS { MATCH (x)-[r]->(:City) } AS home" + cypher: "MATCH (a)-[r]->(b) RETURN a.name AS a, b.name AS b, EXISTS { MATCH (x)-[r]->(:City) } AS home" + expect: + rows: + - [Alix, Gus, false] + - [Gus, Alix, false] + - [Gus, Vincent, false] + - [Vincent, Mia, false] + - [Alix, Amsterdam, true] + - [Mia, Amsterdam, true] + + # The walks of one to three KNOWS edges from Alix, each bound to rs as a + # list: the subquery has to follow those edges in order, within its own + # bounds, direction and end. + - name: return_exists_through_a_bound_edge_list + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS*1..3]->(b) RETURN b.name AS b, size(rs) AS hops, EXISTS { MATCH (x)-[rs:KNOWS*1..2]->(y) } AS short, EXISTS { MATCH (x)-[rs:KNOWS*]->(a) } AS back_to_a, EXISTS { MATCH (x)<-[rs:KNOWS*]-(y) } AS reversed" + cypher: "MATCH (a:Person {name: 'Alix'})-[rs:KNOWS*1..3]->(b) RETURN b.name AS b, size(rs) AS hops, EXISTS { MATCH (x)-[rs:KNOWS*1..2]->(y) } AS short, EXISTS { MATCH (x)-[rs:KNOWS*]->(a) } AS back_to_a, EXISTS { MATCH (x)<-[rs:KNOWS*]-(y) } AS reversed" + expect: + rows: + - [Gus, 1, true, false, true] + - [Alix, 2, true, true, true] + - [Vincent, 2, true, false, false] + - [Gus, 3, false, false, true] + - [Mia, 3, false, false, false] + + - name: return_count_with_a_new_named_start + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (c:City) RETURN c.name AS c, COUNT { MATCH (x)-[:KNOWS]->(y) } AS knows" + cypher: "MATCH (c:City) RETURN c.name AS c, COUNT { MATCH (x)-[:KNOWS]->(y) } AS knows" + expect: + rows: + - [Amsterdam, 4] + + - name: return_exists_and_count_with_only_the_end_bound + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (c:City) RETURN c.name AS c, EXISTS { MATCH (x)-[:LIVES_IN]->(c) } AS lived_in, COUNT { MATCH (x)-[:LIVES_IN]->(c) } AS residents" + cypher: "MATCH (c:City) RETURN c.name AS c, EXISTS { MATCH (x)-[:LIVES_IN]->(c) } AS lived_in, COUNT { MATCH (x)-[:LIVES_IN]->(c) } AS residents" + expect: + rows: + - [Amsterdam, true, 2] + + - name: return_exists_a_path_to_a_bound_node + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a:Person {name: 'Alix'}), (b:Person) RETURN b.name AS b, EXISTS { MATCH (a)-[:KNOWS*1..2]->(b) } AS near" + cypher: "MATCH (a:Person {name: 'Alix'}), (b:Person) RETURN b.name AS b, EXISTS { MATCH (a)-[:KNOWS*1..2]->(b) } AS near" + expect: + rows: + - [Alix, true] + - [Gus, true] + - [Vincent, true] + - [Mia, false] + + - name: return_exists_and_count_from_a_null_node + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams)" + variants: + gql: "MATCH (a:Person) OPTIONAL MATCH (a)-[:LIVES_IN]->(c) RETURN a.name AS a, EXISTS { MATCH (c)<-[:LIVES_IN]-() } AS shared, COUNT { MATCH (c)<-[:LIVES_IN]-() } AS residents" + cypher: "MATCH (a:Person) OPTIONAL MATCH (a)-[:LIVES_IN]->(c) RETURN a.name AS a, EXISTS { MATCH (c)<-[:LIVES_IN]-() } AS shared, COUNT { MATCH (c)<-[:LIVES_IN]-() } AS residents" + expect: + rows: + - [Alix, true, 2] + - [Gus, false, 0] + - [Vincent, false, 0] + - [Mia, true, 2] diff --git a/tests/spec/rosetta/exists_subquery_paths.gtest b/tests/spec/rosetta/exists_subquery_paths.gtest new file mode 100644 index 000000000..b69c99e5d --- /dev/null +++ b/tests/spec/rosetta/exists_subquery_paths.gtest @@ -0,0 +1,118 @@ +# Rosetta: EXISTS subqueries over paths +# +# An EXISTS subquery holds when its pattern matches at least once, with every +# condition in it: the hop range of a variable-length edge, a label on the far +# end of the path (not on each hop), and a label on the outer variable. +# +# Setup (GQL): top -[:CONTAINS]-> sub -[:CONTAINS]-> f, and lone -[:CONTAINS]-> f. +# top, sub and lone are Graph:Directory (lone also Archive), f is Graph:File. + +meta: + model: lpg + section: rosetta + title: EXISTS subqueries over paths + dataset: empty + requires: [cypher] + +tests: + + - name: variable_length_path_to_a_labeled_end + setup: + - "INSERT (:Graph:Directory {id: 'top'})-[:CONTAINS]->(:Graph:Directory {id: 'sub'})-[:CONTAINS]->(:Graph:File {id: 'f'})" + - "MATCH (f:File) INSERT (:Graph:Directory:Archive {id: 'lone'})-[:CONTAINS]->(f)" + variants: + gql: "MATCH (n:Directory) WHERE EXISTS { MATCH (n)-[:CONTAINS]->{1,}(f:File) } RETURN n.id ORDER BY n.id" + cypher: "MATCH (n:Directory) WHERE EXISTS { MATCH (n)-[:CONTAINS*]->(f:File) } RETURN n.id ORDER BY n.id" + expect: + ordered: true + rows: + - [lone] + - [sub] + - [top] + + - name: bounded_path_to_a_labeled_end + setup: + - "INSERT (:Graph:Directory {id: 'top'})-[:CONTAINS]->(:Graph:Directory {id: 'sub'})-[:CONTAINS]->(:Graph:File {id: 'f'})" + - "MATCH (f:File) INSERT (:Graph:Directory:Archive {id: 'lone'})-[:CONTAINS]->(f)" + variants: + gql: "MATCH (n:Directory) WHERE EXISTS { MATCH (n)-[]->{1,5}(:File) } RETURN n.id ORDER BY n.id" + cypher: "MATCH (n:Directory) WHERE EXISTS { MATCH (n)-[*1..5]->(:File) } RETURN n.id ORDER BY n.id" + expect: + ordered: true + rows: + - [lone] + - [sub] + - [top] + + - name: path_with_a_minimum_of_two_hops + setup: + - "INSERT (:Graph:Directory {id: 'top'})-[:CONTAINS]->(:Graph:Directory {id: 'sub'})-[:CONTAINS]->(:Graph:File {id: 'f'})" + - "MATCH (f:File) INSERT (:Graph:Directory:Archive {id: 'lone'})-[:CONTAINS]->(f)" + variants: + gql: "MATCH (n:Directory) WHERE EXISTS { MATCH (n)-[:CONTAINS]->{2,3}() } RETURN n.id" + cypher: "MATCH (n:Directory) WHERE EXISTS { MATCH (n)-[:CONTAINS*2..3]->() } RETURN n.id" + expect: + rows: + - [top] + + - name: not_exists_over_a_path + setup: + - "INSERT (:Graph:Directory {id: 'top'})-[:CONTAINS]->(:Graph:Directory {id: 'sub'})-[:CONTAINS]->(:Graph:File {id: 'f'})" + - "MATCH (f:File) INSERT (:Graph:Directory:Archive {id: 'lone'})-[:CONTAINS]->(f)" + variants: + gql: "MATCH (n:Directory) WHERE NOT EXISTS { MATCH (n)-[:CONTAINS]->{2,}(:File) } RETURN n.id ORDER BY n.id" + cypher: "MATCH (n:Directory) WHERE NOT EXISTS { MATCH (n)-[:CONTAINS*2..]->(:File) } RETURN n.id ORDER BY n.id" + expect: + ordered: true + rows: + - [lone] + - [sub] + + - name: path_with_no_condition_on_its_end + setup: + - "INSERT (:Graph:Directory {id: 'top'})-[:CONTAINS]->(:Graph:Directory {id: 'sub'})-[:CONTAINS]->(:Graph:File {id: 'f'})" + - "MATCH (f:File) INSERT (:Graph:Directory:Archive {id: 'lone'})-[:CONTAINS]->(f)" + variants: + gql: "MATCH (n:Graph) WHERE EXISTS { MATCH (n)-[:CONTAINS]->{1,}() } RETURN n.id ORDER BY n.id" + cypher: "MATCH (n:Graph) WHERE EXISTS { MATCH (n)-[:CONTAINS*]->() } RETURN n.id ORDER BY n.id" + expect: + ordered: true + rows: + - [lone] + - [sub] + - [top] + + - name: label_on_the_outer_variable + setup: + - "INSERT (:Graph:Directory {id: 'top'})-[:CONTAINS]->(:Graph:Directory {id: 'sub'})-[:CONTAINS]->(:Graph:File {id: 'f'})" + - "MATCH (f:File) INSERT (:Graph:Directory:Archive {id: 'lone'})-[:CONTAINS]->(f)" + variants: + gql: "MATCH (n:Directory) WHERE EXISTS { MATCH (n)-[:CONTAINS]->(m) WHERE n:Archive } RETURN n.id" + cypher: "MATCH (n:Directory) WHERE EXISTS { MATCH (n)-[:CONTAINS]->(m) WHERE n:Archive } RETURN n.id" + expect: + rows: + - [lone] + + - name: label_on_the_outer_variable_in_the_pattern + setup: + - "INSERT (:Graph:Directory {id: 'top'})-[:CONTAINS]->(:Graph:Directory {id: 'sub'})-[:CONTAINS]->(:Graph:File {id: 'f'})" + - "MATCH (f:File) INSERT (:Graph:Directory:Archive {id: 'lone'})-[:CONTAINS]->(f)" + variants: + gql: "MATCH (n:Directory) WHERE EXISTS { MATCH (n:Archive)-[:CONTAINS]->() } RETURN n.id" + cypher: "MATCH (n:Directory) WHERE EXISTS { MATCH (n:Archive)-[:CONTAINS]->() } RETURN n.id" + expect: + rows: + - [lone] + + - name: pattern_predicate_over_a_path + setup: + - "INSERT (:Graph:Directory {id: 'top'})-[:CONTAINS]->(:Graph:Directory {id: 'sub'})-[:CONTAINS]->(:Graph:File {id: 'f'})" + - "MATCH (f:File) INSERT (:Graph:Directory:Archive {id: 'lone'})-[:CONTAINS]->(f)" + variants: + cypher: "MATCH (n:Directory) WHERE (n)-[:CONTAINS*]->(:File) RETURN n.id ORDER BY n.id" + expect: + ordered: true + rows: + - [lone] + - [sub] + - [top] diff --git a/tests/spec/rosetta/exists_whole_pattern.gtest b/tests/spec/rosetta/exists_whole_pattern.gtest new file mode 100644 index 000000000..47444eb7e --- /dev/null +++ b/tests/spec/rosetta/exists_whole_pattern.gtest @@ -0,0 +1,201 @@ +# Rosetta: an EXISTS subquery matches its whole pattern +# +# Besides its edge, the subquery's pattern can hold another node pattern, +# which has to match as well, and a path mode, which limits the paths of a +# variable-length edge. +# +# Setup (GQL, one statement): Alix, Gus, Vincent and Mia, and Amsterdam. +# KNOWS Alix->Gus, Gus->Alix, Gus->Vincent and Vincent->Mia; LIVES_IN +# Alix->Amsterdam and Mia->Amsterdam; LIKES Alix->Gus and Mia->Mia. There is +# no Robot. + +meta: + model: lpg + section: rosetta + title: An EXISTS subquery matches its whole pattern + dataset: empty + requires: [cypher] + +tests: + + # --------------------------------------------------------------------------- + # A node pattern apart from the edge + # --------------------------------------------------------------------------- + + - name: exists_with_a_second_node_pattern + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b), (c:Robot) } RETURN a.name AS a" + cypher: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b), (c:Robot) } RETURN a.name AS a" + expect: + empty: true + + - name: not_exists_with_a_node_pattern_before_the_edge + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) WHERE NOT EXISTS { MATCH (c:Robot), (a)-[:KNOWS]->(b) } RETURN a.name AS a" + cypher: "MATCH (a:Person) WHERE NOT EXISTS { MATCH (c:Robot), (a)-[:KNOWS]->(b) } RETURN a.name AS a" + expect: + rows: + - [Alix] + - [Gus] + - [Vincent] + - [Mia] + + - name: exists_with_a_second_match + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b) MATCH (c:Robot) } RETURN a.name AS a" + cypher: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b) MATCH (c:Robot) } RETURN a.name AS a" + expect: + empty: true + + # A second MATCH on the end of the edge adds its label to that end: the + # MATCH clauses of a subquery share their variables. + - name: exists_with_a_label_on_the_end_in_a_second_match + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b) MATCH (b:City) } RETURN a.name AS a" + cypher: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b) MATCH (b:City) } RETURN a.name AS a" + expect: + empty: true + + - name: return_count_with_a_label_on_the_end_in_a_second_match + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) RETURN a.name AS a, COUNT { MATCH (a)-[r]->(b) MATCH (b:City) } AS cities" + cypher: "MATCH (a:Person) RETURN a.name AS a, COUNT { MATCH (a)-[r]->(b) MATCH (b:City) } AS cities" + expect: + rows: + - [Alix, 1] + - [Gus, 0] + - [Vincent, 0] + - [Mia, 1] + + # The second MATCH goes on from the end of the first: Gus knows Alix and + # Vincent knows Mia, who live in Amsterdam; Gus, whom Alix knows, lives + # nowhere. + - name: exists_through_a_second_match_from_the_end + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b) MATCH (b)-[:LIVES_IN]->(c:City) } RETURN a.name AS a" + cypher: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b) MATCH (b)-[:LIVES_IN]->(c:City) } RETURN a.name AS a" + expect: + rows: + - [Gus] + - [Vincent] + + - name: value_count_through_a_second_match_from_the_end + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) RETURN a.name AS a, VALUE { MATCH (a)-[:KNOWS]->(b) MATCH (b)-[:LIVES_IN]->(c) RETURN count(*) } AS c" + expect: + rows: + - [Alix, 0] + - [Gus, 1] + - [Vincent, 1] + - [Mia, 0] + + # An OPTIONAL MATCH keeps every row of the MATCH before it. + - name: count_with_an_optional_match_from_the_end + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) RETURN a.name AS a, COUNT { MATCH (a)-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:LIVES_IN]->(c) } AS c" + cypher: "MATCH (a:Person) RETURN a.name AS a, COUNT { MATCH (a)-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:LIVES_IN]->(c) } AS c" + expect: + rows: + - [Alix, 1] + - [Gus, 2] + - [Vincent, 1] + - [Mia, 0] + + # The same in EXISTS: the OPTIONAL MATCH does not change who has a match. + - name: exists_with_an_optional_match_from_the_end + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:LIVES_IN]->(c) } RETURN a.name AS a" + cypher: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:LIVES_IN]->(c) } RETURN a.name AS a" + expect: + rows: + - [Alix] + - [Gus] + - [Vincent] + + # The OPTIONAL MATCH reads a variable of the outer row (a.age) and one of the + # MATCH before it (b). Alix knows Gus, who knows the older Vincent, and + # Jules, who knows nobody older than Alix: 2 rows. A row count does not show + # whether the condition held (a failed match keeps a null row), so + # `optional_match_conditions.gtest` checks the matches themselves. + - name: count_with_an_optional_match_reading_earlier_variables + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30}), (gus:Person {name: 'Gus', age: 25}), (vincent:Person {name: 'Vincent', age: 40}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(alix), (jules)-[:KNOWS]->(mia), (alix)-[:KNOWS]->(jules)" + variants: + gql: "MATCH (a:Person) RETURN a.name AS a, COUNT { MATCH (a)-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c WHERE c.age > a.age) } AS n" + cypher: "MATCH (a:Person) RETURN a.name AS a, COUNT { MATCH (a)-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c) WHERE c.age > a.age } AS n" + expect: + rows: + - [Alix, 2] + - [Gus, 1] + - [Vincent, 1] + - [Jules, 1] + - [Mia, 0] + + # --------------------------------------------------------------------------- + # A path mode (GQL only, except the plain walk: Cypher has no path modes) + # --------------------------------------------------------------------------- + + # Mia's only LIKES edge returns to her: a walk, not an acyclic path. + - name: exists_a_walk + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:LIKES*1..2]->(x) } RETURN a.name AS a" + cypher: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:LIKES*1..2]->(x) } RETURN a.name AS a" + expect: + rows: + - [Alix] + - [Mia] + + - name: exists_an_acyclic_path + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) WHERE EXISTS { MATCH ACYCLIC (a)-[:LIKES*1..2]->(x) } RETURN a.name AS a" + expect: + rows: + - [Alix] + + # Back to the start in at most two KNOWS edges, either way: Vincent and Mia + # only over the edge they left by, Alix and Gus also over the other edge + # between them. GQL only: openCypher uses each relationship at most once per + # pattern, so Vincent and Mia would not come back. + - name: exists_a_closed_walk + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) WITH a, a AS b WHERE EXISTS { MATCH (a)-[:KNOWS*1..2]-(b) } RETURN a.name AS a" + expect: + rows: + - [Alix] + - [Gus] + - [Vincent] + - [Mia] + + - name: exists_a_closed_trail + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (vincent:Person {name: 'Vincent'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (alix)-[:KNOWS]->(gus), (gus)-[:KNOWS]->(alix), (gus)-[:KNOWS]->(vincent), (vincent)-[:KNOWS]->(mia), (alix)-[:LIVES_IN]->(ams), (mia)-[:LIVES_IN]->(ams), (alix)-[:LIKES]->(gus), (mia)-[:LIKES]->(mia)" + variants: + gql: "MATCH (a:Person) WITH a, a AS b WHERE EXISTS { MATCH TRAIL (a)-[:KNOWS*1..2]-(b) } RETURN a.name AS a" + expect: + rows: + - [Alix] + - [Gus] diff --git a/tests/spec/rosetta/merge_matches.gtest b/tests/spec/rosetta/merge_matches.gtest new file mode 100644 index 000000000..0d79d0f43 --- /dev/null +++ b/tests/spec/rosetta/merge_matches.gtest @@ -0,0 +1,72 @@ +# Rosetta: MERGE binds every match +# +# As in openCypher, a MERGE whose pattern matches more than one node or +# relationship binds each of them: one row per match, and ON MATCH and a later +# SET apply to all of them. It used to bind the first match only. +# +# Each case sets up two Tag nodes with k = 1 (or two U edges with id 'u1' +# between the same two nodes), so binding only one shows. The later MATCH +# reads what the MERGE wrote. + +meta: + model: lpg + section: rosetta + title: MERGE binds every match + dataset: empty + requires: [cypher] + +tests: + + - name: merge_returns_a_row_per_matching_node + setup: + - "INSERT (:Tag {k: 1}), (:Tag {k: 1}), (:Tag {k: 2})" + variants: + gql: "MERGE (t:Tag {k: 1}) RETURN count(*) AS n" + cypher: "MERGE (t:Tag {k: 1}) RETURN count(*) AS n" + expect: + rows: + - [2] + + - name: a_set_after_merge_writes_every_matching_node + setup: + - "INSERT (:Tag {k: 1}), (:Tag {k: 1}), (:Tag {k: 2})" + variants: + gql: "MERGE (t:Tag {k: 1}) SET t.seen = true WITH count(*) AS merged MATCH (t:Tag) RETURN merged, t.k AS k, t.seen AS seen" + cypher: "MERGE (t:Tag {k: 1}) SET t.seen = true WITH count(*) AS merged MATCH (t:Tag) RETURN merged, t.k AS k, t.seen AS seen" + expect: + rows: + - [2, 1, true] + - [2, 1, true] + - [2, 2, null] + + - name: on_match_writes_every_matching_node + setup: + - "INSERT (:Tag {k: 1}), (:Tag {k: 1})" + variants: + cypher: "MERGE (t:Tag {k: 1}) ON MATCH SET t.hits = 1 ON CREATE SET t.hits = 0 WITH count(*) AS merged MATCH (t:Tag) RETURN merged, t.hits AS hits" + expect: + rows: + - [2, 1] + - [2, 1] + + - name: merge_creates_one_node_when_none_matches + setup: + - "INSERT (:Tag {k: 1}), (:Tag {k: 1})" + variants: + gql: "MERGE (t:Tag {k: 3}) WITH count(*) AS merged MATCH (t:Tag) RETURN merged, count(t) AS n" + cypher: "MERGE (t:Tag {k: 3}) WITH count(*) AS merged MATCH (t:Tag) RETURN merged, count(t) AS n" + expect: + rows: + - [1, 3] + + - name: a_set_after_merge_writes_every_matching_relationship + setup: + - "INSERT (a:Node {id: 'a'}), (b:Node {id: 'b'}), (a)-[:U {id: 'u1', w: 1}]->(b), (a)-[:U {id: 'u1', w: 1}]->(b), (a)-[:U {id: 'u2', w: 1}]->(b)" + variants: + gql: "MATCH (a:Node {id: 'a'}), (b:Node {id: 'b'}) MERGE (a)-[r:U {id: 'u1'}]->(b) SET r.w = 5 WITH count(*) AS merged MATCH ()-[s:U]->() RETURN merged, s.id AS id, s.w AS w" + cypher: "MATCH (a:Node {id: 'a'}), (b:Node {id: 'b'}) MERGE (a)-[r:U {id: 'u1'}]->(b) SET r.w = 5 WITH count(*) AS merged MATCH ()-[s:U]->() RETURN merged, s.id AS id, s.w AS w" + expect: + rows: + - [2, u1, 5] + - [2, u1, 5] + - [2, u2, 1] diff --git a/tests/spec/rosetta/multi_label_patterns.gtest b/tests/spec/rosetta/multi_label_patterns.gtest index 08c582137..e8080ff70 100644 --- a/tests/spec/rosetta/multi_label_patterns.gtest +++ b/tests/spec/rosetta/multi_label_patterns.gtest @@ -58,6 +58,72 @@ tests: rows: - [c] + - name: expand_target_count_keeps_all_labels + setup: + - "INSERT (d:Graph:Directory {id: 'd'})-[:REL]->(:Graph:Concept {id: 'c'})" + - "MATCH (d:Directory) INSERT (d)-[:REL]->(:Graph:Tech {id: 't'})" + variants: + gql: "MATCH (a)-[r]->(n:Graph:Concept) RETURN count(r)" + cypher: "MATCH (a)-[r]->(n:Graph:Concept) RETURN count(r)" + expect: + rows: + - [1] + + - name: typed_expand_target_keeps_all_labels + setup: + - "INSERT (d:Graph:Directory {id: 'd'})-[:`Graph:CONTAINS`]->(:Graph:Concept {id: 'c'})" + - "MATCH (d:Directory) INSERT (d)-[:`Graph:CONTAINS`]->(:Graph:Tech {id: 't'})" + variants: + gql: "MATCH (a)-[r:`Graph:CONTAINS`]->(n:Graph:Concept) RETURN count(r)" + cypher: "MATCH (a)-[r:`Graph:CONTAINS`]->(n:Graph:Concept) RETURN count(r)" + expect: + rows: + - [1] + + - name: source_and_target_keep_all_labels + setup: + - "INSERT (d:Graph:Directory {id: 'd'})-[:REL]->(:Graph:Concept {id: 'c'})" + - "MATCH (d:Directory) INSERT (d)-[:REL]->(:Graph:Tech {id: 't'})" + variants: + gql: "MATCH (a:Graph:Directory)-[r]->(n:Graph:Concept) RETURN count(r)" + cypher: "MATCH (a:Graph:Directory)-[r]->(n:Graph:Concept) RETURN count(r)" + expect: + rows: + - [1] + + - name: incoming_target_keeps_all_labels + setup: + - "INSERT (d:Graph:Directory {id: 'd'})-[:REL]->(:Graph:Concept {id: 'c'})" + - "MATCH (d:Directory) INSERT (d)-[:REL]->(:Graph:Tech {id: 't'})" + variants: + gql: "MATCH (n:Graph:Directory)<-[r]-(a) RETURN count(r)" + cypher: "MATCH (n:Graph:Directory)<-[r]-(a) RETURN count(r)" + expect: + rows: + - [0] + + # A target inside EXISTS needs all its labels too: t is a Tech, not a Concept. + - name: exists_subquery_target_keeps_all_labels + setup: + - "INSERT (d:Graph:Directory {id: 'd'})-[:REL]->(:Graph:Concept {id: 'c'})" + - "MATCH (d:Directory) INSERT (d)-[:REL]->(:Graph:Tech {id: 't'})" + variants: + gql: "MATCH (d:Graph:Directory) WHERE EXISTS { MATCH (d)-[:REL]->(n:Graph:Concept) WHERE n.id = 't' } RETURN d.id" + cypher: "MATCH (d:Graph:Directory) WHERE EXISTS { MATCH (d)-[:REL]->(n:Graph:Concept) WHERE n.id = 't' } RETURN d.id" + expect: + empty: true + + - name: exists_subquery_finds_a_target_with_all_labels + setup: + - "INSERT (d:Graph:Directory {id: 'd'})-[:REL]->(:Graph:Concept {id: 'c'})" + - "MATCH (d:Directory) INSERT (d)-[:REL]->(:Graph:Tech {id: 't'})" + variants: + gql: "MATCH (d:Graph:Directory) WHERE EXISTS { MATCH (d)-[:REL]->(n:Graph:Concept) WHERE n.id = 'c' } RETURN d.id" + cypher: "MATCH (d:Graph:Directory) WHERE EXISTS { MATCH (d)-[:REL]->(n:Graph:Concept) WHERE n.id = 'c' } RETURN d.id" + expect: + rows: + - [d] + - name: variable_length_target_keeps_all_labels setup: - "INSERT (d:Graph:Directory {id: 'd'})-[:REL]->(:Graph:Concept {id: 'c'})" diff --git a/tests/spec/rosetta/node_and_edge_variables.gtest b/tests/spec/rosetta/node_and_edge_variables.gtest new file mode 100644 index 000000000..750cafb6f --- /dev/null +++ b/tests/spec/rosetta/node_and_edge_variables.gtest @@ -0,0 +1,123 @@ +# Rosetta: a variable is a node or an edge, and keys() of both +# +# keys() lists the property keys of an edge as it does for a node, sorted (as +# properties() returns them), so the order is the same on every run. A variable +# bound to a node cannot later be used as an edge, nor an edge as a node, nor a +# value as either: the query fails with an error instead of matching by a +# coincidence of IDs. A variable imported into CALL keeps what it is. +# +# Dataset `entity_kinds` (tests/spec/datasets/entity_kinds.setup): KNOWS edges +# have `since` and `w`, LIVES_IN edges `years` and `w`; Alix has `name`, `age` +# and `w`. + +meta: + model: lpg + section: rosetta + title: A variable is a node or an edge + dataset: entity_kinds + requires: [cypher] + +tests: + + # --------------------------------------------------------------------------- + # keys() + # --------------------------------------------------------------------------- + + - name: keys_of_an_edge + variants: + gql: "MATCH ()-[r:KNOWS {w: 1}]->() RETURN keys(r) AS k" + cypher: "MATCH ()-[r:KNOWS {w: 1}]->() RETURN keys(r) AS k" + expect: + rows: + - ["[since, w]"] + + - name: keys_of_an_unwound_edge + variants: + gql: "MATCH ()-[r:LIVES_IN {w: 6}]->() WITH collect(r) AS rs UNWIND rs AS e RETURN keys(e) AS k" + cypher: "MATCH ()-[r:LIVES_IN {w: 6}]->() WITH collect(r) AS rs UNWIND rs AS e RETURN keys(e) AS k" + expect: + rows: + - ["[w, years]"] + + # The five KNOWS edges have `since`; the LIVES_IN edges do not. + - name: edges_filtered_by_their_keys + variants: + gql: "MATCH ()-[r]->() WHERE 'since' IN keys(r) RETURN count(*) AS n" + cypher: "MATCH ()-[r]->() WHERE 'since' IN keys(r) RETURN count(*) AS n" + expect: + rows: + - [5] + + - name: keys_of_a_node + variants: + gql: "MATCH (a:Person {name: 'Alix'}) RETURN keys(a) AS k" + cypher: "MATCH (a:Person {name: 'Alix'}) RETURN keys(a) AS k" + expect: + rows: + - ["[age, name, w]"] + + # --------------------------------------------------------------------------- + # A node, an edge or a value, never two of them + # --------------------------------------------------------------------------- + + - name: a_node_used_as_an_edge_in_a_later_match + variants: + gql: "MATCH (r) MATCH ()-[r]->() RETURN count(*) AS c" + cypher: "MATCH (r) MATCH ()-[r]->() RETURN count(*) AS c" + expect: + error: "cannot also be an edge" + + - name: an_edge_used_as_a_node_in_a_later_match + variants: + gql: "MATCH ()-[r]->() MATCH (r)-[]->() RETURN count(*) AS c" + cypher: "MATCH ()-[r]->() MATCH (r)-->() RETURN count(*) AS c" + expect: + error: "cannot also be a node" + + - name: one_name_for_a_node_and_an_edge_in_one_pattern + variants: + gql: "MATCH (a)-[a]->(b) RETURN count(*) AS c" + cypher: "MATCH (a)-[a]->(b) RETURN count(*) AS c" + expect: + error: "cannot also be an edge" + + - name: one_name_for_an_edge_and_its_end_node + variants: + gql: "MATCH (a)-[r]->(r) RETURN count(*) AS c" + cypher: "MATCH (a)-[r]->(r) RETURN count(*) AS c" + expect: + error: "cannot also be a node" + + # Cypher only: GQL has no WITH of a literal before a MATCH that reaches here + # (LET and FOR bind values of any kind). + - name: a_value_used_as_an_edge + variants: + cypher: "WITH 1 AS r MATCH ()-[r]->() RETURN count(*) AS c" + expect: + error: "cannot be an edge" + + # Cypher only: GQL's CALL does not import outer variables yet. + - name: a_node_imported_into_call_used_as_an_edge + variants: + cypher: "MATCH (a:Person) CALL { WITH a MATCH ()-[a]->(b) RETURN b } RETURN count(*) AS c" + expect: + error: "cannot also be an edge" + + # Each KNOWS edge, imported into CALL, is matched again as an edge. + - name: an_edge_imported_into_call_stays_an_edge + variants: + cypher: "MATCH ()-[r:KNOWS]->() CALL { WITH r MATCH (x)-[r]->(y) RETURN x.name AS xn } RETURN count(*) AS c" + expect: + rows: + - [5] + + # The KNOWS triangle Alix, Gus, Vincent: a node bound again as a node. + - name: a_node_bound_again_as_a_node + variants: + gql: "MATCH (a)-[:KNOWS]->(b)-[:KNOWS]->(c)-[:KNOWS]->(a) RETURN a.name AS a" + cypher: "MATCH (a)-[:KNOWS]->(b)-[:KNOWS]->(c)-[:KNOWS]->(a) RETURN a.name AS a" + expect: + rows: + - [Alix] + - [Gus] + - [Vincent] diff --git a/tests/spec/rosetta/optional_match_conditions.gtest b/tests/spec/rosetta/optional_match_conditions.gtest new file mode 100644 index 000000000..d2ef680e9 --- /dev/null +++ b/tests/spec/rosetta/optional_match_conditions.gtest @@ -0,0 +1,143 @@ +# Rosetta: an OPTIONAL MATCH condition that reads an earlier variable +# +# A condition on an OPTIONAL MATCH (a WHERE after it, or GQL's WHERE inside +# the pattern) decides which matches the optional part keeps; a row none of +# whose matches pass it keeps nulls. When the condition reads a variable bound +# before the OPTIONAL MATCH (here `a`), the trailing WHERE dropped such rows, +# and the inline GQL form read `a` as null and kept no match at all. +# +# Dataset `entity_kinds` (tests/spec/datasets/entity_kinds.setup): Alix (30), +# Gus (25), Vincent (40), Jules (35) and Mia (28); KNOWS Alix->Gus, +# Gus->Vincent, Vincent->Alix, Jules->Mia and Alix->Jules. + +meta: + model: lpg + section: rosetta + title: OPTIONAL MATCH conditions on earlier variables + dataset: entity_kinds + requires: [cypher] + +tests: + + # Whom Alix's friends know that is older than Alix: Gus knows Vincent (40); + # Jules knows only Mia (28), so Jules keeps a null. + - name: a_trailing_where_reads_an_earlier_variable + variants: + gql: "MATCH (a:Person {name: 'Alix'})-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c) WHERE c.age > a.age RETURN b.name AS b, c.name AS c" + cypher: "MATCH (a:Person {name: 'Alix'})-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c) WHERE c.age > a.age RETURN b.name AS b, c.name AS c" + expect: + rows: + - [Gus, Vincent] + - [Jules, null] + + - name: an_inline_where_reads_an_earlier_variable + variants: + gql: "MATCH (a:Person {name: 'Alix'})-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c WHERE c.age > a.age) RETURN b.name AS b, c.name AS c" + expect: + rows: + - [Gus, Vincent] + - [Jules, null] + + # The condition sits in the second pattern of the OPTIONAL MATCH: Gus knows + # Vincent, who is older than Alix (and lives nowhere); Jules knows Mia, who + # is younger and lives in Paris. + - name: an_inline_where_in_a_second_pattern + variants: + gql: "MATCH (a:Person {name: 'Alix'})-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c), (c)-[:LIVES_IN]->(y WHERE c.age < a.age) RETURN b.name AS b, y.name AS y" + expect: + rows: + - [Gus, null] + - [Jules, Paris] + + # Per person: the rows of the optional part (one per friend, kept with a + # null) and the matches that pass the condition. + - name: the_condition_in_a_subquery + variants: + gql: "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c) WHERE c.age > a.age RETURN count(*) AS rows, count(c) AS found } RETURN a.name AS a, rows, found" + cypher: "MATCH (a:Person) CALL (a) { MATCH (a)-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c) WHERE c.age > a.age RETURN count(*) AS rows, count(c) AS found } RETURN a.name AS a, rows, found" + expect: + rows: + - [Alix, 2, 1] + - [Gus, 1, 1] + - [Vincent, 1, 0] + - [Jules, 1, 0] + - [Mia, 0, 0] + + # A condition that reads only the optional part's own variables works. + - name: a_condition_on_the_optional_part_only + variants: + gql: "MATCH (a:Person {name: 'Alix'})-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c) WHERE c.age > 30 RETURN b.name AS b, c.name AS c" + cypher: "MATCH (a:Person {name: 'Alix'})-[:KNOWS]->(b) OPTIONAL MATCH (b)-[:KNOWS]->(c) WHERE c.age > 30 RETURN b.name AS b, c.name AS c" + expect: + rows: + - [Gus, Vincent] + - [Jules, null] + + # --------------------------------------------------------------------------- + # An OPTIONAL MATCH that comes first: one row, with nulls when nothing matches + # --------------------------------------------------------------------------- + + - name: a_leading_optional_match_without_a_match + variants: + gql: "OPTIONAL MATCH (x:Robot) RETURN x.name AS x" + cypher: "OPTIONAL MATCH (x:Robot) RETURN x.name AS x" + expect: + rows: + - [null] + + - name: a_leading_optional_match_with_a_match + variants: + gql: "OPTIONAL MATCH (p:Person {name: 'Alix'}) RETURN p.name AS p" + cypher: "OPTIONAL MATCH (p:Person {name: 'Alix'}) RETURN p.name AS p" + expect: + rows: + - [Alix] + + # In a subquery the OPTIONAL MATCH starts from the outer row: Mia knows + # nobody, so her optional part is one row with a null. + - name: a_value_subquery_that_starts_with_optional_match + variants: + gql: "MATCH (p:Person) RETURN p.name AS p, VALUE { OPTIONAL MATCH (p)-[:KNOWS]->(f) RETURN count(f) } AS friends" + expect: + rows: + - [Alix, 2] + - [Gus, 1] + - [Vincent, 1] + - [Jules, 1] + - [Mia, 0] + + - name: a_count_subquery_that_starts_with_optional_match + variants: + gql: "MATCH (p:Person) RETURN p.name AS p, COUNT { OPTIONAL MATCH (p)-[:KNOWS]->(f) } AS n" + cypher: "MATCH (p:Person) RETURN p.name AS p, COUNT { OPTIONAL MATCH (p)-[:KNOWS]->(f) } AS n" + expect: + rows: + - [Alix, 2] + - [Gus, 1] + - [Vincent, 1] + - [Jules, 1] + - [Mia, 1] + + - name: an_exists_subquery_that_starts_with_optional_match + variants: + gql: "MATCH (p:Person) WHERE EXISTS { OPTIONAL MATCH (p)-[:LIVES_IN]->(c) } RETURN p.name AS p" + cypher: "MATCH (p:Person) WHERE EXISTS { OPTIONAL MATCH (p)-[:LIVES_IN]->(c) } RETURN p.name AS p" + expect: + rows: + - [Alix] + - [Gus] + - [Vincent] + - [Jules] + - [Mia] + + - name: a_call_subquery_that_starts_with_optional_match + variants: + gql: "MATCH (p:Person) CALL (p) { OPTIONAL MATCH (p)-[:LIVES_IN]->(c) RETURN c.name AS city } RETURN p.name AS p, city" + cypher: "MATCH (p:Person) CALL (p) { OPTIONAL MATCH (p)-[:LIVES_IN]->(c) RETURN c.name AS city } RETURN p.name AS p, city" + expect: + rows: + - [Alix, Amsterdam] + - [Gus, Berlin] + - [Vincent, null] + - [Jules, null] + - [Mia, Paris] diff --git a/tests/spec/rosetta/order_by_mixed_values.gtest b/tests/spec/rosetta/order_by_mixed_values.gtest new file mode 100644 index 000000000..79aabe024 --- /dev/null +++ b/tests/spec/rosetta/order_by_mixed_values.gtest @@ -0,0 +1,300 @@ +# Rosetta: ORDER BY over values of different types +# +# ORDER BY has one total order over every value, the one openCypher defines: +# maps, lists, paths, temporal values, strings, booleans, numbers, then null. +# Integers and floats compare as numbers, with NaN after infinity; lists +# compare element by element and maps by size, keys and values. Nulls go +# last ascending and first descending, unless NULLS FIRST or NULLS LAST (GQL) +# says otherwise, in either direction. + +meta: + model: lpg + section: rosetta + title: ORDER BY over values of different types + dataset: empty + requires: [cypher] + +tests: + + # --------------------------------------------------------------------------- + # One order across types + # --------------------------------------------------------------------------- + + - name: mixed_types_in_one_order + variants: + gql: "UNWIND [3, 'a', 2.5, true, null, [1], {k: 1}] AS x RETURN x ORDER BY x" + cypher: "UNWIND [3, 'a', 2.5, true, null, [1], {k: 1}] AS x RETURN x ORDER BY x" + expect: + ordered: true + rows: + - ["{k: 1}"] + - ["[1]"] + - [a] + - [true] + - [2.5] + - [3] + - [null] + + - name: mixed_types_descending + variants: + gql: "UNWIND [3, 'a', 2.5, true, null, [1], {k: 1}] AS x RETURN x ORDER BY x DESC" + cypher: "UNWIND [3, 'a', 2.5, true, null, [1], {k: 1}] AS x RETURN x ORDER BY x DESC" + expect: + ordered: true + rows: + - [null] + - [3] + - [2.5] + - [true] + - [a] + - ["[1]"] + - ["{k: 1}"] + + - name: the_order_does_not_depend_on_the_input_order + variants: + gql: "UNWIND [2.5, 'a', 3] AS x RETURN x ORDER BY x" + cypher: "UNWIND [2.5, 'a', 3] AS x RETURN x ORDER BY x" + expect: + ordered: true + rows: + - [a] + - [2.5] + - [3] + + - name: the_order_does_not_depend_on_the_input_order_reversed + variants: + gql: "UNWIND [3, 'a', 2.5] AS x RETURN x ORDER BY x" + cypher: "UNWIND [3, 'a', 2.5] AS x RETURN x ORDER BY x" + expect: + ordered: true + rows: + - [a] + - [2.5] + - [3] + + # Comparing values of different types (and NaN) as equal was not a total + # order: the sort failed on this input with "user-provided comparison + # function does not correctly implement a total order". + - name: mixed_values_with_nan_sort_without_failing + variants: + gql: "UNWIND [9, 0.0 / 0.0, false, 0.0 / 0.0, -3.8, 'e', -11, 19, -15, -2, -17, -8.2, 2, -13.5, -10, true, 18.5, 15, 6, -2, 'e'] AS x RETURN x ORDER BY x" + cypher: "UNWIND [9, 0.0 / 0.0, false, 0.0 / 0.0, -3.8, 'e', -11, 19, -15, -2, -17, -8.2, 2, -13.5, -10, true, 18.5, 15, 6, -2, 'e'] AS x RETURN x ORDER BY x" + expect: + ordered: true + rows: + - [e] + - [e] + - [false] + - [true] + - [-17] + - [-15] + - [-13.5] + - [-11] + - [-10] + - [-8.2] + - [-3.8] + - [-2] + - [-2] + - [2] + - [6] + - [9] + - [15] + - [18.5] + - [19] + - [NaN] + - [NaN] + + - name: property_values_of_different_types + setup: + - "INSERT (:N {val: 1}), (:N {val: 'hello'}), (:N {val: true}), (:N {val: 2.5}), (:N)" + variants: + gql: "MATCH (n:N) RETURN n.val AS v ORDER BY v" + cypher: "MATCH (n:N) RETURN n.val AS v ORDER BY v" + expect: + ordered: true + rows: + - [hello] + - [true] + - [1] + - [2.5] + - [null] + + # --------------------------------------------------------------------------- + # Within a type + # --------------------------------------------------------------------------- + + - name: integers_and_floats_order_as_numbers + variants: + gql: "UNWIND [3, 2.5, -1, 2, 1.5, 0, -0.5] AS x RETURN x ORDER BY x" + cypher: "UNWIND [3, 2.5, -1, 2, 1.5, 0, -0.5] AS x RETURN x ORDER BY x" + expect: + ordered: true + rows: + - [-1] + - [-0.5] + - [0] + - [1.5] + - [2] + - [2.5] + - [3] + + - name: nan_sorts_after_infinity + variants: + gql: "UNWIND [1, 0.0 / 0.0, -1, 1.0 / 0.0, -1.0 / 0.0] AS x RETURN x ORDER BY x" + cypher: "UNWIND [1, 0.0 / 0.0, -1, 1.0 / 0.0, -1.0 / 0.0] AS x RETURN x ORDER BY x" + expect: + ordered: true + rows: + - [-Infinity] + - [-1] + - [1] + - [Infinity] + - [NaN] + + # Percentiles sort their values with the same number order. Sorting NaN as + # equal to everything made the sort fail on this input. + - name: percentiles_over_values_with_nan + variants: + gql: "UNWIND [0.0 / 0.0, 0.0 / 0.0, 0.0 / 0.0, 2, -9, -6, -2, 6, 4, -0.7, 9, 0.0 / 0.0, 1, 0.0 / 0.0, 0.0 / 0.0, -7, -5.2, -7.6, 0.0 / 0.0, 4.8, -7.2] AS x RETURN percentileDisc(x, 0.25) AS d, percentileCont(x, 0.25) AS c" + cypher: "UNWIND [0.0 / 0.0, 0.0 / 0.0, 0.0 / 0.0, 2, -9, -6, -2, 6, 4, -0.7, 9, 0.0 / 0.0, 1, 0.0 / 0.0, 0.0 / 0.0, -7, -5.2, -7.6, 0.0 / 0.0, 4.8, -7.2] AS x RETURN percentileDisc(x, 0.25) AS d, percentileCont(x, 0.25) AS c" + expect: + rows: + - [-5.2, -5.2] + + - name: lists_order_element_by_element + variants: + gql: "UNWIND [[1, 2], [1], [1, null], [null], [], [0, 'z']] AS x RETURN x ORDER BY x" + cypher: "UNWIND [[1, 2], [1], [1, null], [null], [], [0, 'z']] AS x RETURN x ORDER BY x" + expect: + ordered: true + rows: + - ["[]"] + - ["[0, z]"] + - ["[1]"] + - ["[1, 2]"] + - ["[1, null]"] + - ["[null]"] + + - name: maps_order_by_size_then_keys_then_values + variants: + gql: "UNWIND [{b: 1}, {a: 2}, {a: 1}, {a: 1, b: 1}] AS m RETURN m ORDER BY m" + cypher: "UNWIND [{b: 1}, {a: 2}, {a: 1}, {a: 1, b: 1}] AS m RETURN m ORDER BY m" + expect: + ordered: true + rows: + - ["{a: 1}"] + - ["{a: 2}"] + - ["{b: 1}"] + - ["{a: 1, b: 1}"] + + # --------------------------------------------------------------------------- + # Many rows, across row batches (sort and top-K) + # --------------------------------------------------------------------------- + + - name: mixed_values_across_row_batches_top_k + variants: + gql: "UNWIND range(1, 3000) AS i RETURN CASE WHEN i % 2 = 0 THEN i ELSE toString(i) END AS v ORDER BY v DESC LIMIT 3" + cypher: "UNWIND range(1, 3000) AS i RETURN CASE WHEN i % 2 = 0 THEN i ELSE toString(i) END AS v ORDER BY v DESC LIMIT 3" + expect: + ordered: true + rows: + - [3000] + - [2998] + - [2996] + + - name: mixed_values_across_row_batches_sorted + variants: + gql: "UNWIND range(1, 3000) AS i RETURN CASE WHEN i % 2 = 0 THEN i ELSE toString(i) END AS v ORDER BY v SKIP 1500 LIMIT 2" + cypher: "UNWIND range(1, 3000) AS i RETURN CASE WHEN i % 2 = 0 THEN i ELSE toString(i) END AS v ORDER BY v SKIP 1500 LIMIT 2" + expect: + ordered: true + rows: + - [2] + - [4] + + # --------------------------------------------------------------------------- + # Where nulls go + # --------------------------------------------------------------------------- + + - name: nulls_last_by_default_ascending + variants: + gql: "UNWIND [1, null, 3, 2] AS x RETURN x ORDER BY x" + cypher: "UNWIND [1, null, 3, 2] AS x RETURN x ORDER BY x" + expect: + ordered: true + rows: + - [1] + - [2] + - [3] + - [null] + + - name: nulls_first_by_default_descending + variants: + gql: "UNWIND [1, null, 3, 2] AS x RETURN x ORDER BY x DESC" + cypher: "UNWIND [1, null, 3, 2] AS x RETURN x ORDER BY x DESC" + expect: + ordered: true + rows: + - [null] + - [3] + - [2] + - [1] + + - name: nulls_last_descending + query: "UNWIND [1, null, 3, 2] AS x RETURN x ORDER BY x DESC NULLS LAST" + expect: + ordered: true + rows: + - [3] + - [2] + - [1] + - [null] + + - name: nulls_first_descending + query: "UNWIND [1, null, 3, 2] AS x RETURN x ORDER BY x DESC NULLS FIRST" + expect: + ordered: true + rows: + - [null] + - [3] + - [2] + - [1] + + - name: nulls_first_ascending + query: "UNWIND [1, null, 3, 2] AS x RETURN x ORDER BY x ASC NULLS FIRST" + expect: + ordered: true + rows: + - [null] + - [1] + - [2] + - [3] + + - name: nulls_last_descending_top_k + query: "UNWIND [1, null, 3, 2] AS x RETURN x ORDER BY x DESC NULLS LAST LIMIT 2" + expect: + ordered: true + rows: + - [3] + - [2] + + - name: nulls_last_descending_on_a_property + setup: + - "INSERT (:T {x: 1}), (:T {x: 3}), (:T), (:T {x: 2})" + query: "MATCH (t:T) RETURN t.x AS x ORDER BY x DESC NULLS LAST" + expect: + ordered: true + rows: + - [3] + - [2] + - [1] + - [null] + + - name: nulls_last_on_the_second_key_descending + query: "UNWIND [[1, null], [1, 2], [0, 5]] AS p RETURN p[0] AS a, p[1] AS b ORDER BY a DESC, b DESC NULLS LAST" + expect: + ordered: true + rows: + - [1, 2] + - [1, null] + - [0, 5] diff --git a/tests/spec/rosetta/parameters.gtest b/tests/spec/rosetta/parameters.gtest new file mode 100644 index 000000000..cfa9ad5f1 --- /dev/null +++ b/tests/spec/rosetta/parameters.gtest @@ -0,0 +1,94 @@ +# Rosetta: parameters +# +# A statement that uses a parameter nobody supplied fails with "Missing +# parameter" before it writes anything, in every path (a write used to store +# the text "$e"). An unaliased column that reads a parameter is named after +# the query text ($x), not after the value that replaces it. +# +# Dataset `entity_kinds` (tests/spec/datasets/entity_kinds.setup): Alix (30), +# Gus (25), Vincent (40), Jules (35) and Mia (28). + +meta: + model: lpg + section: rosetta + title: Parameters + dataset: entity_kinds + requires: [cypher] + +tests: + + # --------------------------------------------------------------------------- + # A parameter nobody supplied + # --------------------------------------------------------------------------- + + - name: unsupplied_parameter_in_a_write + variants: + gql: "INSERT (:P {e: $e}) RETURN 1 AS one" + cypher: "CREATE (:P {e: $e}) RETURN 1 AS one" + expect: + error: "Missing parameter: $e" + + - name: unsupplied_parameter_in_return + variants: + gql: "RETURN $x AS x" + cypher: "RETURN $x AS x" + expect: + error: "Missing parameter: $x" + + - name: unsupplied_parameter_in_where + variants: + gql: "MATCH (p:Person) WHERE p.age = $age RETURN p.name AS n" + cypher: "MATCH (p:Person) WHERE p.age = $age RETURN p.name AS n" + expect: + error: "Missing parameter: $age" + + # --------------------------------------------------------------------------- + # Column names + # --------------------------------------------------------------------------- + + - name: an_unaliased_parameter_is_named_after_it + params: + x: 5 + variants: + gql: "RETURN $x" + cypher: "RETURN $x" + expect: + columns: ["$x"] + rows: + - [5] + + - name: an_unaliased_expression_with_a_parameter_is_named_after_its_text + params: + x: 5 + variants: + gql: "MATCH (p:Person) WHERE p.age > 35 RETURN p.age + $x" + cypher: "MATCH (p:Person) WHERE p.age > 35 RETURN p.age + $x" + expect: + columns: ["p.age + $x"] + rows: + - [45] + + # The ORDER BY repeats the parameter expression: it still sorts by it. + - name: an_unaliased_parameter_expression_in_order_by + params: + x: 5 + variants: + gql: "MATCH (p:Person) WHERE p.age < 30 RETURN p.age + $x ORDER BY p.age + $x DESC" + cypher: "MATCH (p:Person) WHERE p.age < 30 RETURN p.age + $x ORDER BY p.age + $x DESC" + expect: + ordered: true + columns: ["p.age + $x"] + rows: + - [33] + - [30] + + - name: an_aliased_parameter_keeps_its_alias + params: + x: 5 + variants: + gql: "RETURN $x AS five" + cypher: "RETURN $x AS five" + expect: + columns: [five] + rows: + - [5] diff --git a/tests/spec/rosetta/return_star_ordered.gtest b/tests/spec/rosetta/return_star_ordered.gtest new file mode 100644 index 000000000..613ddb5e1 --- /dev/null +++ b/tests/spec/rosetta/return_star_ordered.gtest @@ -0,0 +1,121 @@ +# Rosetta: RETURN * with ORDER BY, SKIP, LIMIT and DISTINCT +# +# RETURN * returns every variable in scope. ORDER BY may sort by one of them, +# by a property of one or by an expression, and adds no column for it. + +meta: + model: lpg + section: rosetta + title: RETURN * with ORDER BY, SKIP, LIMIT and DISTINCT + dataset: empty + requires: [cypher] + +tests: + + - name: ordered_by_a_returned_variable + variants: + gql: "UNWIND [3, 1, 2] AS x RETURN * ORDER BY x" + cypher: "UNWIND [3, 1, 2] AS x RETURN * ORDER BY x" + expect: + ordered: true + columns: [x] + rows: + - [1] + - [2] + - [3] + + - name: ordered_by_an_expression + variants: + gql: "UNWIND [3, 1, 2] AS x RETURN * ORDER BY -x" + cypher: "UNWIND [3, 1, 2] AS x RETURN * ORDER BY -x" + expect: + ordered: true + columns: [x] + rows: + - [3] + - [2] + - [1] + + - name: ordered_by_a_property + variants: + gql: "UNWIND [{k: 2, v: 'b'}, {k: 1, v: 'a'}] AS m RETURN * ORDER BY m.k" + cypher: "UNWIND [{k: 2, v: 'b'}, {k: 1, v: 'a'}] AS m RETURN * ORDER BY m.k" + expect: + ordered: true + columns: [m] + rows: + - ["{k: 1, v: a}"] + - ["{k: 2, v: b}"] + + - name: ordered_with_limit + variants: + gql: "UNWIND [3, 1, 2] AS x RETURN * ORDER BY x DESC LIMIT 2" + cypher: "UNWIND [3, 1, 2] AS x RETURN * ORDER BY x DESC LIMIT 2" + expect: + ordered: true + columns: [x] + rows: + - [3] + - [2] + + - name: ordered_with_skip + variants: + gql: "UNWIND [3, 1, 2] AS x RETURN * ORDER BY x SKIP 1" + cypher: "UNWIND [3, 1, 2] AS x RETURN * ORDER BY x SKIP 1" + expect: + ordered: true + columns: [x] + rows: + - [2] + - [3] + + - name: distinct_ordered + variants: + gql: "UNWIND [2, 1, 2, 3] AS x RETURN DISTINCT * ORDER BY x" + cypher: "UNWIND [2, 1, 2, 3] AS x RETURN DISTINCT * ORDER BY x" + expect: + ordered: true + columns: [x] + rows: + - [1] + - [2] + - [3] + + - name: two_columns_ordered_by_one + variants: + gql: "UNWIND [3, 1, 2] AS x WITH x, x * 10 AS y RETURN * ORDER BY y DESC" + cypher: "UNWIND [3, 1, 2] AS x WITH x, x * 10 AS y RETURN * ORDER BY y DESC" + expect: + ordered: true + columns: [x, y] + rows: + - [3, 30] + - [2, 20] + - [1, 10] + + # Nodes and edges print differently in each binding, so this case checks the + # columns and the count; the next one checks the order by name, and + # crates/grafeo-engine/tests/returned_entities.rs compares the entities. + - name: nodes_and_edges_ordered_by_an_edge_property + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris'}), (alix)-[:LIVES_IN {years: 5}]->(ams), (gus)-[:LIVES_IN {years: 3}]->(ber), (mia)-[:LIVES_IN {years: 1}]->(par)" + variants: + gql: "MATCH (p:Person)-[r:LIVES_IN]->(c:City) RETURN * ORDER BY r.years LIMIT 2" + cypher: "MATCH (p:Person)-[r:LIVES_IN]->(c:City) RETURN * ORDER BY r.years LIMIT 2" + expect: + columns: [p, r, c] + count: 2 + + # GQL has no ORDER BY after WITH; RETURN * ... NEXT passes the ordered rows + # on instead. + - name: names_ordered_by_an_edge_property_through_with_star + setup: + - "INSERT (alix:Person {name: 'Alix'}), (gus:Person {name: 'Gus'}), (mia:Person {name: 'Mia'}), (ams:City {name: 'Amsterdam'}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris'}), (alix)-[:LIVES_IN {years: 5}]->(ams), (gus)-[:LIVES_IN {years: 3}]->(ber), (mia)-[:LIVES_IN {years: 1}]->(par)" + variants: + gql: "MATCH (p:Person)-[r:LIVES_IN]->(c:City) RETURN * ORDER BY r.years LIMIT 2 NEXT RETURN p.name AS p, r.years AS years, c.name AS c" + cypher: "MATCH (p:Person)-[r:LIVES_IN]->(c:City) WITH * ORDER BY r.years LIMIT 2 RETURN p.name AS p, r.years AS years, c.name AS c" + expect: + ordered: true + rows: + - [Mia, 1, Paris] + - [Gus, 3, Berlin] diff --git a/tests/spec/rosetta/set_operations_on_entities.gtest b/tests/spec/rosetta/set_operations_on_entities.gtest new file mode 100644 index 000000000..dd10c23d4 --- /dev/null +++ b/tests/spec/rosetta/set_operations_on_entities.gtest @@ -0,0 +1,200 @@ +# Rosetta: set operations and ORDER BY on returned nodes and edges +# +# Every branch of a set operation returns its nodes and edges the same way, +# so the same entity from two branches is one row of a UNION, and EXCEPT and +# INTERSECT compare entities. A column added to sort by a RETURN alias's +# property is not part of the result. +# +# The two sides of a set operation reach the same entity through different +# patterns, so a side that returned it differently would not match. Rows are +# counted here because nodes and edges print differently in each binding; +# crates/grafeo-engine/tests/returned_entities.rs compares the entities +# themselves. EXCEPT, INTERSECT and OTHERWISE are GQL only (Cypher has UNION). +# +# Setup (GQL): (:A {n: 1})-[:K {w: 1}]->(:B {n: 2}), the same with n 3, 4 and +# w 2, and with n 5, 6 and w 3. + +meta: + model: lpg + section: rosetta + title: Set operations and ORDER BY on returned entities + dataset: empty + requires: [cypher] + +tests: + + - name: union_of_one_edge_from_two_branches + setup: + - "INSERT (:A {n: 1})-[:K {w: 1}]->(:B {n: 2})" + - "INSERT (:A {n: 3})-[:K {w: 2}]->(:B {n: 4})" + - "INSERT (:A {n: 5})-[:K {w: 3}]->(:B {n: 6})" + variants: + gql: "MATCH ()-[r]->() WHERE r.w = 1 RETURN r UNION MATCH (:A {n: 1})-[r:K]->(:B) RETURN r" + cypher: "MATCH ()-[r]->() WHERE r.w = 1 RETURN r UNION MATCH (:A {n: 1})-[r:K]->(:B) RETURN r" + expect: + count: 1 + + - name: union_of_one_node_from_two_branches + setup: + - "INSERT (:A {n: 1})-[:K {w: 1}]->(:B {n: 2})" + - "INSERT (:A {n: 3})-[:K {w: 2}]->(:B {n: 4})" + - "INSERT (:A {n: 5})-[:K {w: 3}]->(:B {n: 6})" + variants: + gql: "MATCH (a:A) WHERE a.n = 3 RETURN a UNION MATCH (a:A)-[:K]->(:B {n: 4}) RETURN a" + cypher: "MATCH (a:A) WHERE a.n = 3 RETURN a UNION MATCH (a:A)-[:K]->(:B {n: 4}) RETURN a" + expect: + count: 1 + + - name: except_on_edges + setup: + - "INSERT (:A {n: 1})-[:K {w: 1}]->(:B {n: 2})" + - "INSERT (:A {n: 3})-[:K {w: 2}]->(:B {n: 4})" + - "INSERT (:A {n: 5})-[:K {w: 3}]->(:B {n: 6})" + query: "MATCH ()-[r]->() RETURN r EXCEPT MATCH (:A {n: 3})-[r:K]->() RETURN r" + expect: + count: 2 + + - name: intersect_on_edges + setup: + - "INSERT (:A {n: 1})-[:K {w: 1}]->(:B {n: 2})" + - "INSERT (:A {n: 3})-[:K {w: 2}]->(:B {n: 4})" + - "INSERT (:A {n: 5})-[:K {w: 3}]->(:B {n: 6})" + query: "MATCH ()-[r]->() RETURN r INTERSECT MATCH (:A {n: 3})-[r:K]->() RETURN r" + expect: + count: 1 + + - name: except_on_nodes + setup: + - "INSERT (:A {n: 1})-[:K {w: 1}]->(:B {n: 2})" + - "INSERT (:A {n: 3})-[:K {w: 2}]->(:B {n: 4})" + - "INSERT (:A {n: 5})-[:K {w: 3}]->(:B {n: 6})" + query: "MATCH (a:A) RETURN a EXCEPT MATCH (a:A)-[:K]->(:B {n: 4}) RETURN a" + expect: + count: 2 + + - name: intersect_on_nodes + setup: + - "INSERT (:A {n: 1})-[:K {w: 1}]->(:B {n: 2})" + - "INSERT (:A {n: 3})-[:K {w: 2}]->(:B {n: 4})" + - "INSERT (:A {n: 5})-[:K {w: 3}]->(:B {n: 6})" + query: "MATCH (a:A) RETURN a INTERSECT MATCH (a:A)-[:K]->(:B {n: 4}) RETURN a" + expect: + count: 1 + + - name: a_sort_key_on_an_alias_is_not_a_column + setup: + - "INSERT (:A {n: 1})-[:K {w: 1}]->(:B {n: 2})" + - "INSERT (:A {n: 3})-[:K {w: 2}]->(:B {n: 4})" + - "INSERT (:A {n: 5})-[:K {w: 3}]->(:B {n: 6})" + variants: + gql: "MATCH ()-[r]->() RETURN r AS e ORDER BY e.w" + cypher: "MATCH ()-[r]->() RETURN r AS e ORDER BY e.w" + expect: + columns: [e] + count: 3 + + # --------------------------------------------------------------------------- + # A Person/City graph whose nodes and edges both carry `w` (see + # values_through_ordering.gtest): duplicates, nulls and mixed branches + # --------------------------------------------------------------------------- + + # One row per outgoing edge: Alix 3, Gus 2, Vincent, Jules, Mia; one Alix removed. + - name: except_all_on_nodes_with_duplicates + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + query: "MATCH (a:Person)-[]->(b) RETURN a EXCEPT ALL MATCH (a:Person {name: 'Alix'}) RETURN a" + expect: + count: 7 + + # Each person as often as both sides have them: Alix 2, Gus, Vincent, Jules. + - name: intersect_all_on_nodes_with_duplicates + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + query: "MATCH (a:Person)-[]->(b) RETURN a INTERSECT ALL MATCH (a:Person)-[:KNOWS]->(b) RETURN a" + expect: + count: 5 + + # The KNOWS edges of Gus, Vincent and Jules. + - name: except_on_node_and_edge_pairs + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + query: "MATCH (a:Person)-[r:KNOWS]->(b) RETURN a, r EXCEPT MATCH (a:Person {name: 'Alix'})-[r:KNOWS]->(b) RETURN a, r" + expect: + count: 3 + + # Edges with w below 3 and above 1: all 8 once. + - name: union_of_overlapping_edge_sets + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH (a:Person)-[r]->(b) WHERE r.w < 3 RETURN r UNION MATCH (a:Person)-[r]->(b) WHERE r.w > 1 RETURN r" + cypher: "MATCH (a:Person)-[r]->(b) WHERE r.w < 3 RETURN r UNION MATCH (a:Person)-[r]->(b) WHERE r.w > 1 RETURN r" + expect: + count: 8 + + # One name, three branches: a node, an edge and a node. + - name: union_all_of_a_node_an_edge_and_a_node + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH (x:Person {name: 'Gus'}) RETURN x UNION ALL MATCH ()-[x:KNOWS]->() WHERE x.w = 1 RETURN x UNION ALL MATCH (x:City {name: 'Paris'}) RETURN x" + cypher: "MATCH (x:Person {name: 'Gus'}) RETURN x UNION ALL MATCH ()-[x:KNOWS]->() WHERE x.w = 1 RETURN x UNION ALL MATCH (x:City {name: 'Paris'}) RETURN x" + expect: + count: 3 + + # Each branch reads `w` of its own entities. + - name: union_all_of_node_and_edge_properties + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + variants: + gql: "MATCH (a:Person) RETURN a.w AS w UNION ALL MATCH ()-[r:KNOWS]->() RETURN r.w AS w" + cypher: "MATCH (a:Person) RETURN a.w AS w UNION ALL MATCH ()-[r:KNOWS]->() RETURN r.w AS w" + expect: + rows: + - [100] + - [101] + - [103] + - [null] + - [null] + - [1] + - [2] + - [3] + - [4] + - [5] + + # Persons not older than 29. + - name: except_on_names + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + query: "MATCH (a:Person) RETURN a.name AS n EXCEPT MATCH (a:Person) WHERE a.age > 29 RETURN a.name AS n" + expect: + rows: + - [Gus] + - [Mia] + + # No person is older than 100. + - name: otherwise_takes_the_right_branch_when_the_left_is_empty + setup: + - "INSERT (alix:Person {name: 'Alix', age: 30, w: 100}), (gus:Person {name: 'Gus', age: 25, w: 101}), (vincent:Person {name: 'Vincent', age: 40, w: 103}), (jules:Person {name: 'Jules', age: 35}), (mia:Person {name: 'Mia', age: 28}), (ams:City {name: 'Amsterdam', w: 104}), (ber:City {name: 'Berlin'}), (par:City {name: 'Paris', w: 105}), (alix)-[:KNOWS {since: 2010, w: 1}]->(gus), (gus)-[:KNOWS {since: 2012, w: 2}]->(vincent), (vincent)-[:KNOWS {since: 2015, w: 3}]->(alix), (jules)-[:KNOWS {since: 2020, w: 4}]->(mia), (alix)-[:KNOWS {since: 2018, w: 5}]->(jules), (alix)-[:LIVES_IN {years: 5, w: 6}]->(ams), (gus)-[:LIVES_IN {years: 3, w: 7}]->(ber), (mia)-[:LIVES_IN {years: 1, w: 8}]->(par)" + query: "MATCH (a:Person) WHERE a.age > 100 RETURN a.name AS n OTHERWISE MATCH (c:City) RETURN c.name AS n" + expect: + rows: + - [Amsterdam] + - [Berlin] + - [Paris] + + # Null equals null in a set operation. + - name: except_with_nulls + query: "UNWIND [1, 2, 2, 3, null] AS x RETURN x EXCEPT UNWIND [2, null] AS x RETURN x" + expect: + rows: + - [1] + - [3] + + # 2 appears twice on the left, three times on the right. + - name: intersect_all_with_duplicates + query: "UNWIND [1, 2, 2, 3] AS x RETURN x INTERSECT ALL UNWIND [2, 2, 2, 4] AS x RETURN x" + expect: + rows: + - [2] + - [2] diff --git a/tests/spec/rosetta/values_through_joins_and_writes.gtest b/tests/spec/rosetta/values_through_joins_and_writes.gtest new file mode 100644 index 000000000..0a68182f1 --- /dev/null +++ b/tests/spec/rosetta/values_through_joins_and_writes.gtest @@ -0,0 +1,187 @@ +# Rosetta: values through joins, CALL, UNWIND, grouping and writes +# +# A node or edge that goes through a join, a CALL subquery, UNWIND, a group +# key or a write stays what it is. The setup gives nodes and edges a property +# with the same name (`w`) and overlapping IDs, so one that loses track of what +# it is reads the other entity's `w` (nodes have 100 and up, edges 1 to 8): +# every expected `w` below tells which entity was read. Jules, Mia and Berlin +# have no `w`, so for them the wrong entity shows as a number instead of null. +# +# Dataset `entity_kinds` (tests/spec/datasets/entity_kinds.setup): Alix, Gus, +# Vincent (with w 100, 101, 103), Jules and Mia; Amsterdam (w 104), Berlin, +# Paris (w 105). KNOWS edges w 1 to 5 (Alix->Gus, Gus->Vincent, Vincent->Alix, +# Jules->Mia, Alix->Jules), LIVES_IN edges w 6 to 8 (Alix->Amsterdam, +# Gus->Berlin, Mia->Paris). + +meta: + model: lpg + section: rosetta + title: Values through joins, CALL, UNWIND, grouping and writes + dataset: entity_kinds + requires: [cypher] + +tests: + + # --------------------------------------------------------------------------- + # Joins + # --------------------------------------------------------------------------- + + - name: nodes_without_a_property_after_optional_match + variants: + gql: "MATCH (a:Person) OPTIONAL MATCH (a)-[:LIVES_IN]->(c) RETURN a.name AS a, a.w AS w, c.w AS cw" + cypher: "MATCH (a:Person) OPTIONAL MATCH (a)-[:LIVES_IN]->(c) RETURN a.name AS a, a.w AS w, c.w AS cw" + expect: + rows: + - [Alix, 100, 104] + - [Gus, 101, null] + - [Vincent, 103, null] + - [Jules, null, null] + - [Mia, null, 105] + + - name: a_node_without_a_property_after_a_comma_join + variants: + gql: "MATCH (a:Person)-[:KNOWS]->(b), (b)-[r:LIVES_IN]->(c) RETURN a.name AS a, b.name AS b, b.w AS bw, r.w AS rw" + cypher: "MATCH (a:Person)-[:KNOWS]->(b), (b)-[r:LIVES_IN]->(c) RETURN a.name AS a, b.name AS b, b.w AS bw, r.w AS rw" + expect: + rows: + - [Alix, Gus, 101, 7] + - [Jules, Mia, null, 8] + - [Vincent, Alix, 100, 6] + + # Jules knows Mia, who lives in Paris: Jules passes the semi-join. + - name: a_node_without_a_property_after_a_semi_join + variants: + gql: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(m)-[:LIVES_IN]->() } RETURN a.name AS a, a.w AS w" + cypher: "MATCH (a:Person) WHERE EXISTS { MATCH (a)-[:KNOWS]->(m)-[:LIVES_IN]->() } RETURN a.name AS a, a.w AS w" + expect: + rows: + - [Alix, 100] + - [Vincent, 103] + - [Jules, null] + + - name: nodes_without_a_property_after_an_anti_join + variants: + gql: "MATCH (a:Person) WHERE NOT EXISTS { MATCH (a)-[:KNOWS]->()-[:KNOWS]->() } RETURN a.name AS a, a.w AS w" + cypher: "MATCH (a:Person) WHERE NOT EXISTS { MATCH (a)-[:KNOWS]->()-[:KNOWS]->() } RETURN a.name AS a, a.w AS w" + expect: + rows: + - [Jules, null] + - [Mia, null] + + # --------------------------------------------------------------------------- + # CALL subqueries and UNWIND + # --------------------------------------------------------------------------- + + - name: nodes_after_a_call_subquery_that_imports_them + variants: + gql: "MATCH (a:Person) CALL { WITH a RETURN 1 AS one } RETURN a.name AS a, a.w AS w" + cypher: "MATCH (a:Person) CALL { WITH a RETURN 1 AS one } RETURN a.name AS a, a.w AS w" + expect: + rows: + - [Alix, 100] + - [Gus, 101] + - [Vincent, 103] + - [Jules, null] + - [Mia, null] + + - name: edges_after_a_call_subquery + variants: + gql: "MATCH ()-[r:KNOWS]->() CALL { RETURN 1 AS one } RETURN r.w AS w" + cypher: "MATCH ()-[r:KNOWS]->() CALL { RETURN 1 AS one } RETURN r.w AS w" + expect: + rows: + - [1] + - [2] + - [3] + - [4] + - [5] + + - name: nodes_after_unwind + variants: + gql: "MATCH (a:Person) UNWIND [1, 2] AS k RETURN a.name AS a, a.w AS w, k" + cypher: "MATCH (a:Person) UNWIND [1, 2] AS k RETURN a.name AS a, a.w AS w, k" + expect: + rows: + - [Alix, 100, 1] + - [Alix, 100, 2] + - [Gus, 101, 1] + - [Gus, 101, 2] + - [Vincent, 103, 1] + - [Vincent, 103, 2] + - [Jules, null, 1] + - [Jules, null, 2] + - [Mia, null, 1] + - [Mia, null, 2] + + # --------------------------------------------------------------------------- + # Group keys + # --------------------------------------------------------------------------- + + - name: nodes_as_group_keys + variants: + gql: "MATCH (a:Person)-[r]->() WITH a, count(r) AS n RETURN a.name AS a, a.w AS w, n" + cypher: "MATCH (a:Person)-[r]->() WITH a, count(r) AS n RETURN a.name AS a, a.w AS w, n" + expect: + rows: + - [Alix, 100, 3] + - [Gus, 101, 2] + - [Vincent, 103, 1] + - [Jules, null, 1] + - [Mia, null, 1] + + - name: edges_as_group_keys + variants: + gql: "MATCH ()-[r:KNOWS]->() WITH r, count(*) AS n RETURN r.w AS w, n" + cypher: "MATCH ()-[r:KNOWS]->() WITH r, count(*) AS n RETURN r.w AS w, n" + expect: + rows: + - [1, 1] + - [2, 1] + - [3, 1] + - [4, 1] + - [5, 1] + + # --------------------------------------------------------------------------- + # Writes that pass their input rows on + # --------------------------------------------------------------------------- + + - name: a_node_without_a_property_after_set + variants: + gql: "MATCH (a:Person {name: 'Jules'}) SET a.seen = true RETURN a.w AS w, a.seen AS seen" + cypher: "MATCH (a:Person {name: 'Jules'}) SET a.seen = true RETURN a.w AS w, a.seen AS seen" + expect: + rows: + - [null, true] + + - name: a_node_without_a_property_after_an_insert + variants: + gql: "MATCH (a:Person {name: 'Mia'}) INSERT (a)-[:LIKES]->(:Person {name: 'Hans'}) RETURN a.w AS w" + cypher: "MATCH (a:Person {name: 'Mia'}) CREATE (a)-[:LIKES]->(:Person {name: 'Hans'}) RETURN a.w AS w" + expect: + rows: + - [null] + + - name: a_node_without_a_property_after_merge + variants: + gql: "MATCH (a:Person {name: 'Jules'}) MERGE (a)-[:VISITED]->(c:City {name: 'Prague'}) RETURN a.w AS w, c.name AS c" + cypher: "MATCH (a:Person {name: 'Jules'}) MERGE (a)-[:VISITED]->(c:City {name: 'Prague'}) RETURN a.w AS w, c.name AS c" + expect: + rows: + - [null, Prague] + + # A SET with a computed value works it out in a column of its own first; that + # column must not take the place of what the next write creates. Cypher only: + # GQL does not allow INSERT or MERGE after SET yet (#483). + - name: an_edge_created_after_a_computed_set + variants: + cypher: "MATCH (a:Person {name: 'Alix'}), (b:Person {name: 'Gus'}) SET a.x = b.w + 1 CREATE (a)-[r:LIKES {w: 9}]->(b) RETURN type(r) AS t, a.x AS x, r.w AS rw" + expect: + rows: + - [LIKES, 102, 9] + + - name: a_node_merged_after_a_computed_set + variants: + cypher: "MATCH (a:Person {name: 'Jules'}), (b:Person {name: 'Gus'}) SET a.x = b.w + 99 MERGE (a)-[:VISITED]->(c:City {name: 'Prague'}) RETURN a.x AS x, c.name AS c" + expect: + rows: + - [200, Prague] diff --git a/tests/spec/rosetta/values_through_lists.gtest b/tests/spec/rosetta/values_through_lists.gtest new file mode 100644 index 000000000..58bd31914 --- /dev/null +++ b/tests/spec/rosetta/values_through_lists.gtest @@ -0,0 +1,159 @@ +# Rosetta: nodes and edges in lists, through collect, UNWIND and list functions +# +# A list of nodes or edges stays one: collect of a node or edge, the nodes and +# relationships of a path, and UNWIND of such a list give nodes and edges +# back, so a property read takes the right entity. The dataset gives nodes and +# edges a property with the same name (`w`) and overlapping IDs (nodes have +# 100 and up, edges 1 to 8; Jules, Mia and Berlin have none), so an entity +# read as the other kind shows a value from the wrong range. +# +# Dataset `entity_kinds` (tests/spec/datasets/entity_kinds.setup). + +meta: + model: lpg + section: rosetta + title: Nodes and edges in lists + dataset: entity_kinds + requires: [cypher] + +tests: + + # --------------------------------------------------------------------------- + # collect, then UNWIND or a list comprehension + # --------------------------------------------------------------------------- + + - name: unwind_collected_nodes + variants: + gql: "MATCH (a:Person) WITH collect(a) AS people UNWIND people AS p RETURN p.name AS n, p.w AS w" + cypher: "MATCH (a:Person) WITH collect(a) AS people UNWIND people AS p RETURN p.name AS n, p.w AS w" + expect: + rows: + - [Alix, 100] + - [Gus, 101] + - [Vincent, 103] + - [Jules, null] + - [Mia, null] + + - name: unwind_collected_edges + variants: + gql: "MATCH ()-[r:KNOWS]->() WITH collect(r) AS rs UNWIND rs AS e RETURN e.w AS w" + cypher: "MATCH ()-[r:KNOWS]->() WITH collect(r) AS rs UNWIND rs AS e RETURN e.w AS w" + expect: + rows: + - [1] + - [2] + - [3] + - [4] + - [5] + + - name: type_and_property_of_unwound_edges + variants: + gql: "MATCH ()-[r:LIVES_IN]->() WITH collect(r) AS rs UNWIND rs AS e RETURN type(e) AS t, e.w AS w" + cypher: "MATCH ()-[r:LIVES_IN]->() WITH collect(r) AS rs UNWIND rs AS e RETURN type(e) AS t, e.w AS w" + expect: + rows: + - [LIVES_IN, 6] + - [LIVES_IN, 7] + - [LIVES_IN, 8] + + # Every KNOWS edge has a `w` below 10; read as nodes they would not. + - name: comprehension_over_collected_edges + variants: + gql: "MATCH ()-[r:KNOWS]->() WITH collect(r) AS rs RETURN size([x IN rs WHERE x.w < 10]) AS n" + cypher: "MATCH ()-[r:KNOWS]->() WITH collect(r) AS rs RETURN size([x IN rs WHERE x.w < 10]) AS n" + expect: + rows: + - [5] + + # Jules and Mia have no `w`; read as edges they would. + - name: comprehension_over_collected_nodes + variants: + gql: "MATCH (a:Person) WITH collect(a) AS people RETURN size([x IN people WHERE x.w IS NULL]) AS n" + cypher: "MATCH (a:Person) WITH collect(a) AS people RETURN size([x IN people WHERE x.w IS NULL]) AS n" + expect: + rows: + - [2] + + # The edge collected, unwound and matched again: Alix, Gus and Mia to the + # city each lives in. + - name: match_through_an_unwound_edge + variants: + gql: "MATCH ()-[r:LIVES_IN]->() WITH collect(r) AS rs UNWIND rs AS e MATCH (x)-[e]->(y) RETURN x.name AS x, y.name AS y" + cypher: "MATCH ()-[r:LIVES_IN]->() WITH collect(r) AS rs UNWIND rs AS e MATCH (x)-[e]->(y) RETURN x.name AS x, y.name AS y" + expect: + rows: + - [Alix, Amsterdam] + - [Gus, Berlin] + - [Mia, Paris] + + # --------------------------------------------------------------------------- + # A property of one item of such a list + # --------------------------------------------------------------------------- + + - name: property_of_one_collected_edge + variants: + gql: "MATCH ()-[r:KNOWS {w: 5}]->() WITH collect(r) AS rs RETURN rs[0].w AS w, rs[-1].since AS since" + cypher: "MATCH ()-[r:KNOWS {w: 5}]->() WITH collect(r) AS rs RETURN rs[0].w AS w, rs[-1].since AS since" + expect: + rows: + - [5, 2018] + + # Jules has no `w`; read as the edge with the same ID, Jules would have 4. + - name: property_of_one_collected_node + variants: + gql: "MATCH (a:Person {name: 'Jules'}) WITH collect(a) AS people RETURN people[0].w AS w, people[0].name AS n" + cypher: "MATCH (a:Person {name: 'Jules'}) WITH collect(a) AS people RETURN people[0].w AS w, people[0].name AS n" + expect: + rows: + - [null, Jules] + + # The first edge of each two-hop walk from Alix: to Gus (w 1) and to Jules (w 5). + - name: property_of_one_edge_of_a_variable_length_pattern + variants: + gql: "MATCH (:Person {name: 'Alix'})-[r:KNOWS]->{2}() RETURN r[0].w AS w" + cypher: "MATCH (:Person {name: 'Alix'})-[r:KNOWS*2]->() RETURN r[0].w AS w" + expect: + rows: + - [1] + - [5] + + # --------------------------------------------------------------------------- + # The nodes and relationships of a path, and other lists computed per row + # --------------------------------------------------------------------------- + + # Alix->Gus->Vincent (w 1, 2) and Alix->Jules->Mia (w 5, 4). + - name: unwind_relationships_of_paths + variants: + gql: "MATCH p = (:Person {name: 'Alix'})-[:KNOWS*2]->() UNWIND relationships(p) AS e RETURN e.w AS w" + cypher: "MATCH p = (:Person {name: 'Alix'})-[:KNOWS*2]->() UNWIND relationships(p) AS e RETURN e.w AS w" + expect: + rows: + - [1] + - [2] + - [5] + - [4] + + - name: unwind_nodes_of_a_path + variants: + gql: "MATCH p = (:Person {name: 'Jules'})-[:KNOWS]->() UNWIND nodes(p) AS n RETURN n.name AS n, n.w AS w" + cypher: "MATCH p = (:Person {name: 'Jules'})-[:KNOWS]->() UNWIND nodes(p) AS n RETURN n.name AS n, n.w AS w" + expect: + rows: + - [Jules, null] + - [Mia, null] + + # A list built from each row: 1 for Alix (w 100), 1 and 2 for Gus (w 101), + # 1 to 4 for Vincent (w 103), none for Jules and Mia (no w, so no list). + - name: unwind_a_list_computed_per_row + variants: + gql: "MATCH (a:Person) UNWIND range(1, a.w - 99) AS i RETURN a.name AS a, i" + cypher: "MATCH (a:Person) UNWIND range(1, a.w - 99) AS i RETURN a.name AS a, i" + expect: + rows: + - [Alix, 1] + - [Gus, 1] + - [Gus, 2] + - [Vincent, 1] + - [Vincent, 2] + - [Vincent, 3] + - [Vincent, 4] diff --git a/tests/spec/rosetta/values_through_ordering.gtest b/tests/spec/rosetta/values_through_ordering.gtest new file mode 100644 index 000000000..67d876abf --- /dev/null +++ b/tests/spec/rosetta/values_through_ordering.gtest @@ -0,0 +1,259 @@ +# Rosetta: values through ORDER BY, LIMIT, SKIP and DISTINCT +# +# Ordering, cutting and deduplicating rows must not change a value. The +# setup gives nodes and edges a property with the same name (`w`) and +# overlapping IDs, so a node or edge that loses track of what it is reads the +# other entity's `w` (nodes have 100 and up, edges 1 to 8): every expected `w` +# below tells which entity was read. The row-batch tests cross the 2048-row +# boundary where LIMIT and SKIP cut a batch. +# +# Dataset `entity_kinds` (tests/spec/datasets/entity_kinds.setup): Alix, Gus, +# Vincent (with w 100, 101, 103), Jules and Mia; Amsterdam (w 104), Berlin, +# Paris (w 105). KNOWS edges w 1 to 5, LIVES_IN edges w 6 to 8. The row-batch +# tests add `:N` nodes of their own and read only those. + +meta: + model: lpg + section: rosetta + title: Values through ORDER BY, LIMIT, SKIP and DISTINCT + dataset: entity_kinds + requires: [cypher] + +tests: + + # --------------------------------------------------------------------------- + # A node or edge ordered, cut or deduplicated before RETURN stays itself + # --------------------------------------------------------------------------- + + - name: edge_properties_after_an_ordered_cut + variants: + cypher: "MATCH (a:Person)-[r]->(b) WITH a, r ORDER BY r.w LIMIT 4 RETURN a.name AS n, r.w AS w, type(r) AS t" + expect: + rows: + - [Alix, 1, KNOWS] + - [Gus, 2, KNOWS] + - [Vincent, 3, KNOWS] + - [Jules, 4, KNOWS] + + - name: edge_properties_after_skip + variants: + cypher: "MATCH (a:Person)-[r]->(b) WITH r, b ORDER BY r.w SKIP 2 RETURN r.w AS w, b.name AS n" + expect: + rows: + - [3, Alix] + - [4, Mia] + - [5, Jules] + - [6, Amsterdam] + - [7, Berlin] + - [8, Paris] + + - name: an_aliased_edge_after_an_ordered_cut + variants: + cypher: "MATCH (a:Person)-[r]->(b) WITH r AS e ORDER BY e.w LIMIT 3 RETURN e.w AS w, type(e) AS t" + expect: + rows: + - [1, KNOWS] + - [2, KNOWS] + - [3, KNOWS] + + - name: edge_properties_after_a_computed_sort_key + variants: + cypher: "MATCH (a:Person)-[r]->(b) WITH a, r ORDER BY toString(r.w) RETURN r.w AS w, a.name AS n" + expect: + rows: + - [1, Alix] + - [2, Gus] + - [3, Vincent] + - [4, Jules] + - [5, Alix] + - [6, Alix] + - [7, Gus] + - [8, Mia] + + - name: edge_properties_after_sorting_by_type + variants: + cypher: "MATCH (a:Person)-[r]->(b) WITH r ORDER BY type(r) DESC, r.w LIMIT 3 RETURN r.w AS w" + expect: + rows: + - [6] + - [7] + - [8] + + - name: node_properties_after_distinct + variants: + cypher: "MATCH (a:Person)-[r]->(b) WITH DISTINCT a RETURN a.w AS w, a.name AS n" + expect: + rows: + - [100, Alix] + - [101, Gus] + - [103, Vincent] + - [null, Jules] + - [null, Mia] + + - name: node_properties_after_an_ordered_cut + variants: + cypher: "MATCH (a:Person) WITH a ORDER BY a.age DESC LIMIT 2 RETURN a.w AS w, a.name AS n" + expect: + rows: + - [103, Vincent] + - [null, Jules] + + - name: node_properties_after_skip + variants: + cypher: "MATCH (a:Person) WITH a ORDER BY a.name SKIP 1 RETURN a.w AS w, a.name AS n" + expect: + rows: + - [101, Gus] + - [103, Vincent] + - [null, Jules] + - [null, Mia] + + - name: end_node_properties_after_an_ordered_cut + variants: + cypher: "MATCH (a:Person)-[r]->(b) WITH b ORDER BY b.name LIMIT 3 RETURN b.w AS w, b.name AS n" + expect: + rows: + - [100, Alix] + - [104, Amsterdam] + - [null, Berlin] + + - name: node_properties_sorted_with_an_edge_key + variants: + cypher: "MATCH (a:Person)-[r]->(b) WITH r, a ORDER BY a.age, r.w SKIP 1 LIMIT 3 RETURN r.w AS w, a.w AS aw" + expect: + rows: + - [7, 101] + - [8, null] + - [1, 100] + + - name: a_let_copy_of_an_edge + query: "MATCH (a:Person)-[r]->(b) LET x = r RETURN x.w AS w ORDER BY w" + expect: + ordered: true + rows: + - [1] + - [2] + - [3] + - [4] + - [5] + - [6] + - [7] + - [8] + + # --------------------------------------------------------------------------- + # ORDER BY on a property: only the RETURN items are columns + # --------------------------------------------------------------------------- + + - name: sort_by_a_property_that_is_not_returned + variants: + gql: "MATCH (a:Person) RETURN a.name AS n ORDER BY a.age DESC" + cypher: "MATCH (a:Person) RETURN a.name AS n ORDER BY a.age DESC" + expect: + ordered: true + columns: [n] + rows: + - [Vincent] + - [Jules] + - [Alix] + - [Mia] + - [Gus] + + - name: sort_by_a_property_of_a_returned_alias + variants: + gql: "MATCH (a:Person) RETURN a.name AS n, a AS x ORDER BY x.age SKIP 2" + cypher: "MATCH (a:Person) RETURN a.name AS n, a AS x ORDER BY x.age SKIP 2" + expect: + columns: [n, x] + count: 3 + + - name: sort_by_an_expression_and_a_property_of_an_alias + variants: + gql: "MATCH (a:Person) RETURN a AS x ORDER BY labels(x)[0], x.name" + cypher: "MATCH (a:Person) RETURN a AS x ORDER BY labels(x)[0], x.name" + expect: + columns: [x] + count: 5 + + # --------------------------------------------------------------------------- + # Cuts across the 2048-row batch boundary (the edges each have a pair of nodes + # of their own: a setup linear in the row count, fast enough for WASM) + # --------------------------------------------------------------------------- + + - name: skip_across_a_row_batch + setup: + - "UNWIND range(0, 2099) AS i INSERT (:N {i: i, w: 1000 + i})" + variants: + gql: "MATCH (n:N) RETURN n.i AS i ORDER BY n.i SKIP 2046 LIMIT 5" + cypher: "MATCH (n:N) RETURN n.i AS i ORDER BY n.i SKIP 2046 LIMIT 5" + expect: + ordered: true + rows: + - [2046] + - [2047] + - [2048] + - [2049] + - [2050] + + - name: edge_properties_after_a_cut_across_row_batches + setup: + - "UNWIND range(0, 2098) AS i INSERT (:N {i: i, w: 1000 + i})-[:NEXT {k: i, w: i}]->(:N {i: i + 1, w: 1001 + i})" + variants: + cypher: "MATCH (n:N)-[r]->(m) WITH r ORDER BY r.k SKIP 2047 LIMIT 3 RETURN r.w AS w" + expect: + rows: + - [2047] + - [2048] + - [2049] + + - name: top_k_across_row_batches + setup: + - "UNWIND range(0, 2098) AS i INSERT (:N {i: i, w: 1000 + i})-[:NEXT {k: i, w: i}]->(:N {i: i + 1, w: 1001 + i})" + variants: + gql: "MATCH (n:N)-[r]->(m) RETURN r.w AS w ORDER BY r.k DESC LIMIT 2" + cypher: "MATCH (n:N)-[r]->(m) RETURN r.w AS w ORDER BY r.k DESC LIMIT 2" + expect: + ordered: true + rows: + - [2098] + - [2097] + + # --------------------------------------------------------------------------- + # Values of several types in one column + # --------------------------------------------------------------------------- + + - name: mixed_values_through_skip_and_limit + variants: + gql: "UNWIND [1, 'a', 2.5, null] AS x RETURN x SKIP 1 LIMIT 2" + cypher: "UNWIND [1, 'a', 2.5, null] AS x RETURN x SKIP 1 LIMIT 2" + expect: + rows: + - [a] + - [2.5] + + - name: mixed_values_in_a_later_row_batch + variants: + gql: "UNWIND range(1, 3000) AS i RETURN CASE WHEN i < 2500 THEN i ELSE 'x' END AS v SKIP 2498 LIMIT 3" + cypher: "UNWIND range(1, 3000) AS i RETURN CASE WHEN i < 2500 THEN i ELSE 'x' END AS v SKIP 2498 LIMIT 3" + expect: + rows: + - [2499] + - [x] + - [x] + + - name: mixed_values_sorted_across_row_batches + variants: + gql: "UNWIND range(1, 3000) AS i RETURN CASE WHEN i > 2048 THEN 'late' ELSE i END AS v ORDER BY i DESC LIMIT 2" + cypher: "UNWIND range(1, 3000) AS i RETURN CASE WHEN i > 2048 THEN 'late' ELSE i END AS v ORDER BY i DESC LIMIT 2" + expect: + ordered: true + rows: + - [late] + - [late] + + - name: for_loop_ordered_and_cut + query: "FOR x IN [3, 1, 2] RETURN x ORDER BY x LIMIT 2" + expect: + ordered: true + rows: + - [1] + - [2] diff --git a/tests/spec/rosetta/writes_after_a_delete.gtest b/tests/spec/rosetta/writes_after_a_delete.gtest new file mode 100644 index 000000000..5c940eff3 --- /dev/null +++ b/tests/spec/rosetta/writes_after_a_delete.gtest @@ -0,0 +1,65 @@ +# Rosetta: a write to a node or edge deleted earlier in the statement fails +# +# SET, REMOVE and label changes on a node or edge the statement (or its +# transaction) deleted before fail with an error, as in openCypher, instead of +# writing to an entity nobody sees. With a node type defined, the write used to +# skip the type's checks as well. +# +# Dataset `entity_kinds` (tests/spec/datasets/entity_kinds.setup): Mia has no +# `w`; the KNOWS edge with `w: 1` goes from Alix to Gus. + +meta: + model: lpg + section: rosetta + title: A write to a deleted node or edge fails + dataset: entity_kinds + requires: [cypher] + +tests: + + - name: set_a_property_of_a_deleted_node + variants: + gql: "MATCH (n:Person {name: 'Mia'}) DETACH DELETE n SET n.x = 1 RETURN n.x AS x" + cypher: "MATCH (n:Person {name: 'Mia'}) DETACH DELETE n SET n.x = 1 RETURN n.x AS x" + expect: + error: "has been deleted in this transaction" + + - name: remove_a_property_of_a_deleted_node + variants: + gql: "MATCH (n:Person {name: 'Alix'}) DETACH DELETE n REMOVE n.w RETURN n.w AS w" + cypher: "MATCH (n:Person {name: 'Alix'}) DETACH DELETE n REMOVE n.w RETURN n.w AS w" + expect: + error: "has been deleted in this transaction" + + - name: add_a_label_to_a_deleted_node + variants: + gql: "MATCH (n:Person {name: 'Mia'}) DETACH DELETE n SET n:Admin RETURN labels(n) AS l" + cypher: "MATCH (n:Person {name: 'Mia'}) DETACH DELETE n SET n:Admin RETURN labels(n) AS l" + expect: + error: "has been deleted in this transaction" + + - name: set_a_property_of_a_deleted_edge + variants: + gql: "MATCH ()-[r:KNOWS {w: 1}]->() DELETE r SET r.w = 9 RETURN r.w AS w" + cypher: "MATCH ()-[r:KNOWS {w: 1}]->() DELETE r SET r.w = 9 RETURN r.w AS w" + expect: + error: "has been deleted in this transaction" + + # The node-type check used to be skipped for a node it could not see. + - name: set_a_property_of_a_deleted_typed_node + setup: + - "CREATE NODE TYPE Robot (serial STRING NOT NULL)" + - "INSERT (:Robot {serial: 'R1'})" + variants: + gql: "MATCH (n:Robot) DETACH DELETE n SET n.serial = 7 RETURN n.serial AS s" + expect: + error: "has been deleted in this transaction" + + # A SET before the DELETE is an ordinary write. Cypher only: GQL takes no + # DELETE after a WITH yet. + - name: set_before_a_delete + variants: + cypher: "MATCH (n:Person {name: 'Mia'}) SET n.x = 1 WITH n DETACH DELETE n RETURN count(*) AS c" + expect: + rows: + - [1] diff --git a/tests/spec/runners/dart/spec_runner_test.dart b/tests/spec/runners/dart/spec_runner_test.dart index dfe30a5c6..6860cee82 100644 --- a/tests/spec/runners/dart/spec_runner_test.dart +++ b/tests/spec/runners/dart/spec_runner_test.dart @@ -1092,6 +1092,7 @@ void main() { setup: tc.setup, expect: tc.expect, tags: tc.tags, + params: tc.params, ); _runTestCase(db, variantTc, lang, meta.language); } finally { diff --git a/tests/spec/runners/wasm/spec-runner.test.mjs b/tests/spec/runners/wasm/spec-runner.test.mjs index b5dd0ec2b..7407d80ab 100644 --- a/tests/spec/runners/wasm/spec-runner.test.mjs +++ b/tests/spec/runners/wasm/spec-runner.test.mjs @@ -209,6 +209,8 @@ for (const filePath of gtestFiles) { for (const req of (tc.requires || [])) { if (!isAvailable(db, req)) return ctx.skip() } + // WASM executeRaw does not support params yet + if (tc.params && Object.keys(tc.params).length > 0) return ctx.skip() const effectiveDataset = tc.dataset || meta.dataset if (effectiveDataset && effectiveDataset !== 'empty') { loadDataset(db, effectiveDataset)