Compare commits
10 Commits
release/2.
...
free-memor
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
23474fd6cc | ||
|
|
91017b7f36 | ||
|
|
00f8d54249 | ||
|
|
4fcdd52f88 | ||
|
|
6c947947eb | ||
|
|
64fd281b2e | ||
|
|
97e250129e | ||
|
|
b02b201129 | ||
|
|
2c6a55775d | ||
|
|
940bf6722c |
2
.github/workflows/diff.yaml
vendored
2
.github/workflows/diff.yaml
vendored
@@ -437,7 +437,7 @@ jobs:
|
||||
- name: Run mgbench
|
||||
run: |
|
||||
cd tests/mgbench
|
||||
./benchmark.py --num-workers-for-benchmark 12 --export-results benchmark_result.json pokec/medium/*/*
|
||||
./benchmark.py vendor-native --num-workers-for-benchmark 12 --export-results benchmark_result.json pokec/medium/*/*
|
||||
|
||||
- name: Upload mgbench results
|
||||
run: |
|
||||
|
||||
63
.github/workflows/release_mgbench_client.yaml
vendored
Normal file
63
.github/workflows/release_mgbench_client.yaml
vendored
Normal file
@@ -0,0 +1,63 @@
|
||||
name: "Mgbench Bolt Client Publish Docker Image"
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
version:
|
||||
description: "Mgbench bolt client version to publish on Dockerhub."
|
||||
required: true
|
||||
force_release:
|
||||
type: boolean
|
||||
required: false
|
||||
default: false
|
||||
|
||||
jobs:
|
||||
mgbench_docker_publish:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
DOCKER_ORGANIZATION_NAME: memgraph
|
||||
DOCKER_REPOSITORY_NAME: mgbench-client
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v3
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v2
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
uses: docker/setup-buildx-action@v2
|
||||
|
||||
- name: Log in to Docker Hub
|
||||
uses: docker/login-action@v2
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Check if specified version is already pushed
|
||||
run: |
|
||||
EXISTS=$(docker manifest inspect $DOCKER_ORGANIZATION_NAME/$DOCKER_REPOSITORY_NAME:${{ github.event.inputs.version }} > /dev/null; echo $?)
|
||||
echo $EXISTS
|
||||
if [[ ${EXISTS} -eq 0 ]]; then
|
||||
echo 'The specified version has been already released to DockerHub.'
|
||||
if [[ ${{ github.event.inputs.force_release }} = true ]]; then
|
||||
echo 'Forcing the release!'
|
||||
else
|
||||
echo 'Stopping the release!'
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
echo 'All good the specified version has not been release to DockerHub.'
|
||||
fi
|
||||
|
||||
- name: Build & push docker images
|
||||
run: |
|
||||
cd tests/mgbench
|
||||
docker buildx build \
|
||||
--build-arg TOOLCHAIN_VERSION=toolchain-v4 \
|
||||
--platform linux/amd64,linux/arm64 \
|
||||
--tag $DOCKER_ORGANIZATION_NAME/$DOCKER_REPOSITORY_NAME:${{ github.event.inputs.version }} \
|
||||
--tag $DOCKER_ORGANIZATION_NAME/$DOCKER_REPOSITORY_NAME:latest \
|
||||
--file Dockerfile.mgbench_client \
|
||||
--push .
|
||||
@@ -103,6 +103,10 @@ modifications:
|
||||
value: "true"
|
||||
override: false
|
||||
|
||||
- name: "storage_parallel_index_recovery"
|
||||
value: "false"
|
||||
override: true
|
||||
|
||||
undocumented:
|
||||
- "flag_file"
|
||||
- "also_log_to_stderr"
|
||||
|
||||
@@ -117,7 +117,7 @@ declare -A primary_urls=(
|
||||
["mgconsole"]="http://$local_cache_host/git/mgconsole.git"
|
||||
["spdlog"]="http://$local_cache_host/git/spdlog"
|
||||
["nlohmann"]="http://$local_cache_host/file/nlohmann/json/4f8fba14066156b73f1189a2b8bd568bde5284c5/single_include/nlohmann/json.hpp"
|
||||
["neo4j"]="http://$local_cache_host/file/neo4j-community-3.2.3-unix.tar.gz"
|
||||
["neo4j"]="http://$local_cache_host/file/neo4j-community-5.6.0-unix.tar.gz"
|
||||
["librdkafka"]="http://$local_cache_host/git/librdkafka.git"
|
||||
["protobuf"]="http://$local_cache_host/git/protobuf.git"
|
||||
["pulsar"]="http://$local_cache_host/git/pulsar.git"
|
||||
@@ -142,7 +142,7 @@ declare -A secondary_urls=(
|
||||
["mgconsole"]="http://github.com/memgraph/mgconsole.git"
|
||||
["spdlog"]="https://github.com/gabime/spdlog"
|
||||
["nlohmann"]="https://raw.githubusercontent.com/nlohmann/json/4f8fba14066156b73f1189a2b8bd568bde5284c5/single_include/nlohmann/json.hpp"
|
||||
["neo4j"]="https://s3-eu-west-1.amazonaws.com/deps.memgraph.io/neo4j-community-3.2.3-unix.tar.gz"
|
||||
["neo4j"]="https://dist.neo4j.org/neo4j-community-5.6.0-unix.tar.gz"
|
||||
["librdkafka"]="https://github.com/edenhill/librdkafka.git"
|
||||
["protobuf"]="https://github.com/protocolbuffers/protobuf.git"
|
||||
["pulsar"]="https://github.com/apache/pulsar.git"
|
||||
@@ -180,9 +180,9 @@ repo_clone_try_double "${primary_urls[libbcrypt]}" "${secondary_urls[libbcrypt]}
|
||||
|
||||
# neo4j
|
||||
file_get_try_double "${primary_urls[neo4j]}" "${secondary_urls[neo4j]}"
|
||||
tar -xzf neo4j-community-3.2.3-unix.tar.gz
|
||||
mv neo4j-community-3.2.3 neo4j
|
||||
rm neo4j-community-3.2.3-unix.tar.gz
|
||||
tar -xzf neo4j-community-5.6.0-unix.tar.gz
|
||||
mv neo4j-community-5.6.0 neo4j
|
||||
rm neo4j-community-5.6.0-unix.tar.gz
|
||||
|
||||
# nlohmann json
|
||||
# We wget header instead of cloning repo since repo is huge (lots of test data).
|
||||
|
||||
202
licenses/third-party/ldbc/LICENSE
vendored
Normal file
202
licenses/third-party/ldbc/LICENSE
vendored
Normal file
@@ -0,0 +1,202 @@
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
@@ -192,6 +192,20 @@ DEFINE_VALIDATED_uint64(storage_wal_file_flush_every_n_tx,
|
||||
// NOLINTNEXTLINE(cppcoreguidelines-avoid-non-const-global-variables)
|
||||
DEFINE_bool(storage_snapshot_on_exit, false, "Controls whether the storage creates another snapshot on exit.");
|
||||
|
||||
// NOLINTNEXTLINE(cppcoreguidelines-avoid-non-const-global-variables)
|
||||
DEFINE_uint64(storage_items_per_batch, memgraph::storage::Config::Durability().items_per_batch,
|
||||
"The number of edges and vertices stored in a batch in a snapshot file.");
|
||||
|
||||
// NOLINTNEXTLINE(cppcoreguidelines-avoid-non-const-global-variables)
|
||||
DEFINE_bool(storage_parallel_index_recovery, false,
|
||||
"Controls whether the index creation can be done in a multithreaded fashion.");
|
||||
|
||||
// NOLINTNEXTLINE(cppcoreguidelines-avoid-non-const-global-variables)
|
||||
DEFINE_uint64(storage_recovery_thread_count,
|
||||
std::max(static_cast<uint64_t>(std::thread::hardware_concurrency()),
|
||||
memgraph::storage::Config::Durability().recovery_thread_count),
|
||||
"The number of threads used to recover persisted data from disk.");
|
||||
|
||||
// NOLINTNEXTLINE(cppcoreguidelines-avoid-non-const-global-variables)
|
||||
DEFINE_bool(telemetry_enabled, false,
|
||||
"Set to true to enable telemetry. We collect information about the "
|
||||
@@ -852,7 +866,10 @@ int main(int argc, char **argv) {
|
||||
.wal_file_size_kibibytes = FLAGS_storage_wal_file_size_kib,
|
||||
.wal_file_flush_every_n_tx = FLAGS_storage_wal_file_flush_every_n_tx,
|
||||
.snapshot_on_exit = FLAGS_storage_snapshot_on_exit,
|
||||
.restore_replicas_on_startup = true},
|
||||
.restore_replicas_on_startup = true,
|
||||
.items_per_batch = FLAGS_storage_items_per_batch,
|
||||
.recovery_thread_count = FLAGS_storage_recovery_thread_count,
|
||||
.allow_parallel_index_creation = FLAGS_storage_parallel_index_recovery},
|
||||
.transaction = {.isolation_level = ParseIsolationLevel()}};
|
||||
if (FLAGS_storage_snapshot_interval_sec == 0) {
|
||||
if (FLAGS_storage_wal_enabled) {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
|
||||
@@ -1068,14 +1068,14 @@ std::optional<plan::ProfilingStatsWithTotalTime> PullPlan::Pull(AnyStream *strea
|
||||
std::optional<utils::PoolResource> pool_memory;
|
||||
|
||||
if (!use_monotonic_memory_) {
|
||||
pool_memory.emplace(8, kExecutionPoolMaxBlockSize, utils::NewDeleteResource(), utils::NewDeleteResource());
|
||||
pool_memory.emplace(8, kExecutionPoolMaxBlockSize, &resource_with_exception, &resource_with_exception);
|
||||
} else {
|
||||
// We can throw on every query because a simple queries for deleting will use only
|
||||
// the stack allocated buffer.
|
||||
// Also, we want to throw only when the query engine requests more memory and not the storage
|
||||
// so we add the exception to the allocator.
|
||||
// TODO (mferencevic): Tune the parameters accordingly.
|
||||
pool_memory.emplace(128, 1024, &monotonic_memory, utils::NewDeleteResource());
|
||||
pool_memory.emplace(128, 1024, &monotonic_memory, &resource_with_exception);
|
||||
}
|
||||
|
||||
std::optional<utils::LimitedMemoryResource> maybe_limited_resource;
|
||||
@@ -1375,6 +1375,16 @@ PreparedQuery PrepareProfileQuery(ParsedQuery parsed_query, bool in_explicit_tra
|
||||
&interpreter_context->ast_cache, interpreter_context->config.query);
|
||||
|
||||
auto *cypher_query = utils::Downcast<CypherQuery>(parsed_inner_query.query);
|
||||
|
||||
bool contains_csv = false;
|
||||
auto clauses = cypher_query->single_query_->clauses_;
|
||||
if (std::any_of(clauses.begin(), clauses.end(),
|
||||
[](const auto *clause) { return clause->GetTypeInfo() == LoadCsv::kType; })) {
|
||||
contains_csv = true;
|
||||
}
|
||||
// If this is LOAD CSV query, use PoolResource without MonotonicMemoryResource as we want to reuse allocated memory
|
||||
auto use_monotonic_memory = !contains_csv;
|
||||
|
||||
MG_ASSERT(cypher_query, "Cypher grammar should not allow other queries in PROFILE");
|
||||
Frame frame(0);
|
||||
SymbolTable symbol_table;
|
||||
@@ -1399,14 +1409,14 @@ PreparedQuery PrepareProfileQuery(ParsedQuery parsed_query, bool in_explicit_tra
|
||||
// We want to execute the query we are profiling lazily, so we delay
|
||||
// the construction of the corresponding context.
|
||||
stats_and_total_time = std::optional<plan::ProfilingStatsWithTotalTime>{},
|
||||
pull_plan = std::shared_ptr<PullPlanVector>(nullptr), transaction_status](
|
||||
pull_plan = std::shared_ptr<PullPlanVector>(nullptr), transaction_status, use_monotonic_memory](
|
||||
AnyStream *stream, std::optional<int> n) mutable -> std::optional<QueryHandlerResult> {
|
||||
// No output symbols are given so that nothing is streamed.
|
||||
if (!stats_and_total_time) {
|
||||
stats_and_total_time =
|
||||
PullPlan(plan, parameters, true, dba, interpreter_context, execution_memory,
|
||||
optional_username, transaction_status, nullptr, memory_limit)
|
||||
.Pull(stream, {}, {}, summary);
|
||||
stats_and_total_time = PullPlan(plan, parameters, true, dba, interpreter_context,
|
||||
execution_memory, optional_username, transaction_status,
|
||||
nullptr, memory_limit, use_monotonic_memory)
|
||||
.Pull(stream, {}, {}, summary);
|
||||
pull_plan = std::make_shared<PullPlanVector>(ProfilingStatsToTable(*stats_and_total_time));
|
||||
}
|
||||
|
||||
@@ -2721,15 +2731,22 @@ Interpreter::PrepareResult Interpreter::Prepare(const std::string &query_string,
|
||||
ParseQuery(query_string, params, &interpreter_context_->ast_cache, interpreter_context_->config.query);
|
||||
TypedValue parsing_time{parsing_timer.Elapsed().count()};
|
||||
|
||||
if (utils::Downcast<CypherQuery>(parsed_query.query)) {
|
||||
auto *cypher_query = utils::Downcast<CypherQuery>(parsed_query.query);
|
||||
if ((utils::Downcast<CypherQuery>(parsed_query.query) || utils::Downcast<ProfileQuery>(parsed_query.query))) {
|
||||
CypherQuery *cypher_query = nullptr;
|
||||
if (utils::Downcast<CypherQuery>(parsed_query.query)) {
|
||||
cypher_query = utils::Downcast<CypherQuery>(parsed_query.query);
|
||||
} else {
|
||||
auto *profile_query = utils::Downcast<ProfileQuery>(parsed_query.query);
|
||||
cypher_query = profile_query->cypher_query_;
|
||||
}
|
||||
|
||||
if (const auto &clauses = cypher_query->single_query_->clauses_;
|
||||
std::any_of(clauses.begin(), clauses.end(),
|
||||
[](const auto *clause) { return clause->GetTypeInfo() == LoadCsv::kType; })) {
|
||||
// Using PoolResource without MonotonicMemoryResouce for LOAD CSV reduces memory usage.
|
||||
// QueryExecution MemoryResource is mostly used for allocations done on Frame and storing `row`s
|
||||
query_executions_[query_executions_.size() - 1] = std::make_unique<QueryExecution>(
|
||||
utils::PoolResource(1, kExecutionPoolMaxBlockSize, utils::NewDeleteResource(), utils::NewDeleteResource()));
|
||||
utils::PoolResource(8, kExecutionPoolMaxBlockSize, utils::NewDeleteResource(), utils::NewDeleteResource()));
|
||||
query_execution_ptr = &query_executions_.back();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -51,7 +51,7 @@ extern const Event FailedQuery;
|
||||
namespace memgraph::query {
|
||||
|
||||
inline constexpr size_t kExecutionMemoryBlockSize = 1UL * 1024UL * 1024UL;
|
||||
inline constexpr size_t kExecutionPoolMaxBlockSize = 32768UL; // 2 ^ 15
|
||||
inline constexpr size_t kExecutionPoolMaxBlockSize = 2048UL; // 2 ^ 11
|
||||
|
||||
class AuthQueryHandler {
|
||||
public:
|
||||
|
||||
@@ -53,6 +53,7 @@
|
||||
#include "utils/likely.hpp"
|
||||
#include "utils/logging.hpp"
|
||||
#include "utils/memory.hpp"
|
||||
#include "utils/pmr/deque.hpp"
|
||||
#include "utils/pmr/list.hpp"
|
||||
#include "utils/pmr/unordered_map.hpp"
|
||||
#include "utils/pmr/unordered_set.hpp"
|
||||
@@ -3248,7 +3249,7 @@ class AccumulateCursor : public Cursor {
|
||||
private:
|
||||
const Accumulate &self_;
|
||||
const UniqueCursorPtr input_cursor_;
|
||||
utils::pmr::vector<utils::pmr::vector<TypedValue>> cache_;
|
||||
utils::pmr::deque<utils::pmr::vector<TypedValue>> cache_;
|
||||
decltype(cache_.begin()) cache_it_ = cache_.begin();
|
||||
bool pulled_all_input_{false};
|
||||
};
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -50,6 +50,11 @@ struct Config {
|
||||
|
||||
bool snapshot_on_exit{false};
|
||||
bool restore_replicas_on_startup{false};
|
||||
|
||||
uint64_t items_per_batch{1'000'000};
|
||||
uint64_t recovery_thread_count{8};
|
||||
|
||||
bool allow_parallel_index_creation{false};
|
||||
} durability;
|
||||
|
||||
struct Transaction {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -113,13 +113,15 @@ std::optional<std::vector<WalDurabilityInfo>> GetWalFiles(const std::filesystem:
|
||||
// to ensure that the indices and constraints are consistent at the end of the
|
||||
// recovery process.
|
||||
void RecoverIndicesAndConstraints(const RecoveredIndicesAndConstraints &indices_constraints, Indices *indices,
|
||||
Constraints *constraints, utils::SkipList<Vertex> *vertices) {
|
||||
Constraints *constraints, utils::SkipList<Vertex> *vertices,
|
||||
const std::optional<ParalellizedIndexCreationInfo> ¶lell_exec_info) {
|
||||
spdlog::info("Recreating indices from metadata.");
|
||||
// Recover label indices.
|
||||
spdlog::info("Recreating {} label indices from metadata.", indices_constraints.indices.label.size());
|
||||
for (const auto &item : indices_constraints.indices.label) {
|
||||
if (!indices->label_index.CreateIndex(item, vertices->access()))
|
||||
if (!indices->label_index.CreateIndex(item, vertices->access(), paralell_exec_info))
|
||||
throw RecoveryFailure("The label index must be created here!");
|
||||
|
||||
spdlog::info("A label index is recreated from metadata.");
|
||||
}
|
||||
spdlog::info("Label indices are recreated.");
|
||||
@@ -163,7 +165,7 @@ std::optional<RecoveryInfo> RecoverData(const std::filesystem::path &snapshot_di
|
||||
std::deque<std::pair<std::string, uint64_t>> *epoch_history,
|
||||
utils::SkipList<Vertex> *vertices, utils::SkipList<Edge> *edges,
|
||||
std::atomic<uint64_t> *edge_count, NameIdMapper *name_id_mapper,
|
||||
Indices *indices, Constraints *constraints, Config::Items items,
|
||||
Indices *indices, Constraints *constraints, const Config &config,
|
||||
uint64_t *wal_seq_num) {
|
||||
utils::MemoryTracker::OutOfMemoryExceptionEnabler oom_exception;
|
||||
spdlog::info("Recovering persisted data using snapshot ({}) and WAL directory ({}).", snapshot_directory,
|
||||
@@ -195,7 +197,7 @@ std::optional<RecoveryInfo> RecoverData(const std::filesystem::path &snapshot_di
|
||||
}
|
||||
spdlog::info("Starting snapshot recovery from {}.", path);
|
||||
try {
|
||||
recovered_snapshot = LoadSnapshot(path, vertices, edges, epoch_history, name_id_mapper, edge_count, items);
|
||||
recovered_snapshot = LoadSnapshot(path, vertices, edges, epoch_history, name_id_mapper, edge_count, config);
|
||||
spdlog::info("Snapshot recovery successful!");
|
||||
break;
|
||||
} catch (const RecoveryFailure &e) {
|
||||
@@ -213,7 +215,11 @@ std::optional<RecoveryInfo> RecoverData(const std::filesystem::path &snapshot_di
|
||||
*epoch_id = std::move(recovered_snapshot->snapshot_info.epoch_id);
|
||||
|
||||
if (!utils::DirExists(wal_directory)) {
|
||||
RecoverIndicesAndConstraints(indices_constraints, indices, constraints, vertices);
|
||||
const auto par_exec_info = config.durability.allow_parallel_index_creation
|
||||
? std::make_optional(std::make_pair(recovery_info.vertex_batches,
|
||||
config.durability.recovery_thread_count))
|
||||
: std::nullopt;
|
||||
RecoverIndicesAndConstraints(indices_constraints, indices, constraints, vertices, par_exec_info);
|
||||
return recovered_snapshot->recovery_info;
|
||||
}
|
||||
} else {
|
||||
@@ -319,7 +325,7 @@ std::optional<RecoveryInfo> RecoverData(const std::filesystem::path &snapshot_di
|
||||
}
|
||||
try {
|
||||
auto info = LoadWal(wal_file.path, &indices_constraints, last_loaded_timestamp, vertices, edges, name_id_mapper,
|
||||
edge_count, items);
|
||||
edge_count, config.items);
|
||||
recovery_info.next_vertex_id = std::max(recovery_info.next_vertex_id, info.next_vertex_id);
|
||||
recovery_info.next_edge_id = std::max(recovery_info.next_edge_id, info.next_edge_id);
|
||||
recovery_info.next_timestamp = std::max(recovery_info.next_timestamp, info.next_timestamp);
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -91,13 +91,18 @@ std::optional<std::vector<WalDurabilityInfo>> GetWalFiles(const std::filesystem:
|
||||
std::string_view uuid = "",
|
||||
std::optional<size_t> current_seq_num = {});
|
||||
|
||||
using ParalellizedIndexCreationInfo =
|
||||
std::pair<std::vector<std::pair<Gid, uint64_t>> /*vertex_recovery_info*/, uint64_t /*thread_count*/>;
|
||||
|
||||
// Helper function used to recover all discovered indices and constraints. The
|
||||
// indices and constraints must be recovered after the data recovery is done
|
||||
// to ensure that the indices and constraints are consistent at the end of the
|
||||
// recovery process.
|
||||
/// @throw RecoveryFailure
|
||||
void RecoverIndicesAndConstraints(const RecoveredIndicesAndConstraints &indices_constraints, Indices *indices,
|
||||
Constraints *constraints, utils::SkipList<Vertex> *vertices);
|
||||
void RecoverIndicesAndConstraints(
|
||||
const RecoveredIndicesAndConstraints &indices_constraints, Indices *indices, Constraints *constraints,
|
||||
utils::SkipList<Vertex> *vertices,
|
||||
const std::optional<ParalellizedIndexCreationInfo> ¶lell_exec_info = std::nullopt);
|
||||
|
||||
/// Recovers data either from a snapshot and/or WAL files.
|
||||
/// @throw RecoveryFailure
|
||||
@@ -108,7 +113,7 @@ std::optional<RecoveryInfo> RecoverData(const std::filesystem::path &snapshot_di
|
||||
std::deque<std::pair<std::string, uint64_t>> *epoch_history,
|
||||
utils::SkipList<Vertex> *vertices, utils::SkipList<Edge> *edges,
|
||||
std::atomic<uint64_t> *edge_count, NameIdMapper *name_id_mapper,
|
||||
Indices *indices, Constraints *constraints, Config::Items items,
|
||||
Indices *indices, Constraints *constraints, const Config &config,
|
||||
uint64_t *wal_seq_num);
|
||||
|
||||
} // namespace memgraph::storage::durability
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -12,6 +12,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <algorithm>
|
||||
#include <optional>
|
||||
#include <set>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
@@ -29,6 +30,8 @@ struct RecoveryInfo {
|
||||
|
||||
// last timestamp read from a WAL file
|
||||
std::optional<uint64_t> last_commit_timestamp;
|
||||
|
||||
std::vector<std::pair<Gid /*first vertex gid*/, uint64_t /*batch size*/>> vertex_batches;
|
||||
};
|
||||
|
||||
/// Structure used to track indices and constraints during recovery.
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -11,18 +11,26 @@
|
||||
|
||||
#include "storage/v2/durability/snapshot.hpp"
|
||||
|
||||
#include <thread>
|
||||
|
||||
#include "storage/v2/durability/exceptions.hpp"
|
||||
#include "storage/v2/durability/paths.hpp"
|
||||
#include "storage/v2/durability/serialization.hpp"
|
||||
#include "storage/v2/durability/version.hpp"
|
||||
#include "storage/v2/durability/wal.hpp"
|
||||
#include "storage/v2/edge.hpp"
|
||||
#include "storage/v2/edge_accessor.hpp"
|
||||
#include "storage/v2/edge_ref.hpp"
|
||||
#include "storage/v2/id_types.hpp"
|
||||
#include "storage/v2/mvcc.hpp"
|
||||
#include "storage/v2/vertex.hpp"
|
||||
#include "storage/v2/vertex_accessor.hpp"
|
||||
#include "utils/concepts.hpp"
|
||||
#include "utils/file_locker.hpp"
|
||||
#include "utils/logging.hpp"
|
||||
#include "utils/message.hpp"
|
||||
#include "utils/spin_lock.hpp"
|
||||
#include "utils/synchronized.hpp"
|
||||
|
||||
namespace memgraph::storage::durability {
|
||||
|
||||
@@ -40,6 +48,8 @@ namespace memgraph::storage::durability {
|
||||
// * offset to the constraints section
|
||||
// * offset to the mapper section
|
||||
// * offset to the metadata section
|
||||
// * offset to the offset-count pair of the first edge batch (`0` if properties on edges are disabled)
|
||||
// * offset to the offset-count pair of the first vertex batch
|
||||
//
|
||||
// 4) Encoded edges (if properties on edges are enabled); each edge is written
|
||||
// in the following format:
|
||||
@@ -87,9 +97,23 @@ namespace memgraph::storage::durability {
|
||||
// * number of edges
|
||||
// * number of vertices
|
||||
//
|
||||
// 10) Batch infos
|
||||
// * number of edge batch infos
|
||||
// * edge batch infos
|
||||
// * starting offset of the batch
|
||||
// * number of edges in the batch
|
||||
// * vertex batch infos
|
||||
// * starting offset of the batch
|
||||
// * number of vertices in the batch
|
||||
//
|
||||
// IMPORTANT: When changing snapshot encoding/decoding bump the snapshot/WAL
|
||||
// version in `version.hpp`.
|
||||
|
||||
struct BatchInfo {
|
||||
uint64_t offset;
|
||||
uint64_t count;
|
||||
};
|
||||
|
||||
// Function used to read information about the snapshot file.
|
||||
SnapshotInfo ReadSnapshotInfo(const std::filesystem::path &path) {
|
||||
// Check magic and version.
|
||||
@@ -124,6 +148,13 @@ SnapshotInfo ReadSnapshotInfo(const std::filesystem::path &path) {
|
||||
info.offset_mapper = read_offset();
|
||||
info.offset_epoch_history = read_offset();
|
||||
info.offset_metadata = read_offset();
|
||||
if (*version >= 15U) {
|
||||
info.offset_edge_batches = read_offset();
|
||||
info.offset_vertex_batches = read_offset();
|
||||
} else {
|
||||
info.offset_edge_batches = 0U;
|
||||
info.offset_vertex_batches = 0U;
|
||||
}
|
||||
}
|
||||
|
||||
// Read metadata.
|
||||
@@ -157,17 +188,385 @@ SnapshotInfo ReadSnapshotInfo(const std::filesystem::path &path) {
|
||||
return info;
|
||||
}
|
||||
|
||||
RecoveredSnapshot LoadSnapshot(const std::filesystem::path &path, utils::SkipList<Vertex> *vertices,
|
||||
utils::SkipList<Edge> *edges,
|
||||
std::deque<std::pair<std::string, uint64_t>> *epoch_history,
|
||||
NameIdMapper *name_id_mapper, std::atomic<uint64_t> *edge_count, Config::Items items) {
|
||||
std::vector<BatchInfo> ReadBatchInfos(Decoder &snapshot) {
|
||||
std::vector<BatchInfo> infos;
|
||||
const auto infos_size = snapshot.ReadUint();
|
||||
if (!infos_size.has_value()) {
|
||||
throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
infos.reserve(*infos_size);
|
||||
|
||||
for (auto i{0U}; i < *infos_size; ++i) {
|
||||
const auto offset = snapshot.ReadUint();
|
||||
if (!offset.has_value()) {
|
||||
throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
|
||||
const auto count = snapshot.ReadUint();
|
||||
if (!count.has_value()) {
|
||||
throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
infos.push_back(BatchInfo{*offset, *count});
|
||||
}
|
||||
return infos;
|
||||
}
|
||||
|
||||
template <typename TFunc>
|
||||
void LoadPartialEdges(const std::filesystem::path &path, utils::SkipList<Edge> &edges, const uint64_t from_offset,
|
||||
const uint64_t edges_count, const Config::Items items, TFunc get_property_from_id) {
|
||||
Decoder snapshot;
|
||||
snapshot.Initialize(path, kSnapshotMagic);
|
||||
|
||||
// Recover edges.
|
||||
auto edge_acc = edges.access();
|
||||
uint64_t last_edge_gid = 0;
|
||||
spdlog::info("Recovering {} edges.", edges_count);
|
||||
if (!snapshot.SetPosition(from_offset)) throw RecoveryFailure("Couldn't read data from snapshot!");
|
||||
|
||||
std::vector<std::pair<PropertyId, PropertyValue>> read_properties;
|
||||
for (uint64_t i = 0; i < edges_count; ++i) {
|
||||
{
|
||||
const auto marker = snapshot.ReadMarker();
|
||||
if (!marker || *marker != Marker::SECTION_EDGE) throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
|
||||
// Read edge GID.
|
||||
auto gid = snapshot.ReadUint();
|
||||
if (!gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
if (i > 0 && *gid <= last_edge_gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
last_edge_gid = *gid;
|
||||
|
||||
if (items.properties_on_edges) {
|
||||
spdlog::debug("Recovering edge {} with properties.", *gid);
|
||||
auto [it, inserted] = edge_acc.insert(Edge{Gid::FromUint(*gid), nullptr});
|
||||
if (!inserted) throw RecoveryFailure("The edge must be inserted here!");
|
||||
|
||||
// Recover properties.
|
||||
{
|
||||
auto props_size = snapshot.ReadUint();
|
||||
if (!props_size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto &props = it->properties;
|
||||
read_properties.clear();
|
||||
read_properties.reserve(*props_size);
|
||||
for (uint64_t j = 0; j < *props_size; ++j) {
|
||||
auto key = snapshot.ReadUint();
|
||||
if (!key) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto value = snapshot.ReadPropertyValue();
|
||||
if (!value) throw RecoveryFailure("Invalid snapshot data!");
|
||||
read_properties.emplace_back(get_property_from_id(*key), std::move(*value));
|
||||
}
|
||||
props.InitProperties(std::move(read_properties));
|
||||
}
|
||||
} else {
|
||||
spdlog::debug("Ensuring edge {} doesn't have any properties.", *gid);
|
||||
// Read properties.
|
||||
{
|
||||
auto props_size = snapshot.ReadUint();
|
||||
if (!props_size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
if (*props_size != 0)
|
||||
throw RecoveryFailure(
|
||||
"The snapshot has properties on edges, but the storage is "
|
||||
"configured without properties on edges!");
|
||||
}
|
||||
}
|
||||
}
|
||||
spdlog::info("Partial edges are recovered.");
|
||||
}
|
||||
|
||||
// Returns the gid of the last recovered vertex
|
||||
template <typename TLabelFromIdFunc, typename TPropertyFromIdFunc>
|
||||
uint64_t LoadPartialVertices(const std::filesystem::path &path, utils::SkipList<Vertex> &vertices,
|
||||
const uint64_t from_offset, const uint64_t vertices_count,
|
||||
TLabelFromIdFunc get_label_from_id, TPropertyFromIdFunc get_property_from_id) {
|
||||
Decoder snapshot;
|
||||
snapshot.Initialize(path, kSnapshotMagic);
|
||||
if (!snapshot.SetPosition(from_offset)) throw RecoveryFailure("Couldn't read data from snapshot!");
|
||||
|
||||
auto vertex_acc = vertices.access();
|
||||
uint64_t last_vertex_gid = 0;
|
||||
spdlog::info("Recovering {} vertices.", vertices_count);
|
||||
std::vector<std::pair<PropertyId, PropertyValue>> read_properties;
|
||||
for (uint64_t i = 0; i < vertices_count; ++i) {
|
||||
{
|
||||
auto marker = snapshot.ReadMarker();
|
||||
if (!marker || *marker != Marker::SECTION_VERTEX) throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
|
||||
// Insert vertex.
|
||||
auto gid = snapshot.ReadUint();
|
||||
if (!gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
if (i > 0 && *gid <= last_vertex_gid) {
|
||||
throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
last_vertex_gid = *gid;
|
||||
spdlog::debug("Recovering vertex {}.", *gid);
|
||||
auto [it, inserted] = vertex_acc.insert(Vertex{Gid::FromUint(*gid), nullptr});
|
||||
if (!inserted) throw RecoveryFailure("The vertex must be inserted here!");
|
||||
|
||||
// Recover labels.
|
||||
spdlog::trace("Recovering labels for vertex {}.", *gid);
|
||||
{
|
||||
auto labels_size = snapshot.ReadUint();
|
||||
if (!labels_size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto &labels = it->labels;
|
||||
labels.reserve(*labels_size);
|
||||
for (uint64_t j = 0; j < *labels_size; ++j) {
|
||||
auto label = snapshot.ReadUint();
|
||||
if (!label) throw RecoveryFailure("Invalid snapshot data!");
|
||||
labels.emplace_back(get_label_from_id(*label));
|
||||
}
|
||||
}
|
||||
|
||||
// Recover properties.
|
||||
spdlog::trace("Recovering properties for vertex {}.", *gid);
|
||||
{
|
||||
auto props_size = snapshot.ReadUint();
|
||||
if (!props_size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto &props = it->properties;
|
||||
read_properties.clear();
|
||||
read_properties.reserve(*props_size);
|
||||
for (uint64_t j = 0; j < *props_size; ++j) {
|
||||
auto key = snapshot.ReadUint();
|
||||
if (!key) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto value = snapshot.ReadPropertyValue();
|
||||
if (!value) throw RecoveryFailure("Invalid snapshot data!");
|
||||
read_properties.emplace_back(get_property_from_id(*key), std::move(*value));
|
||||
}
|
||||
props.InitProperties(std::move(read_properties));
|
||||
}
|
||||
|
||||
// Skip in edges.
|
||||
{
|
||||
auto in_size = snapshot.ReadUint();
|
||||
if (!in_size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
for (uint64_t j = 0; j < *in_size; ++j) {
|
||||
auto edge_gid = snapshot.ReadUint();
|
||||
if (!edge_gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto from_gid = snapshot.ReadUint();
|
||||
if (!from_gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto edge_type = snapshot.ReadUint();
|
||||
if (!edge_type) throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
}
|
||||
|
||||
// Skip out edges.
|
||||
auto out_size = snapshot.ReadUint();
|
||||
if (!out_size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
for (uint64_t j = 0; j < *out_size; ++j) {
|
||||
auto edge_gid = snapshot.ReadUint();
|
||||
if (!edge_gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto to_gid = snapshot.ReadUint();
|
||||
if (!to_gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto edge_type = snapshot.ReadUint();
|
||||
if (!edge_type) throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
}
|
||||
spdlog::info("Partial vertices are recovered.");
|
||||
|
||||
return last_vertex_gid;
|
||||
}
|
||||
|
||||
// Returns the number of edges recovered
|
||||
|
||||
struct LoadPartialConnectivityResult {
|
||||
uint64_t edge_count;
|
||||
uint64_t highest_edge_id;
|
||||
Gid first_vertex_gid;
|
||||
};
|
||||
|
||||
template <typename TEdgeTypeFromIdFunc>
|
||||
LoadPartialConnectivityResult LoadPartialConnectivity(const std::filesystem::path &path,
|
||||
utils::SkipList<Vertex> &vertices, utils::SkipList<Edge> &edges,
|
||||
const uint64_t from_offset, const uint64_t vertices_count,
|
||||
const Config::Items items, const bool snapshot_has_edges,
|
||||
TEdgeTypeFromIdFunc get_edge_type_from_id) {
|
||||
Decoder snapshot;
|
||||
snapshot.Initialize(path, kSnapshotMagic);
|
||||
if (!snapshot.SetPosition(from_offset)) throw RecoveryFailure("Couldn't read data from snapshot!");
|
||||
|
||||
auto vertex_acc = vertices.access();
|
||||
auto edge_acc = edges.access();
|
||||
|
||||
// Read the first gid to find the necessary iterator in vertices
|
||||
const auto first_vertex_gid = std::invoke([&]() mutable {
|
||||
{
|
||||
auto marker = snapshot.ReadMarker();
|
||||
if (!marker || *marker != Marker::SECTION_VERTEX) throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
|
||||
auto gid = snapshot.ReadUint();
|
||||
if (!gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
return Gid::FromUint(*gid);
|
||||
});
|
||||
|
||||
uint64_t edge_count{0};
|
||||
uint64_t highest_edge_gid{0};
|
||||
auto vertex_it = vertex_acc.find(first_vertex_gid);
|
||||
if (vertex_it == vertex_acc.end()) {
|
||||
throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
|
||||
spdlog::info("Recovering connectivity for {} vertices.", vertices_count);
|
||||
|
||||
if (!snapshot.SetPosition(from_offset)) throw RecoveryFailure("Couldn't read data from snapshot!");
|
||||
|
||||
for (uint64_t i = 0; i < vertices_count; ++i) {
|
||||
auto &vertex = *vertex_it;
|
||||
{
|
||||
auto marker = snapshot.ReadMarker();
|
||||
if (!marker || *marker != Marker::SECTION_VERTEX) throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
|
||||
auto gid = snapshot.ReadUint();
|
||||
if (!gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
if (gid != vertex.gid.AsUint()) throw RecoveryFailure("Invalid snapshot data!");
|
||||
|
||||
// Skip labels.
|
||||
{
|
||||
auto labels_size = snapshot.ReadUint();
|
||||
if (!labels_size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
for (uint64_t j = 0; j < *labels_size; ++j) {
|
||||
auto label = snapshot.ReadUint();
|
||||
if (!label) throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
}
|
||||
|
||||
// Skip properties.
|
||||
{
|
||||
auto props_size = snapshot.ReadUint();
|
||||
if (!props_size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
for (uint64_t j = 0; j < *props_size; ++j) {
|
||||
auto key = snapshot.ReadUint();
|
||||
if (!key) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto value = snapshot.SkipPropertyValue();
|
||||
if (!value) throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
}
|
||||
|
||||
// Recover in edges.
|
||||
{
|
||||
spdlog::trace("Recovering inbound edges for vertex {}.", vertex.gid.AsUint());
|
||||
auto in_size = snapshot.ReadUint();
|
||||
if (!in_size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
vertex.in_edges.reserve(*in_size);
|
||||
for (uint64_t j = 0; j < *in_size; ++j) {
|
||||
auto edge_gid = snapshot.ReadUint();
|
||||
if (!edge_gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
highest_edge_gid = std::max(highest_edge_gid, *edge_gid);
|
||||
|
||||
auto from_gid = snapshot.ReadUint();
|
||||
if (!from_gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto edge_type = snapshot.ReadUint();
|
||||
if (!edge_type) throw RecoveryFailure("Invalid snapshot data!");
|
||||
|
||||
auto from_vertex = vertex_acc.find(Gid::FromUint(*from_gid));
|
||||
if (from_vertex == vertex_acc.end()) throw RecoveryFailure("Invalid from vertex!");
|
||||
|
||||
EdgeRef edge_ref(Gid::FromUint(*edge_gid));
|
||||
if (items.properties_on_edges) {
|
||||
// The snapshot contains the individiual edges only if it was created with a config where properties are
|
||||
// allowed on edges. That means the snapshots that were created without edge properties will only contain the
|
||||
// edges in the in/out edges list of vertices, therefore the edges has to be created here.
|
||||
if (snapshot_has_edges) {
|
||||
auto edge = edge_acc.find(Gid::FromUint(*edge_gid));
|
||||
if (edge == edge_acc.end()) throw RecoveryFailure("Invalid edge!");
|
||||
edge_ref = EdgeRef(&*edge);
|
||||
} else {
|
||||
auto [edge, inserted] = edge_acc.insert(Edge{Gid::FromUint(*edge_gid), nullptr});
|
||||
edge_ref = EdgeRef(&*edge);
|
||||
}
|
||||
}
|
||||
vertex.in_edges.emplace_back(get_edge_type_from_id(*edge_type), &*from_vertex, edge_ref);
|
||||
}
|
||||
}
|
||||
|
||||
// Recover out edges.
|
||||
{
|
||||
spdlog::trace("Recovering outbound edges for vertex {}.", vertex.gid.AsUint());
|
||||
auto out_size = snapshot.ReadUint();
|
||||
if (!out_size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
vertex.out_edges.reserve(*out_size);
|
||||
for (uint64_t j = 0; j < *out_size; ++j) {
|
||||
auto edge_gid = snapshot.ReadUint();
|
||||
if (!edge_gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
|
||||
auto to_gid = snapshot.ReadUint();
|
||||
if (!to_gid) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto edge_type = snapshot.ReadUint();
|
||||
if (!edge_type) throw RecoveryFailure("Invalid snapshot data!");
|
||||
|
||||
auto to_vertex = vertex_acc.find(Gid::FromUint(*to_gid));
|
||||
if (to_vertex == vertex_acc.end()) throw RecoveryFailure("Invalid to vertex!");
|
||||
|
||||
EdgeRef edge_ref(Gid::FromUint(*edge_gid));
|
||||
if (items.properties_on_edges) {
|
||||
// The snapshot contains the individiual edges only if it was created with a config where properties are
|
||||
// allowed on edges. That means the snapshots that were created without edge properties will only contain the
|
||||
// edges in the in/out edges list of vertices, therefore the edges has to be created here.
|
||||
if (snapshot_has_edges) {
|
||||
auto edge = edge_acc.find(Gid::FromUint(*edge_gid));
|
||||
if (edge == edge_acc.end()) throw RecoveryFailure("Invalid edge!");
|
||||
edge_ref = EdgeRef(&*edge);
|
||||
} else {
|
||||
auto [edge, inserted] = edge_acc.insert(Edge{Gid::FromUint(*edge_gid), nullptr});
|
||||
edge_ref = EdgeRef(&*edge);
|
||||
}
|
||||
}
|
||||
vertex.out_edges.emplace_back(get_edge_type_from_id(*edge_type), &*to_vertex, edge_ref);
|
||||
// Increment edge count. We only increment the count here because the
|
||||
// information is duplicated in in_edges.
|
||||
edge_count++;
|
||||
}
|
||||
}
|
||||
++vertex_it;
|
||||
}
|
||||
spdlog::info("Partial connectivities are recovered.");
|
||||
return {edge_count, highest_edge_gid, first_vertex_gid};
|
||||
}
|
||||
|
||||
template <typename TFunc>
|
||||
void RecoverOnMultipleThreads(size_t thread_count, const TFunc &func, const std::vector<BatchInfo> &batches) {
|
||||
utils::Synchronized<std::optional<RecoveryFailure>, utils::SpinLock> maybe_error{};
|
||||
{
|
||||
std::atomic<uint64_t> batch_counter = 0;
|
||||
thread_count = std::min(thread_count, batches.size());
|
||||
std::vector<std::jthread> threads;
|
||||
threads.reserve(thread_count);
|
||||
|
||||
for (auto i{0U}; i < thread_count; ++i) {
|
||||
threads.emplace_back([&func, &batches, &maybe_error, &batch_counter]() {
|
||||
while (!maybe_error.Lock()->has_value()) {
|
||||
const auto batch_index = batch_counter++;
|
||||
if (batch_index >= batches.size()) {
|
||||
return;
|
||||
}
|
||||
const auto &batch = batches[batch_index];
|
||||
try {
|
||||
func(batch_index, batch);
|
||||
} catch (RecoveryFailure &failure) {
|
||||
*maybe_error.Lock() = std::move(failure);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
if (maybe_error.Lock()->has_value()) {
|
||||
throw RecoveryFailure((*maybe_error.Lock())->what());
|
||||
}
|
||||
}
|
||||
|
||||
RecoveredSnapshot LoadSnapshotVersion14(const std::filesystem::path &path, utils::SkipList<Vertex> *vertices,
|
||||
utils::SkipList<Edge> *edges,
|
||||
std::deque<std::pair<std::string, uint64_t>> *epoch_history,
|
||||
NameIdMapper *name_id_mapper, std::atomic<uint64_t> *edge_count,
|
||||
Config::Items items) {
|
||||
RecoveryInfo ret;
|
||||
RecoveredIndicesAndConstraints indices_constraints;
|
||||
|
||||
Decoder snapshot;
|
||||
auto version = snapshot.Initialize(path, kSnapshotMagic);
|
||||
if (!version) throw RecoveryFailure("Couldn't read snapshot magic and/or version!");
|
||||
if (!IsVersionSupported(*version)) throw RecoveryFailure(fmt::format("Invalid snapshot version {}", *version));
|
||||
if (*version != 14U) throw RecoveryFailure(fmt::format("Expected snapshot version is 14, but got {}", *version));
|
||||
|
||||
// Cleanup of loaded data in case of failure.
|
||||
bool success = false;
|
||||
@@ -625,10 +1024,297 @@ RecoveredSnapshot LoadSnapshot(const std::filesystem::path &path, utils::SkipLis
|
||||
return {info, ret, std::move(indices_constraints)};
|
||||
}
|
||||
|
||||
RecoveredSnapshot LoadSnapshot(const std::filesystem::path &path, utils::SkipList<Vertex> *vertices,
|
||||
utils::SkipList<Edge> *edges,
|
||||
std::deque<std::pair<std::string, uint64_t>> *epoch_history,
|
||||
NameIdMapper *name_id_mapper, std::atomic<uint64_t> *edge_count, const Config &config) {
|
||||
RecoveryInfo recovery_info;
|
||||
RecoveredIndicesAndConstraints indices_constraints;
|
||||
|
||||
Decoder snapshot;
|
||||
const auto version = snapshot.Initialize(path, kSnapshotMagic);
|
||||
if (!version) throw RecoveryFailure("Couldn't read snapshot magic and/or version!");
|
||||
|
||||
if (!IsVersionSupported(*version)) throw RecoveryFailure(fmt::format("Invalid snapshot version {}", *version));
|
||||
if (*version == 14U) {
|
||||
return LoadSnapshotVersion14(path, vertices, edges, epoch_history, name_id_mapper, edge_count, config.items);
|
||||
}
|
||||
|
||||
// Cleanup of loaded data in case of failure.
|
||||
bool success = false;
|
||||
utils::OnScopeExit cleanup([&] {
|
||||
if (!success) {
|
||||
edges->clear();
|
||||
vertices->clear();
|
||||
epoch_history->clear();
|
||||
}
|
||||
});
|
||||
|
||||
// Read snapshot info.
|
||||
const auto info = ReadSnapshotInfo(path);
|
||||
spdlog::info("Recovering {} vertices and {} edges.", info.vertices_count, info.edges_count);
|
||||
// Check for edges.
|
||||
bool snapshot_has_edges = info.offset_edges != 0;
|
||||
|
||||
// Recover mapper.
|
||||
std::unordered_map<uint64_t, uint64_t> snapshot_id_map;
|
||||
{
|
||||
spdlog::info("Recovering mapper metadata.");
|
||||
if (!snapshot.SetPosition(info.offset_mapper)) throw RecoveryFailure("Couldn't read data from snapshot!");
|
||||
|
||||
auto marker = snapshot.ReadMarker();
|
||||
if (!marker || *marker != Marker::SECTION_MAPPER) throw RecoveryFailure("Invalid snapshot data!");
|
||||
|
||||
auto size = snapshot.ReadUint();
|
||||
if (!size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
|
||||
for (uint64_t i = 0; i < *size; ++i) {
|
||||
auto id = snapshot.ReadUint();
|
||||
if (!id) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto name = snapshot.ReadString();
|
||||
if (!name) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto my_id = name_id_mapper->NameToId(*name);
|
||||
snapshot_id_map.emplace(*id, my_id);
|
||||
SPDLOG_TRACE("Mapping \"{}\"from snapshot id {} to actual id {}.", *name, *id, my_id);
|
||||
}
|
||||
}
|
||||
auto get_label_from_id = [&snapshot_id_map](uint64_t snapshot_id) {
|
||||
auto it = snapshot_id_map.find(snapshot_id);
|
||||
if (it == snapshot_id_map.end()) throw RecoveryFailure("Invalid snapshot data!");
|
||||
return LabelId::FromUint(it->second);
|
||||
};
|
||||
auto get_property_from_id = [&snapshot_id_map](uint64_t snapshot_id) {
|
||||
auto it = snapshot_id_map.find(snapshot_id);
|
||||
if (it == snapshot_id_map.end()) throw RecoveryFailure("Invalid snapshot data!");
|
||||
return PropertyId::FromUint(it->second);
|
||||
};
|
||||
auto get_edge_type_from_id = [&snapshot_id_map](uint64_t snapshot_id) {
|
||||
auto it = snapshot_id_map.find(snapshot_id);
|
||||
if (it == snapshot_id_map.end()) throw RecoveryFailure("Invalid snapshot data!");
|
||||
return EdgeTypeId::FromUint(it->second);
|
||||
};
|
||||
|
||||
// Reset current edge count.
|
||||
edge_count->store(0, std::memory_order_release);
|
||||
|
||||
{
|
||||
spdlog::info("Recovering edges.");
|
||||
// Recover edges.
|
||||
if (snapshot_has_edges) {
|
||||
// We don't need to check whether we store properties on edge or not, because `LoadPartialEdges` will always
|
||||
// iterate over the edges in the snapshot (if they exist) and the current configuration of properties on edge only
|
||||
// affect what it does:
|
||||
// 1. If properties are allowed on edges, then it loads the edges.
|
||||
// 2. If properties are not allowed on edges, then it checks that none of the edges have any properties.
|
||||
if (!snapshot.SetPosition(info.offset_edge_batches)) {
|
||||
throw RecoveryFailure("Couldn't read data from snapshot!");
|
||||
}
|
||||
const auto edge_batches = ReadBatchInfos(snapshot);
|
||||
|
||||
RecoverOnMultipleThreads(
|
||||
config.durability.recovery_thread_count,
|
||||
[path, edges, items = config.items, &get_property_from_id](const size_t /*batch_index*/,
|
||||
const BatchInfo &batch) {
|
||||
LoadPartialEdges(path, *edges, batch.offset, batch.count, items, get_property_from_id);
|
||||
},
|
||||
edge_batches);
|
||||
}
|
||||
spdlog::info("Edges are recovered.");
|
||||
|
||||
// Recover vertices (labels and properties).
|
||||
spdlog::info("Recovering vertices.", info.vertices_count);
|
||||
uint64_t last_vertex_gid{0};
|
||||
|
||||
if (!snapshot.SetPosition(info.offset_vertex_batches)) {
|
||||
throw RecoveryFailure("Couldn't read data from snapshot!");
|
||||
}
|
||||
|
||||
const auto vertex_batches = ReadBatchInfos(snapshot);
|
||||
RecoverOnMultipleThreads(
|
||||
config.durability.recovery_thread_count,
|
||||
[path, vertices, &vertex_batches, &get_label_from_id, &get_property_from_id, &last_vertex_gid](
|
||||
const size_t batch_index, const BatchInfo &batch) {
|
||||
const auto last_vertex_gid_in_batch =
|
||||
LoadPartialVertices(path, *vertices, batch.offset, batch.count, get_label_from_id, get_property_from_id);
|
||||
if (batch_index == vertex_batches.size() - 1) {
|
||||
last_vertex_gid = last_vertex_gid_in_batch;
|
||||
}
|
||||
},
|
||||
vertex_batches);
|
||||
|
||||
spdlog::info("Vertices are recovered.");
|
||||
|
||||
// Recover vertices (in/out edges).
|
||||
spdlog::info("Recover connectivity.");
|
||||
recovery_info.vertex_batches.reserve(vertex_batches.size());
|
||||
for (const auto batch : vertex_batches) {
|
||||
recovery_info.vertex_batches.emplace_back(std::make_pair(Gid::FromUint(0), batch.count));
|
||||
}
|
||||
std::atomic<uint64_t> highest_edge_gid{0};
|
||||
|
||||
RecoverOnMultipleThreads(
|
||||
config.durability.recovery_thread_count,
|
||||
[path, vertices, edges, edge_count, items = config.items, snapshot_has_edges, &get_edge_type_from_id,
|
||||
&highest_edge_gid, &recovery_info](const size_t batch_index, const BatchInfo &batch) {
|
||||
const auto result = LoadPartialConnectivity(path, *vertices, *edges, batch.offset, batch.count, items,
|
||||
snapshot_has_edges, get_edge_type_from_id);
|
||||
edge_count->fetch_add(result.edge_count);
|
||||
auto known_highest_edge_gid = highest_edge_gid.load();
|
||||
while (known_highest_edge_gid < result.highest_edge_id) {
|
||||
highest_edge_gid.compare_exchange_weak(known_highest_edge_gid, result.highest_edge_id);
|
||||
}
|
||||
recovery_info.vertex_batches[batch_index].first = result.first_vertex_gid;
|
||||
},
|
||||
vertex_batches);
|
||||
|
||||
spdlog::info("Connectivity is recovered.");
|
||||
|
||||
// Set initial values for edge/vertex ID generators.
|
||||
recovery_info.next_edge_id = highest_edge_gid + 1;
|
||||
recovery_info.next_vertex_id = last_vertex_gid + 1;
|
||||
}
|
||||
|
||||
// Recover indices.
|
||||
{
|
||||
spdlog::info("Recovering metadata of indices.");
|
||||
if (!snapshot.SetPosition(info.offset_indices)) throw RecoveryFailure("Couldn't read data from snapshot!");
|
||||
|
||||
auto marker = snapshot.ReadMarker();
|
||||
if (!marker || *marker != Marker::SECTION_INDICES) throw RecoveryFailure("Invalid snapshot data!");
|
||||
|
||||
// Recover label indices.
|
||||
{
|
||||
auto size = snapshot.ReadUint();
|
||||
if (!size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
spdlog::info("Recovering metadata of {} label indices.", *size);
|
||||
for (uint64_t i = 0; i < *size; ++i) {
|
||||
auto label = snapshot.ReadUint();
|
||||
if (!label) throw RecoveryFailure("Invalid snapshot data!");
|
||||
AddRecoveredIndexConstraint(&indices_constraints.indices.label, get_label_from_id(*label),
|
||||
"The label index already exists!");
|
||||
SPDLOG_TRACE("Recovered metadata of label index for :{}", name_id_mapper->IdToName(snapshot_id_map.at(*label)));
|
||||
}
|
||||
spdlog::info("Metadata of label indices are recovered.");
|
||||
}
|
||||
|
||||
// Recover label+property indices.
|
||||
{
|
||||
auto size = snapshot.ReadUint();
|
||||
if (!size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
spdlog::info("Recovering metadata of {} label+property indices.", *size);
|
||||
for (uint64_t i = 0; i < *size; ++i) {
|
||||
auto label = snapshot.ReadUint();
|
||||
if (!label) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto property = snapshot.ReadUint();
|
||||
if (!property) throw RecoveryFailure("Invalid snapshot data!");
|
||||
AddRecoveredIndexConstraint(&indices_constraints.indices.label_property,
|
||||
{get_label_from_id(*label), get_property_from_id(*property)},
|
||||
"The label+property index already exists!");
|
||||
SPDLOG_TRACE("Recovered metadata of label+property index for :{}({})",
|
||||
name_id_mapper->IdToName(snapshot_id_map.at(*label)),
|
||||
name_id_mapper->IdToName(snapshot_id_map.at(*property)));
|
||||
}
|
||||
spdlog::info("Metadata of label+property indices are recovered.");
|
||||
}
|
||||
spdlog::info("Metadata of indices are recovered.");
|
||||
}
|
||||
|
||||
// Recover constraints.
|
||||
{
|
||||
spdlog::info("Recovering metadata of constraints.");
|
||||
if (!snapshot.SetPosition(info.offset_constraints)) throw RecoveryFailure("Couldn't read data from snapshot!");
|
||||
|
||||
auto marker = snapshot.ReadMarker();
|
||||
if (!marker || *marker != Marker::SECTION_CONSTRAINTS) throw RecoveryFailure("Invalid snapshot data!");
|
||||
|
||||
// Recover existence constraints.
|
||||
{
|
||||
auto size = snapshot.ReadUint();
|
||||
if (!size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
spdlog::info("Recovering metadata of {} existence constraints.", *size);
|
||||
for (uint64_t i = 0; i < *size; ++i) {
|
||||
auto label = snapshot.ReadUint();
|
||||
if (!label) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto property = snapshot.ReadUint();
|
||||
if (!property) throw RecoveryFailure("Invalid snapshot data!");
|
||||
AddRecoveredIndexConstraint(&indices_constraints.constraints.existence,
|
||||
{get_label_from_id(*label), get_property_from_id(*property)},
|
||||
"The existence constraint already exists!");
|
||||
SPDLOG_TRACE("Recovered metadata of existence constraint for :{}({})",
|
||||
name_id_mapper->IdToName(snapshot_id_map.at(*label)),
|
||||
name_id_mapper->IdToName(snapshot_id_map.at(*property)));
|
||||
}
|
||||
spdlog::info("Metadata of existence constraints are recovered.");
|
||||
}
|
||||
|
||||
// Recover unique constraints.
|
||||
// Snapshot version should be checked since unique constraints were
|
||||
// implemented in later versions of snapshot.
|
||||
if (*version >= kUniqueConstraintVersion) {
|
||||
auto size = snapshot.ReadUint();
|
||||
if (!size) throw RecoveryFailure("Invalid snapshot data!");
|
||||
spdlog::info("Recovering metadata of {} unique constraints.", *size);
|
||||
for (uint64_t i = 0; i < *size; ++i) {
|
||||
auto label = snapshot.ReadUint();
|
||||
if (!label) throw RecoveryFailure("Invalid snapshot data!");
|
||||
auto properties_count = snapshot.ReadUint();
|
||||
if (!properties_count) throw RecoveryFailure("Invalid snapshot data!");
|
||||
std::set<PropertyId> properties;
|
||||
for (uint64_t j = 0; j < *properties_count; ++j) {
|
||||
auto property = snapshot.ReadUint();
|
||||
if (!property) throw RecoveryFailure("Invalid snapshot data!");
|
||||
properties.insert(get_property_from_id(*property));
|
||||
}
|
||||
AddRecoveredIndexConstraint(&indices_constraints.constraints.unique, {get_label_from_id(*label), properties},
|
||||
"The unique constraint already exists!");
|
||||
SPDLOG_TRACE("Recovered metadata of unique constraints for :{}",
|
||||
name_id_mapper->IdToName(snapshot_id_map.at(*label)));
|
||||
}
|
||||
spdlog::info("Metadata of unique constraints are recovered.");
|
||||
}
|
||||
spdlog::info("Metadata of constraints are recovered.");
|
||||
}
|
||||
|
||||
spdlog::info("Recovering metadata.");
|
||||
// Recover epoch history
|
||||
{
|
||||
if (!snapshot.SetPosition(info.offset_epoch_history)) throw RecoveryFailure("Couldn't read data from snapshot!");
|
||||
|
||||
const auto marker = snapshot.ReadMarker();
|
||||
if (!marker || *marker != Marker::SECTION_EPOCH_HISTORY) throw RecoveryFailure("Invalid snapshot data!");
|
||||
|
||||
const auto history_size = snapshot.ReadUint();
|
||||
if (!history_size) {
|
||||
throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
|
||||
for (int i = 0; i < *history_size; ++i) {
|
||||
auto maybe_epoch_id = snapshot.ReadString();
|
||||
if (!maybe_epoch_id) {
|
||||
throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
const auto maybe_last_commit_timestamp = snapshot.ReadUint();
|
||||
if (!maybe_last_commit_timestamp) {
|
||||
throw RecoveryFailure("Invalid snapshot data!");
|
||||
}
|
||||
epoch_history->emplace_back(std::move(*maybe_epoch_id), *maybe_last_commit_timestamp);
|
||||
}
|
||||
}
|
||||
|
||||
spdlog::info("Metadata recovered.");
|
||||
// Recover timestamp.
|
||||
recovery_info.next_timestamp = info.start_timestamp + 1;
|
||||
|
||||
// Set success flag (to disable cleanup).
|
||||
success = true;
|
||||
|
||||
return {info, recovery_info, std::move(indices_constraints)};
|
||||
}
|
||||
|
||||
void CreateSnapshot(Transaction *transaction, const std::filesystem::path &snapshot_directory,
|
||||
const std::filesystem::path &wal_directory, uint64_t snapshot_retention_count,
|
||||
utils::SkipList<Vertex> *vertices, utils::SkipList<Edge> *edges, NameIdMapper *name_id_mapper,
|
||||
Indices *indices, Constraints *constraints, Config::Items items, const std::string &uuid,
|
||||
Indices *indices, Constraints *constraints, const Config &config, const std::string &uuid,
|
||||
const std::string_view epoch_id, const std::deque<std::pair<std::string, uint64_t>> &epoch_history,
|
||||
utils::FileRetainer *file_retainer) {
|
||||
// Ensure that the storage directory exists.
|
||||
@@ -649,6 +1335,8 @@ void CreateSnapshot(Transaction *transaction, const std::filesystem::path &snaps
|
||||
uint64_t offset_mapper = 0;
|
||||
uint64_t offset_metadata = 0;
|
||||
uint64_t offset_epoch_history = 0;
|
||||
uint64_t offset_edge_batches = 0;
|
||||
uint64_t offset_vertex_batches = 0;
|
||||
{
|
||||
snapshot.WriteMarker(Marker::SECTION_OFFSETS);
|
||||
offset_offsets = snapshot.GetPosition();
|
||||
@@ -659,6 +1347,8 @@ void CreateSnapshot(Transaction *transaction, const std::filesystem::path &snaps
|
||||
snapshot.WriteUint(offset_mapper);
|
||||
snapshot.WriteUint(offset_epoch_history);
|
||||
snapshot.WriteUint(offset_metadata);
|
||||
snapshot.WriteUint(offset_edge_batches);
|
||||
snapshot.WriteUint(offset_vertex_batches);
|
||||
}
|
||||
|
||||
// Object counters.
|
||||
@@ -672,9 +1362,13 @@ void CreateSnapshot(Transaction *transaction, const std::filesystem::path &snaps
|
||||
snapshot.WriteUint(mapping.AsUint());
|
||||
};
|
||||
|
||||
std::vector<BatchInfo> edge_batch_infos;
|
||||
auto items_in_current_batch{0UL};
|
||||
auto batch_start_offset{0UL};
|
||||
// Store all edges.
|
||||
if (items.properties_on_edges) {
|
||||
if (config.items.properties_on_edges) {
|
||||
offset_edges = snapshot.GetPosition();
|
||||
batch_start_offset = offset_edges;
|
||||
auto acc = edges->access();
|
||||
for (auto &edge : acc) {
|
||||
// The edge visibility check must be done here manually because we don't
|
||||
@@ -713,8 +1407,8 @@ void CreateSnapshot(Transaction *transaction, const std::filesystem::path &snaps
|
||||
// type and invalid from/to pointers because we don't know them here,
|
||||
// but that isn't an issue because we won't use that part of the API
|
||||
// here.
|
||||
auto ea =
|
||||
EdgeAccessor{edge_ref, EdgeTypeId::FromUint(0UL), nullptr, nullptr, transaction, indices, constraints, items};
|
||||
auto ea = EdgeAccessor{
|
||||
edge_ref, EdgeTypeId::FromUint(0UL), nullptr, nullptr, transaction, indices, constraints, config.items};
|
||||
|
||||
// Get edge data.
|
||||
auto maybe_props = ea.Properties(View::OLD);
|
||||
@@ -733,16 +1427,29 @@ void CreateSnapshot(Transaction *transaction, const std::filesystem::path &snaps
|
||||
}
|
||||
|
||||
++edges_count;
|
||||
++items_in_current_batch;
|
||||
if (items_in_current_batch == config.durability.items_per_batch) {
|
||||
edge_batch_infos.push_back(BatchInfo{batch_start_offset, items_in_current_batch});
|
||||
batch_start_offset = snapshot.GetPosition();
|
||||
items_in_current_batch = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (items_in_current_batch > 0) {
|
||||
edge_batch_infos.push_back(BatchInfo{batch_start_offset, items_in_current_batch});
|
||||
}
|
||||
|
||||
std::vector<BatchInfo> vertex_batch_infos;
|
||||
// Store all vertices.
|
||||
{
|
||||
items_in_current_batch = 0;
|
||||
offset_vertices = snapshot.GetPosition();
|
||||
batch_start_offset = offset_vertices;
|
||||
auto acc = vertices->access();
|
||||
for (auto &vertex : acc) {
|
||||
// The visibility check is implemented for vertices so we use it here.
|
||||
auto va = VertexAccessor::Create(&vertex, transaction, indices, constraints, items, View::OLD);
|
||||
auto va = VertexAccessor::Create(&vertex, transaction, indices, constraints, config.items, View::OLD);
|
||||
if (!va) continue;
|
||||
|
||||
// Get vertex data.
|
||||
@@ -789,6 +1496,16 @@ void CreateSnapshot(Transaction *transaction, const std::filesystem::path &snaps
|
||||
}
|
||||
|
||||
++vertices_count;
|
||||
++items_in_current_batch;
|
||||
if (items_in_current_batch == config.durability.items_per_batch) {
|
||||
vertex_batch_infos.push_back(BatchInfo{batch_start_offset, items_in_current_batch});
|
||||
batch_start_offset = snapshot.GetPosition();
|
||||
items_in_current_batch = 0;
|
||||
}
|
||||
}
|
||||
|
||||
if (items_in_current_batch > 0) {
|
||||
vertex_batch_infos.push_back(BatchInfo{batch_start_offset, items_in_current_batch});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -879,6 +1596,26 @@ void CreateSnapshot(Transaction *transaction, const std::filesystem::path &snaps
|
||||
snapshot.WriteUint(vertices_count);
|
||||
}
|
||||
|
||||
auto write_batch_infos = [&snapshot](const std::vector<BatchInfo> &batch_infos) {
|
||||
snapshot.WriteUint(batch_infos.size());
|
||||
for (const auto &batch_info : batch_infos) {
|
||||
snapshot.WriteUint(batch_info.offset);
|
||||
snapshot.WriteUint(batch_info.count);
|
||||
}
|
||||
};
|
||||
|
||||
// Write edge batches
|
||||
{
|
||||
offset_edge_batches = snapshot.GetPosition();
|
||||
write_batch_infos(edge_batch_infos);
|
||||
}
|
||||
|
||||
// Write vertex batches
|
||||
{
|
||||
offset_vertex_batches = snapshot.GetPosition();
|
||||
write_batch_infos(vertex_batch_infos);
|
||||
}
|
||||
|
||||
// Write true offsets.
|
||||
{
|
||||
snapshot.SetPosition(offset_offsets);
|
||||
@@ -889,6 +1626,8 @@ void CreateSnapshot(Transaction *transaction, const std::filesystem::path &snaps
|
||||
snapshot.WriteUint(offset_mapper);
|
||||
snapshot.WriteUint(offset_epoch_history);
|
||||
snapshot.WriteUint(offset_metadata);
|
||||
snapshot.WriteUint(offset_edge_batches);
|
||||
snapshot.WriteUint(offset_vertex_batches);
|
||||
}
|
||||
|
||||
// Finalize snapshot file.
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -37,6 +37,8 @@ struct SnapshotInfo {
|
||||
uint64_t offset_mapper;
|
||||
uint64_t offset_epoch_history;
|
||||
uint64_t offset_metadata;
|
||||
uint64_t offset_edge_batches;
|
||||
uint64_t offset_vertex_batches;
|
||||
|
||||
std::string uuid;
|
||||
std::string epoch_id;
|
||||
@@ -62,13 +64,13 @@ SnapshotInfo ReadSnapshotInfo(const std::filesystem::path &path);
|
||||
RecoveredSnapshot LoadSnapshot(const std::filesystem::path &path, utils::SkipList<Vertex> *vertices,
|
||||
utils::SkipList<Edge> *edges,
|
||||
std::deque<std::pair<std::string, uint64_t>> *epoch_history,
|
||||
NameIdMapper *name_id_mapper, std::atomic<uint64_t> *edge_count, Config::Items items);
|
||||
NameIdMapper *name_id_mapper, std::atomic<uint64_t> *edge_count, const Config &config);
|
||||
|
||||
/// Function used to create a snapshot using the given transaction.
|
||||
void CreateSnapshot(Transaction *transaction, const std::filesystem::path &snapshot_directory,
|
||||
const std::filesystem::path &wal_directory, uint64_t snapshot_retention_count,
|
||||
utils::SkipList<Vertex> *vertices, utils::SkipList<Edge> *edges, NameIdMapper *name_id_mapper,
|
||||
Indices *indices, Constraints *constraints, Config::Items items, const std::string &uuid,
|
||||
Indices *indices, Constraints *constraints, const Config &config, const std::string &uuid,
|
||||
std::string_view epoch_id, const std::deque<std::pair<std::string, uint64_t>> &epoch_history,
|
||||
utils::FileRetainer *file_retainer);
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -20,7 +20,7 @@ namespace memgraph::storage::durability {
|
||||
// The current version of snapshot and WAL encoding / decoding.
|
||||
// IMPORTANT: Please bump this version for every snapshot and/or WAL format
|
||||
// change!!!
|
||||
const uint64_t kVersion{14};
|
||||
const uint64_t kVersion{15};
|
||||
|
||||
const uint64_t kOldestSupportedVersion{14};
|
||||
const uint64_t kUniqueConstraintVersion{13};
|
||||
|
||||
@@ -13,12 +13,14 @@
|
||||
#include <algorithm>
|
||||
#include <iterator>
|
||||
#include <limits>
|
||||
#include <thread>
|
||||
|
||||
#include "storage/v2/mvcc.hpp"
|
||||
#include "storage/v2/property_value.hpp"
|
||||
#include "utils/bound.hpp"
|
||||
#include "utils/logging.hpp"
|
||||
#include "utils/memory_tracker.hpp"
|
||||
#include "utils/synchronized.hpp"
|
||||
|
||||
namespace memgraph::storage {
|
||||
|
||||
@@ -263,6 +265,95 @@ bool CurrentVersionHasLabelProperty(const Vertex &vertex, LabelId label, Propert
|
||||
return !deleted && has_label && current_value_equal_to_value;
|
||||
}
|
||||
|
||||
template <typename TIndexAccessor>
|
||||
void TryInsertLabelIndex(Vertex &vertex, LabelId label, TIndexAccessor &index_accessor) {
|
||||
if (vertex.deleted || !utils::Contains(vertex.labels, label)) {
|
||||
return;
|
||||
}
|
||||
|
||||
index_accessor.insert({&vertex, 0});
|
||||
}
|
||||
|
||||
template <typename TIndexAccessor>
|
||||
void TryInsertLabelPropertyIndex(Vertex &vertex, std::pair<LabelId, PropertyId> label_property_pair,
|
||||
TIndexAccessor &index_accessor) {
|
||||
if (vertex.deleted || !utils::Contains(vertex.labels, label_property_pair.first)) {
|
||||
return;
|
||||
}
|
||||
auto value = vertex.properties.GetProperty(label_property_pair.second);
|
||||
if (value.IsNull()) {
|
||||
return;
|
||||
}
|
||||
index_accessor.insert({std::move(value), &vertex, 0});
|
||||
}
|
||||
|
||||
template <typename TSkiplistIter, typename TIndex, typename TIndexKey, typename TFunc>
|
||||
void CreateIndexOnSingleThread(utils::SkipList<Vertex>::Accessor &vertices, TSkiplistIter it, TIndex &index,
|
||||
TIndexKey key, const TFunc &func) {
|
||||
utils::MemoryTracker::OutOfMemoryExceptionEnabler oom_exception;
|
||||
try {
|
||||
auto acc = it->second.access();
|
||||
for (Vertex &vertex : vertices) {
|
||||
func(vertex, key, acc);
|
||||
}
|
||||
} catch (const utils::OutOfMemoryException &) {
|
||||
utils::MemoryTracker::OutOfMemoryExceptionBlocker oom_exception_blocker;
|
||||
index.erase(it);
|
||||
throw;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename TIndex, typename TIndexKey, typename TSKiplistIter, typename TFunc>
|
||||
void CreateIndexOnMultipleThreads(utils::SkipList<Vertex>::Accessor &vertices, TSKiplistIter skiplist_iter,
|
||||
TIndex &index, TIndexKey key, const ParalellizedIndexCreationInfo ¶lell_exec_info,
|
||||
const TFunc &func) {
|
||||
utils::MemoryTracker::OutOfMemoryExceptionEnabler oom_exception;
|
||||
|
||||
const auto &vertex_batches = paralell_exec_info.first;
|
||||
const auto thread_count = std::min(paralell_exec_info.second, vertex_batches.size());
|
||||
|
||||
MG_ASSERT(!vertex_batches.empty(),
|
||||
"The size of batches should always be greater than zero if you want to use the parallel version of index "
|
||||
"creation!");
|
||||
|
||||
std::atomic<uint64_t> batch_counter = 0;
|
||||
|
||||
utils::Synchronized<std::optional<utils::OutOfMemoryException>, utils::SpinLock> maybe_error{};
|
||||
{
|
||||
std::vector<std::jthread> threads;
|
||||
threads.reserve(thread_count);
|
||||
|
||||
for (auto i{0U}; i < thread_count; ++i) {
|
||||
threads.emplace_back(
|
||||
[&skiplist_iter, &func, &index, &vertex_batches, &maybe_error, &batch_counter, &key, &vertices]() {
|
||||
while (!maybe_error.Lock()->has_value()) {
|
||||
const auto batch_index = batch_counter++;
|
||||
if (batch_index >= vertex_batches.size()) {
|
||||
return;
|
||||
}
|
||||
const auto &batch = vertex_batches[batch_index];
|
||||
auto index_accessor = index.at(key).access();
|
||||
auto it = vertices.find(batch.first);
|
||||
|
||||
try {
|
||||
for (auto i{0U}; i < batch.second; ++i, ++it) {
|
||||
func(*it, key, index_accessor);
|
||||
}
|
||||
|
||||
} catch (utils::OutOfMemoryException &failure) {
|
||||
utils::MemoryTracker::OutOfMemoryExceptionBlocker oom_exception_blocker;
|
||||
index.erase(skiplist_iter);
|
||||
*maybe_error.Lock() = std::move(failure);
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
if (maybe_error.Lock()->has_value()) {
|
||||
throw utils::OutOfMemoryException((*maybe_error.Lock())->what());
|
||||
}
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
void LabelIndex::UpdateOnAddLabel(LabelId label, Vertex *vertex, const Transaction &tx) {
|
||||
@@ -272,27 +363,43 @@ void LabelIndex::UpdateOnAddLabel(LabelId label, Vertex *vertex, const Transacti
|
||||
acc.insert(Entry{vertex, tx.start_timestamp});
|
||||
}
|
||||
|
||||
bool LabelIndex::CreateIndex(LabelId label, utils::SkipList<Vertex>::Accessor vertices) {
|
||||
utils::MemoryTracker::OutOfMemoryExceptionEnabler oom_exception;
|
||||
bool LabelIndex::CreateIndex(LabelId label, utils::SkipList<Vertex>::Accessor vertices,
|
||||
const std::optional<ParalellizedIndexCreationInfo> ¶lell_exec_info) {
|
||||
auto create_index_seq = [this](LabelId label, utils::SkipList<Vertex>::Accessor &vertices,
|
||||
std::map<LabelId, utils::SkipList<Entry>>::iterator it) {
|
||||
using IndexAccessor = decltype(it->second.access());
|
||||
|
||||
CreateIndexOnSingleThread(vertices, it, index_, label,
|
||||
[](Vertex &vertex, LabelId label, IndexAccessor &index_accessor) {
|
||||
TryInsertLabelIndex(vertex, label, index_accessor);
|
||||
});
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
auto create_index_par = [this](LabelId label, utils::SkipList<Vertex>::Accessor &vertices,
|
||||
std::map<LabelId, utils::SkipList<Entry>>::iterator label_it,
|
||||
const ParalellizedIndexCreationInfo ¶lell_exec_info) {
|
||||
using IndexAccessor = decltype(label_it->second.access());
|
||||
|
||||
CreateIndexOnMultipleThreads(vertices, label_it, index_, label, paralell_exec_info,
|
||||
[](Vertex &vertex, LabelId label, IndexAccessor &index_accessor) {
|
||||
TryInsertLabelIndex(vertex, label, index_accessor);
|
||||
});
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
auto [it, emplaced] = index_.emplace(std::piecewise_construct, std::forward_as_tuple(label), std::forward_as_tuple());
|
||||
if (!emplaced) {
|
||||
// Index already exists.
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
auto acc = it->second.access();
|
||||
for (Vertex &vertex : vertices) {
|
||||
if (vertex.deleted || !utils::Contains(vertex.labels, label)) {
|
||||
continue;
|
||||
}
|
||||
acc.insert(Entry{&vertex, 0});
|
||||
}
|
||||
} catch (const utils::OutOfMemoryException &) {
|
||||
utils::MemoryTracker::OutOfMemoryExceptionBlocker oom_exception_blocker;
|
||||
index_.erase(it);
|
||||
throw;
|
||||
|
||||
if (paralell_exec_info) {
|
||||
return create_index_par(label, vertices, it, *paralell_exec_info);
|
||||
}
|
||||
return true;
|
||||
return create_index_seq(label, vertices, it);
|
||||
}
|
||||
|
||||
std::vector<LabelId> LabelIndex::ListIndices() const {
|
||||
@@ -418,32 +525,46 @@ void LabelPropertyIndex::UpdateOnSetProperty(PropertyId property, const Property
|
||||
}
|
||||
}
|
||||
|
||||
bool LabelPropertyIndex::CreateIndex(LabelId label, PropertyId property, utils::SkipList<Vertex>::Accessor vertices) {
|
||||
utils::MemoryTracker::OutOfMemoryExceptionEnabler oom_exception;
|
||||
bool LabelPropertyIndex::CreateIndex(LabelId label, PropertyId property, utils::SkipList<Vertex>::Accessor vertices,
|
||||
const std::optional<ParalellizedIndexCreationInfo> ¶lell_exec_info) {
|
||||
auto create_index_seq = [this](LabelId label, PropertyId property, utils::SkipList<Vertex>::Accessor &vertices,
|
||||
std::map<std::pair<LabelId, PropertyId>, utils::SkipList<Entry>>::iterator it) {
|
||||
using IndexAccessor = decltype(it->second.access());
|
||||
|
||||
CreateIndexOnSingleThread(vertices, it, index_, std::make_pair(label, property),
|
||||
[](Vertex &vertex, std::pair<LabelId, PropertyId> key, IndexAccessor &index_accessor) {
|
||||
TryInsertLabelPropertyIndex(vertex, key, index_accessor);
|
||||
});
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
auto create_index_par =
|
||||
[this](LabelId label, PropertyId property, utils::SkipList<Vertex>::Accessor &vertices,
|
||||
std::map<std::pair<LabelId, PropertyId>, utils::SkipList<Entry>>::iterator label_property_it,
|
||||
const ParalellizedIndexCreationInfo ¶lell_exec_info) {
|
||||
using IndexAccessor = decltype(label_property_it->second.access());
|
||||
|
||||
CreateIndexOnMultipleThreads(
|
||||
vertices, label_property_it, index_, std::make_pair(label, property), paralell_exec_info,
|
||||
[](Vertex &vertex, std::pair<LabelId, PropertyId> key, IndexAccessor &index_accessor) {
|
||||
TryInsertLabelPropertyIndex(vertex, key, index_accessor);
|
||||
});
|
||||
|
||||
return true;
|
||||
};
|
||||
|
||||
auto [it, emplaced] =
|
||||
index_.emplace(std::piecewise_construct, std::forward_as_tuple(label, property), std::forward_as_tuple());
|
||||
if (!emplaced) {
|
||||
// Index already exists.
|
||||
return false;
|
||||
}
|
||||
try {
|
||||
auto acc = it->second.access();
|
||||
for (Vertex &vertex : vertices) {
|
||||
if (vertex.deleted || !utils::Contains(vertex.labels, label)) {
|
||||
continue;
|
||||
}
|
||||
auto value = vertex.properties.GetProperty(property);
|
||||
if (value.IsNull()) {
|
||||
continue;
|
||||
}
|
||||
acc.insert(Entry{std::move(value), &vertex, 0});
|
||||
}
|
||||
} catch (const utils::OutOfMemoryException &) {
|
||||
utils::MemoryTracker::OutOfMemoryExceptionBlocker oom_exception_blocker;
|
||||
index_.erase(it);
|
||||
throw;
|
||||
|
||||
if (paralell_exec_info) {
|
||||
return create_index_par(label, property, vertices, it, *paralell_exec_info);
|
||||
}
|
||||
return true;
|
||||
return create_index_seq(label, property, vertices, it);
|
||||
}
|
||||
|
||||
std::vector<std::pair<LabelId, PropertyId>> LabelPropertyIndex::ListIndices() const {
|
||||
|
||||
@@ -28,6 +28,9 @@ namespace memgraph::storage {
|
||||
struct Indices;
|
||||
struct Constraints;
|
||||
|
||||
using ParalellizedIndexCreationInfo =
|
||||
std::pair<std::vector<std::pair<Gid, uint64_t>> /*vertex_recovery_info*/, uint64_t /*thread_count*/>;
|
||||
|
||||
class LabelIndex {
|
||||
private:
|
||||
struct Entry {
|
||||
@@ -58,7 +61,8 @@ class LabelIndex {
|
||||
void UpdateOnAddLabel(LabelId label, Vertex *vertex, const Transaction &tx);
|
||||
|
||||
/// @throw std::bad_alloc
|
||||
bool CreateIndex(LabelId label, utils::SkipList<Vertex>::Accessor vertices);
|
||||
bool CreateIndex(LabelId label, utils::SkipList<Vertex>::Accessor vertices,
|
||||
const std::optional<ParalellizedIndexCreationInfo> ¶lell_exec_info = std::nullopt);
|
||||
|
||||
/// Returns false if there was no index to drop
|
||||
bool DropIndex(LabelId label) { return index_.erase(label) > 0; }
|
||||
@@ -160,7 +164,8 @@ class LabelPropertyIndex {
|
||||
void UpdateOnSetProperty(PropertyId property, const PropertyValue &value, Vertex *vertex, const Transaction &tx);
|
||||
|
||||
/// @throw std::bad_alloc
|
||||
bool CreateIndex(LabelId label, PropertyId property, utils::SkipList<Vertex>::Accessor vertices);
|
||||
bool CreateIndex(LabelId label, PropertyId property, utils::SkipList<Vertex>::Accessor vertices,
|
||||
const std::optional<ParalellizedIndexCreationInfo> ¶lell_exec_info = std::nullopt);
|
||||
|
||||
bool DropIndex(LabelId label, PropertyId property) { return index_.erase({label, property}) > 0; }
|
||||
|
||||
|
||||
@@ -1144,7 +1144,8 @@ bool PropertyStore::SetProperty(PropertyId property, const PropertyValue &value)
|
||||
return !existed;
|
||||
}
|
||||
|
||||
bool PropertyStore::InitProperties(const std::map<storage::PropertyId, storage::PropertyValue> &properties) {
|
||||
template <typename TContainer>
|
||||
bool PropertyStore::DoInitProperties(const TContainer &properties) {
|
||||
uint64_t size = 0;
|
||||
uint8_t *data = nullptr;
|
||||
std::tie(size, data) = GetSizeData(buffer_);
|
||||
@@ -1201,6 +1202,20 @@ bool PropertyStore::InitProperties(const std::map<storage::PropertyId, storage::
|
||||
|
||||
return true;
|
||||
}
|
||||
template bool PropertyStore::DoInitProperties<std::map<PropertyId, PropertyValue>>(
|
||||
const std::map<PropertyId, PropertyValue> &);
|
||||
template bool PropertyStore::DoInitProperties<std::vector<std::pair<PropertyId, PropertyValue>>>(
|
||||
const std::vector<std::pair<PropertyId, PropertyValue>> &);
|
||||
|
||||
bool PropertyStore::InitProperties(const std::map<storage::PropertyId, storage::PropertyValue> &properties) {
|
||||
return DoInitProperties(properties);
|
||||
}
|
||||
|
||||
bool PropertyStore::InitProperties(std::vector<std::pair<storage::PropertyId, storage::PropertyValue>> properties) {
|
||||
std::sort(properties.begin(), properties.end());
|
||||
|
||||
return DoInitProperties(properties);
|
||||
}
|
||||
|
||||
bool PropertyStore::ClearProperties() {
|
||||
bool in_local_buffer = false;
|
||||
|
||||
@@ -60,11 +60,17 @@ class PropertyStore {
|
||||
bool SetProperty(PropertyId property, const PropertyValue &value);
|
||||
|
||||
/// Init property values and return `true` if insertion took place. `false` is
|
||||
/// returned if there exists property in property store and insertion couldn't take place. The time complexity of this
|
||||
/// function is O(n).
|
||||
/// returned if there is any existing property in property store and insertion couldn't take place. The time
|
||||
/// complexity of this function is O(n).
|
||||
/// @throw std::bad_alloc
|
||||
bool InitProperties(const std::map<storage::PropertyId, storage::PropertyValue> &properties);
|
||||
|
||||
/// Init property values and return `true` if insertion took place. `false` is
|
||||
/// returned if there is any existing property in property store and insertion couldn't take place. The time
|
||||
/// complexity of this function is O(n*log(n)):
|
||||
/// @throw std::bad_alloc
|
||||
bool InitProperties(std::vector<std::pair<storage::PropertyId, storage::PropertyValue>> properties);
|
||||
|
||||
/// Remove all properties and return `true` if any removal took place.
|
||||
/// `false` is returned if there were no properties to remove. The time
|
||||
/// complexity of this function is O(1).
|
||||
@@ -72,6 +78,9 @@ class PropertyStore {
|
||||
bool ClearProperties();
|
||||
|
||||
private:
|
||||
template <typename TContainer>
|
||||
bool DoInitProperties(const TContainer &properties);
|
||||
|
||||
uint8_t buffer_[sizeof(uint64_t) + sizeof(uint8_t *)];
|
||||
};
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -173,7 +173,7 @@ void Storage::ReplicationServer::SnapshotHandler(slk::Reader *req_reader, slk::B
|
||||
spdlog::debug("Loading snapshot");
|
||||
auto recovered_snapshot = durability::LoadSnapshot(*maybe_snapshot_path, &storage_->vertices_, &storage_->edges_,
|
||||
&storage_->epoch_history_, &storage_->name_id_mapper_,
|
||||
&storage_->edge_count_, storage_->config_.items);
|
||||
&storage_->edge_count_, storage_->config_);
|
||||
spdlog::debug("Snapshot loaded successfully");
|
||||
// If this step is present it should always be the first step of
|
||||
// the recovery so we use the UUID we read from snasphost
|
||||
|
||||
@@ -12,6 +12,7 @@
|
||||
#include "storage/v2/storage.hpp"
|
||||
#include <algorithm>
|
||||
#include <atomic>
|
||||
#include <limits>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <variant>
|
||||
@@ -360,7 +361,7 @@ Storage::Storage(Config config)
|
||||
if (config_.durability.recover_on_startup) {
|
||||
auto info = durability::RecoverData(snapshot_directory_, wal_directory_, &uuid_, &epoch_id_, &epoch_history_,
|
||||
&vertices_, &edges_, &edge_count_, &name_id_mapper_, &indices_, &constraints_,
|
||||
config_.items, &wal_seq_num_);
|
||||
config_, &wal_seq_num_);
|
||||
if (info) {
|
||||
vertex_id_ = info->next_vertex_id;
|
||||
edge_id_ = info->next_edge_id;
|
||||
@@ -1439,11 +1440,17 @@ void Storage::CollectGarbage() {
|
||||
if constexpr (force) {
|
||||
// We take the unique lock on the main storage lock so we can forcefully clean
|
||||
// everything we can
|
||||
|
||||
// TODO(gvolfing) Verify this! This should not just try, it should lock() always.
|
||||
if (!main_lock_.try_lock()) {
|
||||
CollectGarbage<false>();
|
||||
return;
|
||||
}
|
||||
|
||||
} else {
|
||||
// DEBUG
|
||||
// return;
|
||||
|
||||
// Because the garbage collector iterates through the indices and constraints
|
||||
// to clean them up, it must take the main lock for reading to make sure that
|
||||
// the indices and constraints aren't concurrently being modified.
|
||||
@@ -1484,142 +1491,169 @@ void Storage::CollectGarbage() {
|
||||
deleted_vertices_->swap(current_deleted_vertices);
|
||||
deleted_edges_->swap(current_deleted_edges);
|
||||
|
||||
const bool is_analytical_storage_mode = storage_mode_ == StorageMode::IN_MEMORY_ANALYTICAL;
|
||||
|
||||
// Flag that will be used to determine whether the Index GC should be run. It
|
||||
// should be run when there were any items that were cleaned up (there were
|
||||
// updates between this run of the GC and the previous run of the GC). This
|
||||
// eliminates high CPU usage when the GC doesn't have to clean up anything.
|
||||
bool run_index_cleanup = !committed_transactions_->empty() || !garbage_undo_buffers_->empty();
|
||||
// bool run_index_cleanup = !committed_transactions_->empty() || !garbage_undo_buffers_->empty();
|
||||
bool run_index_cleanup =
|
||||
!committed_transactions_->empty() || !garbage_undo_buffers_->empty() || is_analytical_storage_mode;
|
||||
|
||||
while (true) {
|
||||
// We don't want to hold the lock on commited transactions for too long,
|
||||
// because that prevents other transactions from committing.
|
||||
Transaction *transaction;
|
||||
{
|
||||
auto committed_transactions_ptr = committed_transactions_.Lock();
|
||||
if (committed_transactions_ptr->empty()) {
|
||||
if (is_analytical_storage_mode) {
|
||||
oldest_active_start_timestamp = std::numeric_limits<decltype(oldest_active_start_timestamp)>::max();
|
||||
|
||||
if constexpr (force) {
|
||||
auto vertices = vertices_.access();
|
||||
for (auto &vertex : vertices) {
|
||||
if (vertex.deleted) {
|
||||
current_deleted_vertices.push_back(vertex.gid);
|
||||
}
|
||||
}
|
||||
// TODO add edges as well
|
||||
|
||||
} else {
|
||||
// CollectGarbage<true>();
|
||||
// return;
|
||||
auto vertices = vertices_.access();
|
||||
for (auto &vertex : vertices) {
|
||||
if (vertex.deleted) {
|
||||
current_deleted_vertices.push_back(vertex.gid);
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
while (true) {
|
||||
// We don't want to hold the lock on commited transactions for too long,
|
||||
// because that prevents other transactions from committing.
|
||||
Transaction *transaction;
|
||||
{
|
||||
auto committed_transactions_ptr = committed_transactions_.Lock();
|
||||
if (committed_transactions_ptr->empty() || is_analytical_storage_mode) {
|
||||
break;
|
||||
}
|
||||
transaction = &committed_transactions_ptr->front();
|
||||
}
|
||||
|
||||
auto commit_timestamp = transaction->commit_timestamp->load(std::memory_order_acquire);
|
||||
if (commit_timestamp >= oldest_active_start_timestamp) {
|
||||
break;
|
||||
}
|
||||
transaction = &committed_transactions_ptr->front();
|
||||
}
|
||||
|
||||
auto commit_timestamp = transaction->commit_timestamp->load(std::memory_order_acquire);
|
||||
if (commit_timestamp >= oldest_active_start_timestamp) {
|
||||
break;
|
||||
}
|
||||
// When unlinking a delta which is the first delta in its version chain,
|
||||
// special care has to be taken to avoid the following race condition:
|
||||
//
|
||||
// [Vertex] --> [Delta A]
|
||||
//
|
||||
// GC thread: Delta A is the first in its chain, it must be unlinked from
|
||||
// vertex and marked for deletion
|
||||
// TX thread: Update vertex and add Delta B with Delta A as next
|
||||
//
|
||||
// [Vertex] --> [Delta B] <--> [Delta A]
|
||||
//
|
||||
// GC thread: Unlink delta from Vertex
|
||||
//
|
||||
// [Vertex] --> (nullptr)
|
||||
//
|
||||
// When processing a delta that is the first one in its chain, we
|
||||
// obtain the corresponding vertex or edge lock, and then verify that this
|
||||
// delta still is the first in its chain.
|
||||
// When processing a delta that is in the middle of the chain we only
|
||||
// process the final delta of the given transaction in that chain. We
|
||||
// determine the owner of the chain (either a vertex or an edge), obtain the
|
||||
// corresponding lock, and then verify that this delta is still in the same
|
||||
// position as it was before taking the lock.
|
||||
//
|
||||
// Even though the delta chain is lock-free (both `next` and `prev`) the
|
||||
// chain should not be modified without taking the lock from the object that
|
||||
// owns the chain (either a vertex or an edge). Modifying the chain without
|
||||
// taking the lock will cause subtle race conditions that will leave the
|
||||
// chain in a broken state.
|
||||
// The chain can be only read without taking any locks.
|
||||
|
||||
// When unlinking a delta which is the first delta in its version chain,
|
||||
// special care has to be taken to avoid the following race condition:
|
||||
//
|
||||
// [Vertex] --> [Delta A]
|
||||
//
|
||||
// GC thread: Delta A is the first in its chain, it must be unlinked from
|
||||
// vertex and marked for deletion
|
||||
// TX thread: Update vertex and add Delta B with Delta A as next
|
||||
//
|
||||
// [Vertex] --> [Delta B] <--> [Delta A]
|
||||
//
|
||||
// GC thread: Unlink delta from Vertex
|
||||
//
|
||||
// [Vertex] --> (nullptr)
|
||||
//
|
||||
// When processing a delta that is the first one in its chain, we
|
||||
// obtain the corresponding vertex or edge lock, and then verify that this
|
||||
// delta still is the first in its chain.
|
||||
// When processing a delta that is in the middle of the chain we only
|
||||
// process the final delta of the given transaction in that chain. We
|
||||
// determine the owner of the chain (either a vertex or an edge), obtain the
|
||||
// corresponding lock, and then verify that this delta is still in the same
|
||||
// position as it was before taking the lock.
|
||||
//
|
||||
// Even though the delta chain is lock-free (both `next` and `prev`) the
|
||||
// chain should not be modified without taking the lock from the object that
|
||||
// owns the chain (either a vertex or an edge). Modifying the chain without
|
||||
// taking the lock will cause subtle race conditions that will leave the
|
||||
// chain in a broken state.
|
||||
// The chain can be only read without taking any locks.
|
||||
|
||||
for (Delta &delta : transaction->deltas) {
|
||||
while (true) {
|
||||
auto prev = delta.prev.Get();
|
||||
switch (prev.type) {
|
||||
case PreviousPtr::Type::VERTEX: {
|
||||
Vertex *vertex = prev.vertex;
|
||||
std::lock_guard<utils::SpinLock> vertex_guard(vertex->lock);
|
||||
if (vertex->delta != &delta) {
|
||||
// Something changed, we're not the first delta in the chain
|
||||
// anymore.
|
||||
continue;
|
||||
}
|
||||
vertex->delta = nullptr;
|
||||
if (vertex->deleted) {
|
||||
current_deleted_vertices.push_back(vertex->gid);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case PreviousPtr::Type::EDGE: {
|
||||
Edge *edge = prev.edge;
|
||||
std::lock_guard<utils::SpinLock> edge_guard(edge->lock);
|
||||
if (edge->delta != &delta) {
|
||||
// Something changed, we're not the first delta in the chain
|
||||
// anymore.
|
||||
continue;
|
||||
}
|
||||
edge->delta = nullptr;
|
||||
if (edge->deleted) {
|
||||
current_deleted_edges.push_back(edge->gid);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case PreviousPtr::Type::DELTA: {
|
||||
if (prev.delta->timestamp->load(std::memory_order_acquire) == commit_timestamp) {
|
||||
// The delta that is newer than this one is also a delta from this
|
||||
// transaction. We skip the current delta and will remove it as a
|
||||
// part of the suffix later.
|
||||
for (Delta &delta : transaction->deltas) {
|
||||
while (true) {
|
||||
auto prev = delta.prev.Get();
|
||||
switch (prev.type) {
|
||||
case PreviousPtr::Type::VERTEX: {
|
||||
Vertex *vertex = prev.vertex;
|
||||
std::lock_guard<utils::SpinLock> vertex_guard(vertex->lock);
|
||||
if (vertex->delta != &delta) {
|
||||
// Something changed, we're not the first delta in the chain
|
||||
// anymore.
|
||||
continue;
|
||||
}
|
||||
vertex->delta = nullptr;
|
||||
if (vertex->deleted) {
|
||||
current_deleted_vertices.push_back(vertex->gid);
|
||||
}
|
||||
break;
|
||||
}
|
||||
std::unique_lock<utils::SpinLock> guard;
|
||||
{
|
||||
// We need to find the parent object in order to be able to use
|
||||
// its lock.
|
||||
auto parent = prev;
|
||||
while (parent.type == PreviousPtr::Type::DELTA) {
|
||||
parent = parent.delta->prev.Get();
|
||||
case PreviousPtr::Type::EDGE: {
|
||||
Edge *edge = prev.edge;
|
||||
std::lock_guard<utils::SpinLock> edge_guard(edge->lock);
|
||||
if (edge->delta != &delta) {
|
||||
// Something changed, we're not the first delta in the chain
|
||||
// anymore.
|
||||
continue;
|
||||
}
|
||||
switch (parent.type) {
|
||||
case PreviousPtr::Type::VERTEX:
|
||||
guard = std::unique_lock<utils::SpinLock>(parent.vertex->lock);
|
||||
break;
|
||||
case PreviousPtr::Type::EDGE:
|
||||
guard = std::unique_lock<utils::SpinLock>(parent.edge->lock);
|
||||
break;
|
||||
case PreviousPtr::Type::DELTA:
|
||||
case PreviousPtr::Type::NULLPTR:
|
||||
LOG_FATAL("Invalid database state!");
|
||||
edge->delta = nullptr;
|
||||
if (edge->deleted) {
|
||||
current_deleted_edges.push_back(edge->gid);
|
||||
}
|
||||
break;
|
||||
}
|
||||
if (delta.prev.Get() != prev) {
|
||||
// Something changed, we could now be the first delta in the
|
||||
// chain.
|
||||
continue;
|
||||
case PreviousPtr::Type::DELTA: {
|
||||
if (prev.delta->timestamp->load(std::memory_order_acquire) == commit_timestamp) {
|
||||
// The delta that is newer than this one is also a delta from this
|
||||
// transaction. We skip the current delta and will remove it as a
|
||||
// part of the suffix later.
|
||||
break;
|
||||
}
|
||||
std::unique_lock<utils::SpinLock> guard;
|
||||
{
|
||||
// We need to find the parent object in order to be able to use
|
||||
// its lock.
|
||||
auto parent = prev;
|
||||
while (parent.type == PreviousPtr::Type::DELTA) {
|
||||
parent = parent.delta->prev.Get();
|
||||
}
|
||||
switch (parent.type) {
|
||||
case PreviousPtr::Type::VERTEX:
|
||||
guard = std::unique_lock<utils::SpinLock>(parent.vertex->lock);
|
||||
break;
|
||||
case PreviousPtr::Type::EDGE:
|
||||
guard = std::unique_lock<utils::SpinLock>(parent.edge->lock);
|
||||
break;
|
||||
case PreviousPtr::Type::DELTA:
|
||||
case PreviousPtr::Type::NULLPTR:
|
||||
LOG_FATAL("Invalid database state!");
|
||||
}
|
||||
}
|
||||
if (delta.prev.Get() != prev) {
|
||||
// Something changed, we could now be the first delta in the
|
||||
// chain.
|
||||
continue;
|
||||
}
|
||||
Delta *prev_delta = prev.delta;
|
||||
prev_delta->next.store(nullptr, std::memory_order_release);
|
||||
break;
|
||||
}
|
||||
case PreviousPtr::Type::NULLPTR: {
|
||||
LOG_FATAL("Invalid pointer!");
|
||||
}
|
||||
Delta *prev_delta = prev.delta;
|
||||
prev_delta->next.store(nullptr, std::memory_order_release);
|
||||
break;
|
||||
}
|
||||
case PreviousPtr::Type::NULLPTR: {
|
||||
LOG_FATAL("Invalid pointer!");
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
committed_transactions_.WithLock([&](auto &committed_transactions) {
|
||||
unlinked_undo_buffers.emplace_back(0, std::move(transaction->deltas));
|
||||
committed_transactions.pop_front();
|
||||
});
|
||||
}
|
||||
|
||||
committed_transactions_.WithLock([&](auto &committed_transactions) {
|
||||
unlinked_undo_buffers.emplace_back(0, std::move(transaction->deltas));
|
||||
committed_transactions.pop_front();
|
||||
});
|
||||
}
|
||||
|
||||
// After unlinking deltas from vertices, we refresh the indices. That way
|
||||
// we're sure that none of the vertices from `current_deleted_vertices`
|
||||
// appears in an index, and we can safely remove the from the main storage
|
||||
@@ -1632,22 +1666,26 @@ void Storage::CollectGarbage() {
|
||||
}
|
||||
|
||||
{
|
||||
std::unique_lock<utils::SpinLock> guard(engine_lock_);
|
||||
uint64_t mark_timestamp = timestamp_;
|
||||
// Take garbage_undo_buffers lock while holding the engine lock to make
|
||||
// sure that entries are sorted by mark timestamp in the list.
|
||||
garbage_undo_buffers_.WithLock([&](auto &garbage_undo_buffers) {
|
||||
// Release engine lock because we don't have to hold it anymore and
|
||||
// this could take a long time.
|
||||
guard.unlock();
|
||||
// TODO(mtomic): holding garbage_undo_buffers_ lock here prevents
|
||||
// transactions from aborting until we're done marking, maybe we should
|
||||
// add them one-by-one or something
|
||||
for (auto &[timestamp, undo_buffer] : unlinked_undo_buffers) {
|
||||
timestamp = mark_timestamp;
|
||||
}
|
||||
garbage_undo_buffers.splice(garbage_undo_buffers.end(), unlinked_undo_buffers);
|
||||
});
|
||||
uint64_t mark_timestamp = 0;
|
||||
if (!is_analytical_storage_mode) {
|
||||
std::unique_lock<utils::SpinLock> guard(engine_lock_);
|
||||
// uint64_t mark_timestamp = timestamp_;
|
||||
mark_timestamp = timestamp_;
|
||||
// Take garbage_undo_buffers lock while holding the engine lock to make
|
||||
// sure that entries are sorted by mark timestamp in the list.
|
||||
garbage_undo_buffers_.WithLock([&](auto &garbage_undo_buffers) {
|
||||
// Release engine lock because we don't have to hold it anymore and
|
||||
// this could take a long time.
|
||||
guard.unlock();
|
||||
// TODO(mtomic): holding garbage_undo_buffers_ lock here prevents
|
||||
// transactions from aborting until we're done marking, maybe we should
|
||||
// add them one-by-one or something
|
||||
for (auto &[timestamp, undo_buffer] : unlinked_undo_buffers) {
|
||||
timestamp = mark_timestamp;
|
||||
}
|
||||
garbage_undo_buffers.splice(garbage_undo_buffers.end(), unlinked_undo_buffers);
|
||||
});
|
||||
}
|
||||
for (auto vertex : current_deleted_vertices) {
|
||||
garbage_vertices_.emplace_back(mark_timestamp, vertex);
|
||||
}
|
||||
@@ -1947,8 +1985,7 @@ utils::BasicResult<Storage::CreateSnapshotError> Storage::CreateSnapshot(std::op
|
||||
// Create snapshot.
|
||||
durability::CreateSnapshot(&transaction, snapshot_directory_, wal_directory_,
|
||||
config_.durability.snapshot_retention_count, &vertices_, &edges_, &name_id_mapper_,
|
||||
&indices_, &constraints_, config_.items, uuid_, epoch_id_, epoch_history_,
|
||||
&file_retainer_);
|
||||
&indices_, &constraints_, config_, uuid_, epoch_id_, epoch_history_, &file_retainer_);
|
||||
// Finalize snapshot transaction.
|
||||
commit_log_->MarkFinished(transaction.start_timestamp);
|
||||
};
|
||||
|
||||
@@ -1,3 +1,14 @@
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
// License, and you may not use this file except in compliance with the Business Source License.
|
||||
//
|
||||
// As of the Change Date specified in that file, in accordance with
|
||||
// the Business Source License, use of this software will be governed
|
||||
// by the Apache License, Version 2.0, included in the file
|
||||
// licenses/APL.txt.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
23
src/utils/pmr/deque.hpp
Normal file
23
src/utils/pmr/deque.hpp
Normal file
@@ -0,0 +1,23 @@
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
// License, and you may not use this file except in compliance with the Business Source License.
|
||||
//
|
||||
// As of the Change Date specified in that file, in accordance with
|
||||
// the Business Source License, use of this software will be governed
|
||||
// by the Apache License, Version 2.0, included in the file
|
||||
// licenses/APL.txt.
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <deque>
|
||||
|
||||
#include "utils/memory.hpp"
|
||||
|
||||
namespace memgraph::utils::pmr {
|
||||
|
||||
template <class T>
|
||||
using deque = std::deque<T, utils::Allocator<T>>;
|
||||
|
||||
} // namespace memgraph::utils::pmr
|
||||
@@ -38,7 +38,12 @@ def test_does_default_config_match():
|
||||
flag_name = flag[0]
|
||||
|
||||
# The default value of these is dependent on the given machine.
|
||||
machine_dependent_configurations = ["bolt_num_workers", "data_directory", "log_file"]
|
||||
machine_dependent_configurations = [
|
||||
"bolt_num_workers",
|
||||
"data_directory",
|
||||
"log_file",
|
||||
"storage_recovery_thread_count",
|
||||
]
|
||||
if flag_name in machine_dependent_configurations:
|
||||
continue
|
||||
|
||||
|
||||
@@ -98,6 +98,11 @@ startup_config_dict = {
|
||||
"IP address on which the websocket server for Memgraph monitoring should listen.",
|
||||
),
|
||||
"monitoring_port": ("7444", "7444", "Port on which the websocket server for Memgraph monitoring should listen."),
|
||||
"storage_parallel_index_recovery": (
|
||||
"false",
|
||||
"false",
|
||||
"Controls whether the index creation can be done in a multithreaded fashion.",
|
||||
),
|
||||
"password_encryption_algorithm": ("bcrypt", "bcrypt", "The password encryption algorithm used for authentication."),
|
||||
"pulsar_service_url": ("", "", "Default URL used while connecting to Pulsar brokers."),
|
||||
"query_execution_timeout_sec": (
|
||||
@@ -116,12 +121,18 @@ startup_config_dict = {
|
||||
"The time duration between two replica checks/pings. If < 1, replicas will NOT be checked at all. NOTE: The MAIN instance allocates a new thread for each REPLICA.",
|
||||
),
|
||||
"storage_gc_cycle_sec": ("30", "30", "Storage garbage collector interval (in seconds)."),
|
||||
"storage_items_per_batch": (
|
||||
"1000000",
|
||||
"1000000",
|
||||
"The number of edges and vertices stored in a batch in a snapshot file.",
|
||||
),
|
||||
"storage_properties_on_edges": ("false", "true", "Controls whether edges have properties."),
|
||||
"storage_recover_on_startup": (
|
||||
"false",
|
||||
"false",
|
||||
"Controls whether the storage recovers persisted data on startup.",
|
||||
),
|
||||
"storage_recovery_thread_count": ("12", "12", "The number of threads used to recover persisted data from disk."),
|
||||
"storage_snapshot_interval_sec": (
|
||||
"0",
|
||||
"300",
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
// --storage-items-per-batch is set to 10
|
||||
CREATE INDEX ON :`label2`(`prop2`);
|
||||
CREATE INDEX ON :`label2`(`prop`);
|
||||
CREATE INDEX ON :`label`;
|
||||
CREATE INDEX ON :__mg_vertex__(__mg_id__);
|
||||
CREATE (:__mg_vertex__:`label2` {__mg_id__: 0, `prop2`: ["kaj", 2, Null, {`prop4`: -1.341}], `ext`: 2, `prop`: "joj"});
|
||||
CREATE (:__mg_vertex__:`label2`:`label` {__mg_id__: 1, `ext`: 2, `prop`: "joj"});
|
||||
CREATE (:__mg_vertex__:`label2` {__mg_id__: 2, `prop2`: 2, `prop`: 1});
|
||||
CREATE (:__mg_vertex__:`label2` {__mg_id__: 3, `prop2`: 2, `prop`: 2});
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 1 AND v.__mg_id__ = 0 CREATE (u)-[:`link` {`ext`: [false, {`k`: "l"}], `prop`: -1}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 1 AND v.__mg_id__ = 1 CREATE (u)-[:`link` {`ext`: [false, {`k`: "l"}], `prop`: -1}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 1 AND v.__mg_id__ = 2 CREATE (u)-[:`link` {`ext`: [false, {`k`: "l"}], `prop`: -1}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 1 AND v.__mg_id__ = 3 CREATE (u)-[:`link` {`ext`: [false, {`k`: "l"}], `prop`: -1}]->(v);
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT EXISTS (u.`ext`);
|
||||
CREATE CONSTRAINT ON (u:`label2`) ASSERT u.`prop2`, u.`prop` IS UNIQUE;
|
||||
DROP INDEX ON :__mg_vertex__(__mg_id__);
|
||||
MATCH (u) REMOVE u:__mg_vertex__, u.__mg_id__;
|
||||
@@ -0,0 +1,16 @@
|
||||
CREATE INDEX ON :`label`;
|
||||
CREATE INDEX ON :`label2`(`prop2`);
|
||||
CREATE INDEX ON :`label2`(`prop`);
|
||||
CREATE INDEX ON :__mg_vertex__(__mg_id__);
|
||||
CREATE (:__mg_vertex__:`label2` {__mg_id__: 0, `prop2`: ["kaj", 2, Null, {`prop4`: -1.341}], `ext`: 2, `prop`: "joj"});
|
||||
CREATE (:__mg_vertex__:`label`:`label2` {__mg_id__: 1, `ext`: 2, `prop`: "joj"});
|
||||
CREATE (:__mg_vertex__:`label2` {__mg_id__: 2, `prop2`: 2, `prop`: 1});
|
||||
CREATE (:__mg_vertex__:`label2` {__mg_id__: 3, `prop2`: 2, `prop`: 2});
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 1 AND v.__mg_id__ = 0 CREATE (u)-[:`link` {`ext`: [false, {`k`: "l"}], `prop`: -1}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 1 AND v.__mg_id__ = 1 CREATE (u)-[:`link` {`ext`: [false, {`k`: "l"}], `prop`: -1}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 1 AND v.__mg_id__ = 2 CREATE (u)-[:`link` {`ext`: [false, {`k`: "l"}], `prop`: -1}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 1 AND v.__mg_id__ = 3 CREATE (u)-[:`link` {`ext`: [false, {`k`: "l"}], `prop`: -1}]->(v);
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT EXISTS (u.`ext`);
|
||||
CREATE CONSTRAINT ON (u:`label2`) ASSERT u.`prop2`, u.`prop` IS UNIQUE;
|
||||
DROP INDEX ON :__mg_vertex__(__mg_id__);
|
||||
MATCH (u) REMOVE u:__mg_vertex__, u.__mg_id__;
|
||||
@@ -0,0 +1,16 @@
|
||||
CREATE INDEX ON :`label`;
|
||||
CREATE INDEX ON :`label2`(`prop2`);
|
||||
CREATE INDEX ON :`label2`(`prop`);
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT EXISTS (u.`ext`);
|
||||
CREATE CONSTRAINT ON (u:`label2`) ASSERT u.`prop2`, u.`prop` IS UNIQUE;
|
||||
CREATE INDEX ON :__mg_vertex__(__mg_id__);
|
||||
CREATE (:__mg_vertex__:`label2` {__mg_id__: 0, `prop2`: ["kaj", 2, Null, {`prop4`: -1.341}], `prop`: "joj", `ext`: 2});
|
||||
CREATE (:__mg_vertex__:`label`:`label2` {__mg_id__: 1, `prop`: "joj", `ext`: 2});
|
||||
CREATE (:__mg_vertex__:`label2` {__mg_id__: 2, `prop2`: 2, `prop`: 1});
|
||||
CREATE (:__mg_vertex__:`label2` {__mg_id__: 3, `prop2`: 2, `prop`: 2});
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 1 AND v.__mg_id__ = 0 CREATE (u)-[:`link` {`prop`: -1, `ext`: [false, {`k`: "l"}]}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 1 AND v.__mg_id__ = 1 CREATE (u)-[:`link` {`prop`: -1, `ext`: [false, {`k`: "l"}]}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 1 AND v.__mg_id__ = 2 CREATE (u)-[:`link` {`prop`: -1, `ext`: [false, {`k`: "l"}]}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 1 AND v.__mg_id__ = 3 CREATE (u)-[:`link` {`prop`: -1, `ext`: [false, {`k`: "l"}]}]->(v);
|
||||
DROP INDEX ON :__mg_vertex__(__mg_id__);
|
||||
MATCH (u) REMOVE u:__mg_vertex__, u.__mg_id__;
|
||||
BIN
tests/integration/durability/tests/v15/test_all/snapshot.bin
Normal file
BIN
tests/integration/durability/tests/v15/test_all/snapshot.bin
Normal file
Binary file not shown.
BIN
tests/integration/durability/tests/v15/test_all/wal.bin
Normal file
BIN
tests/integration/durability/tests/v15/test_all/wal.bin
Normal file
Binary file not shown.
@@ -0,0 +1,6 @@
|
||||
CREATE CONSTRAINT ON (u:`label2`) ASSERT EXISTS (u.`ext2`);
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT EXISTS (u.`ext`);
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT u.`a` IS UNIQUE;
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT u.`b` IS UNIQUE;
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT u.`c` IS UNIQUE;
|
||||
CREATE CONSTRAINT ON (u:`label2`) ASSERT u.`a`, u.`b` IS UNIQUE;
|
||||
@@ -0,0 +1,6 @@
|
||||
CREATE CONSTRAINT ON (u:`label2`) ASSERT EXISTS (u.`ext2`);
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT EXISTS (u.`ext`);
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT u.`c` IS UNIQUE;
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT u.`b` IS UNIQUE;
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT u.`a` IS UNIQUE;
|
||||
CREATE CONSTRAINT ON (u:`label2`) ASSERT u.`b`, u.`a` IS UNIQUE;
|
||||
@@ -0,0 +1,6 @@
|
||||
CREATE CONSTRAINT ON (u:`label2`) ASSERT EXISTS (u.`ext2`);
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT EXISTS (u.`ext`);
|
||||
CREATE CONSTRAINT ON (u:`label2`) ASSERT u.`a`, u.`b` IS UNIQUE;
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT u.`a` IS UNIQUE;
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT u.`b` IS UNIQUE;
|
||||
CREATE CONSTRAINT ON (u:`label`) ASSERT u.`c` IS UNIQUE;
|
||||
Binary file not shown.
BIN
tests/integration/durability/tests/v15/test_constraints/wal.bin
Normal file
BIN
tests/integration/durability/tests/v15/test_constraints/wal.bin
Normal file
Binary file not shown.
@@ -0,0 +1,59 @@
|
||||
// --storage-items-per-batch is set to 7
|
||||
CREATE INDEX ON :__mg_vertex__(__mg_id__);
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 0});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 1});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 2});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 3});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 4});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 5});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 6});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 7});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 8});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 9});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 10});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 11});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 12});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 13});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 14});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 15});
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 0 AND v.__mg_id__ = 1 CREATE (u)-[:`edge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 2 AND v.__mg_id__ = 3 CREATE (u)-[:`edge` {`prop`: 11}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 4 AND v.__mg_id__ = 5 CREATE (u)-[:`edge` {`prop`: true}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 6 AND v.__mg_id__ = 7 CREATE (u)-[:`edge2`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 8 AND v.__mg_id__ = 9 CREATE (u)-[:`edge2` {`prop`: -3.141}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 10 AND v.__mg_id__ = 11 CREATE (u)-[:`edgelink` {`prop`: {`prop`: 1, `prop2`: {`prop4`: 9}}}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 12 AND v.__mg_id__ = 13 CREATE (u)-[:`edgelink` {`prop`: [1, Null, false, "\n\n\n\n\\\"\"\n\t"]}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 0 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 1 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 2 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 3 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 4 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 5 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 6 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 7 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 8 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 9 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 10 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 11 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 12 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 13 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 14 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 15 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 0 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 1 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 2 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 3 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 4 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 5 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 6 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 7 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 8 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 9 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 10 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 11 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 12 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 13 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 14 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 15 CREATE (u)-[:`testedge`]->(v);
|
||||
DROP INDEX ON :__mg_vertex__(__mg_id__);
|
||||
MATCH (u) REMOVE u:__mg_vertex__, u.__mg_id__;
|
||||
@@ -0,0 +1,58 @@
|
||||
CREATE INDEX ON :__mg_vertex__(__mg_id__);
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 0});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 1});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 2});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 3});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 4});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 5});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 6});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 7});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 8});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 9});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 10});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 11});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 12});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 13});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 14});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 15});
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 0 AND v.__mg_id__ = 1 CREATE (u)-[:`edge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 2 AND v.__mg_id__ = 3 CREATE (u)-[:`edge` {`prop`: 11}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 4 AND v.__mg_id__ = 5 CREATE (u)-[:`edge` {`prop`: true}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 6 AND v.__mg_id__ = 7 CREATE (u)-[:`edge2`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 8 AND v.__mg_id__ = 9 CREATE (u)-[:`edge2` {`prop`: -3.141}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 10 AND v.__mg_id__ = 11 CREATE (u)-[:`edgelink` {`prop`: {`prop`: 1, `prop2`: {`prop4`: 9}}}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 12 AND v.__mg_id__ = 13 CREATE (u)-[:`edgelink` {`prop`: [1, Null, false, "\n\n\n\n\\\"\"\n\t"]}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 0 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 1 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 2 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 3 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 4 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 5 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 6 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 7 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 8 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 9 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 10 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 11 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 12 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 13 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 14 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 15 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 0 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 1 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 2 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 3 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 4 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 5 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 6 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 7 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 8 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 9 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 10 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 11 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 12 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 13 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 14 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 15 CREATE (u)-[:`testedge`]->(v);
|
||||
DROP INDEX ON :__mg_vertex__(__mg_id__);
|
||||
MATCH (u) REMOVE u:__mg_vertex__, u.__mg_id__;
|
||||
@@ -0,0 +1,58 @@
|
||||
CREATE INDEX ON :__mg_vertex__(__mg_id__);
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 0});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 1});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 2});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 3});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 4});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 5});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 6});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 7});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 8});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 9});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 10});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 11});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 12});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 13});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 14});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 15});
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 0 AND v.__mg_id__ = 1 CREATE (u)-[:`edge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 2 AND v.__mg_id__ = 3 CREATE (u)-[:`edge` {`prop`: 11}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 4 AND v.__mg_id__ = 5 CREATE (u)-[:`edge` {`prop`: true}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 6 AND v.__mg_id__ = 7 CREATE (u)-[:`edge2`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 8 AND v.__mg_id__ = 9 CREATE (u)-[:`edge2` {`prop`: -3.141}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 10 AND v.__mg_id__ = 11 CREATE (u)-[:`edgelink` {`prop`: {`prop`: 1, `prop2`: {`prop4`: 9}}}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 12 AND v.__mg_id__ = 13 CREATE (u)-[:`edgelink` {`prop`: [1, Null, false, "\n\n\n\n\\\"\"\n\t"]}]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 0 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 1 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 2 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 3 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 4 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 5 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 6 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 7 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 8 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 9 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 10 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 11 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 12 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 13 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 14 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 14 AND v.__mg_id__ = 15 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 0 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 1 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 2 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 3 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 4 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 5 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 6 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 7 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 8 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 9 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 10 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 11 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 12 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 13 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 14 CREATE (u)-[:`testedge`]->(v);
|
||||
MATCH (u:__mg_vertex__), (v:__mg_vertex__) WHERE u.__mg_id__ = 15 AND v.__mg_id__ = 15 CREATE (u)-[:`testedge`]->(v);
|
||||
DROP INDEX ON :__mg_vertex__(__mg_id__);
|
||||
MATCH (u) REMOVE u:__mg_vertex__, u.__mg_id__;
|
||||
BIN
tests/integration/durability/tests/v15/test_edges/snapshot.bin
Normal file
BIN
tests/integration/durability/tests/v15/test_edges/snapshot.bin
Normal file
Binary file not shown.
BIN
tests/integration/durability/tests/v15/test_edges/wal.bin
Normal file
BIN
tests/integration/durability/tests/v15/test_edges/wal.bin
Normal file
Binary file not shown.
@@ -0,0 +1,4 @@
|
||||
CREATE INDEX ON :`label2`;
|
||||
CREATE INDEX ON :`label2`(`prop2`);
|
||||
CREATE INDEX ON :`label`(`prop2`);
|
||||
CREATE INDEX ON :`label`(`prop`);
|
||||
@@ -0,0 +1,4 @@
|
||||
CREATE INDEX ON :`label2`;
|
||||
CREATE INDEX ON :`label`(`prop`);
|
||||
CREATE INDEX ON :`label`(`prop2`);
|
||||
CREATE INDEX ON :`label2`(`prop2`);
|
||||
@@ -0,0 +1,4 @@
|
||||
CREATE INDEX ON :`label2`;
|
||||
CREATE INDEX ON :`label2`(`prop2`);
|
||||
CREATE INDEX ON :`label`(`prop2`);
|
||||
CREATE INDEX ON :`label`(`prop`);
|
||||
BIN
tests/integration/durability/tests/v15/test_indices/snapshot.bin
Normal file
BIN
tests/integration/durability/tests/v15/test_indices/snapshot.bin
Normal file
Binary file not shown.
BIN
tests/integration/durability/tests/v15/test_indices/wal.bin
Normal file
BIN
tests/integration/durability/tests/v15/test_indices/wal.bin
Normal file
Binary file not shown.
@@ -0,0 +1,17 @@
|
||||
// --storage-items-per-batch is set to 5
|
||||
CREATE INDEX ON :__mg_vertex__(__mg_id__);
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 0});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 1});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 2, `prop`: false});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 3, `prop`: true});
|
||||
CREATE (:__mg_vertex__:`label2` {__mg_id__: 4, `prop`: 1});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 5, `prop2`: 3.141});
|
||||
CREATE (:__mg_vertex__:`label6` {__mg_id__: 6, `prop3`: true, `prop2`: -314000000});
|
||||
CREATE (:__mg_vertex__:`label3`:`label1`:`label2` {__mg_id__: 7});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 8, `prop3`: "str", `prop2`: 2, `prop`: 1});
|
||||
CREATE (:__mg_vertex__:`label2`:`label1` {__mg_id__: 9, `prop`: {`prop_nes`: "kaj je"}});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 10, `prop_array`: [1, false, Null, "str", {`prop2`: 2}]});
|
||||
CREATE (:__mg_vertex__:`label3`:`label` {__mg_id__: 11, `prop`: {`prop`: [1, false], `prop2`: {}, `prop3`: "test2", `prop4`: "test"}});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 12, `prop`: " \n\"\'\t\\%"});
|
||||
DROP INDEX ON :__mg_vertex__(__mg_id__);
|
||||
MATCH (u) REMOVE u:__mg_vertex__, u.__mg_id__;
|
||||
@@ -0,0 +1,16 @@
|
||||
CREATE INDEX ON :__mg_vertex__(__mg_id__);
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 0});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 1});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 2, `prop`: false});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 3, `prop`: true});
|
||||
CREATE (:__mg_vertex__:`label2` {__mg_id__: 4, `prop`: 1});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 5, `prop2`: 3.141});
|
||||
CREATE (:__mg_vertex__:`label6` {__mg_id__: 6, `prop3`: true, `prop2`: -314000000});
|
||||
CREATE (:__mg_vertex__:`label2`:`label3`:`label1` {__mg_id__: 7});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 8, `prop3`: "str", `prop2`: 2, `prop`: 1});
|
||||
CREATE (:__mg_vertex__:`label1`:`label2` {__mg_id__: 9, `prop`: {`prop_nes`: "kaj je"}});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 10, `prop_array`: [1, false, Null, "str", {`prop2`: 2}]});
|
||||
CREATE (:__mg_vertex__:`label`:`label3` {__mg_id__: 11, `prop`: {`prop`: [1, false], `prop2`: {}, `prop3`: "test2", `prop4`: "test"}});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 12, `prop`: " \n\"\'\t\\%"});
|
||||
DROP INDEX ON :__mg_vertex__(__mg_id__);
|
||||
MATCH (u) REMOVE u:__mg_vertex__, u.__mg_id__;
|
||||
@@ -0,0 +1,16 @@
|
||||
CREATE INDEX ON :__mg_vertex__(__mg_id__);
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 0});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 1});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 2, `prop`: false});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 3, `prop`: true});
|
||||
CREATE (:__mg_vertex__:`label2` {__mg_id__: 4, `prop`: 1});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 5, `prop2`: 3.141});
|
||||
CREATE (:__mg_vertex__:`label6` {__mg_id__: 6, `prop2`: -314000000, `prop3`: true});
|
||||
CREATE (:__mg_vertex__:`label2`:`label3`:`label1` {__mg_id__: 7});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 8, `prop`: 1, `prop2`: 2, `prop3`: "str"});
|
||||
CREATE (:__mg_vertex__:`label1`:`label2` {__mg_id__: 9, `prop`: {`prop_nes`: "kaj je"}});
|
||||
CREATE (:__mg_vertex__:`label` {__mg_id__: 10, `prop_array`: [1, false, Null, "str", {`prop2`: 2}]});
|
||||
CREATE (:__mg_vertex__:`label`:`label3` {__mg_id__: 11, `prop`: {`prop`: [1, false], `prop2`: {}, `prop3`: "test2", `prop4`: "test"}});
|
||||
CREATE (:__mg_vertex__ {__mg_id__: 12, `prop`: " \n\"\'\t\\%"});
|
||||
DROP INDEX ON :__mg_vertex__(__mg_id__);
|
||||
MATCH (u) REMOVE u:__mg_vertex__, u.__mg_id__;
|
||||
Binary file not shown.
BIN
tests/integration/durability/tests/v15/test_vertices/wal.bin
Normal file
BIN
tests/integration/durability/tests/v15/test_vertices/wal.bin
Normal file
Binary file not shown.
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
|
||||
@@ -3,6 +3,23 @@
|
||||
set -Eeuo pipefail
|
||||
script_dir="$( cd "$( dirname "${BASH_SOURCE[0]}" )" && pwd )"
|
||||
|
||||
MEMGRAPH_BINARY_PATH="../../build/memgraph"
|
||||
# NOTE: On Ubuntu 22.04 0.3.2 uses non-existing docker compose --compatibility flag.
|
||||
# NOTE: On Ubuntu 22.04 0.3.1 seems to be working.
|
||||
JEPSEN_VERSION="${JEPSEN_VERSION:-v0.3.0}"
|
||||
JEPSEN_ACTIVE_NODES_NO=5
|
||||
CONTROL_LEIN_RUN_ARGS="test-all --node-configs resources/node-config.edn"
|
||||
CONTROL_LEIN_RUN_STDOUT_LOGS=1
|
||||
CONTROL_LEIN_RUN_STDERR_LOGS=1
|
||||
PRINT_CONTEXT() {
|
||||
echo -e "MEMGRAPH_BINARY_PATH:\t\t $MEMGRAPH_BINARY_PATH"
|
||||
echo -e "JEPSEN_VERSION:\t\t\t $JEPSEN_VERSION"
|
||||
echo -e "JEPSEN_ACTIVE_NODES_NO:\t\t $JEPSEN_ACTIVE_NODES_NO"
|
||||
echo -e "CONTROL_LEIN_RUN_ARGS:\t\t $CONTROL_LEIN_RUN_ARGS"
|
||||
echo -e "CONTROL_LEIN_RUN_STDOUT_LOGS:\t $CONTROL_LEIN_RUN_STDOUT_LOGS"
|
||||
echo -e "CONTROL_LEIN_RUN_STDERR_LOGS:\t $CONTROL_LEIN_RUN_STDERR_LOGS"
|
||||
}
|
||||
|
||||
HELP_EXIT() {
|
||||
echo ""
|
||||
echo "HELP: $0 help|cluster-up|test [args]"
|
||||
@@ -28,15 +45,10 @@ if ! command -v docker > /dev/null 2>&1 || ! command -v docker-compose > /dev/nu
|
||||
ERROR "docker and docker-compose have to be installed."
|
||||
exit 1
|
||||
fi
|
||||
|
||||
MEMGRAPH_BINARY_PATH="../../build/memgraph"
|
||||
JEPSEN_ACTIVE_NODES_NO=5
|
||||
CONTROL_LEIN_RUN_ARGS="test-all --node-configs resources/node-config.edn"
|
||||
CONTROL_LEIN_RUN_STDOUT_LOGS=1
|
||||
CONTROL_LEIN_RUN_STDERR_LOGS=1
|
||||
PRINT_CONTEXT
|
||||
|
||||
if [ ! -d "$script_dir/jepsen" ]; then
|
||||
git clone https://github.com/jepsen-io/jepsen.git -b "0.2.1" "$script_dir/jepsen"
|
||||
git clone https://github.com/jepsen-io/jepsen.git -b "$JEPSEN_VERSION" "$script_dir/jepsen"
|
||||
fi
|
||||
|
||||
if [ "$#" -lt 1 ]; then
|
||||
@@ -96,21 +108,25 @@ case $1 in
|
||||
esac
|
||||
done
|
||||
|
||||
# Resolve binary path if it is a link.
|
||||
# Copy Memgraph binary, handles both cases, when binary is a sym link
|
||||
# or a regular file.
|
||||
binary_path="$MEMGRAPH_BINARY_PATH"
|
||||
if [ -L "$binary_path" ]; then
|
||||
binary_path=$(readlink "$binary_path")
|
||||
fi
|
||||
binary_name=$(basename -- "$binary_path")
|
||||
|
||||
# Copy Memgraph binary.
|
||||
for iter in $(seq 1 "$JEPSEN_ACTIVE_NODES_NO"); do
|
||||
jepsen_node_name="jepsen-n$iter"
|
||||
# Cleanup the node folder with previous binaries.
|
||||
docker exec "$jepsen_node_name" rm -rf /opt/memgraph/
|
||||
docker exec "$jepsen_node_name" mkdir -p /opt/memgraph
|
||||
docker cp "$binary_path" "$jepsen_node_name":/opt/memgraph/"$binary_name"
|
||||
docker exec "$jepsen_node_name" bash -c "rm -f /opt/memgraph/memgraph && ln -s /opt/memgraph/$binary_name /opt/memgraph/memgraph"
|
||||
docker_exec="docker exec $jepsen_node_name bash -c"
|
||||
if [ "$binary_name" == "memgraph" ]; then
|
||||
_binary_name="memgraph_tmp"
|
||||
else
|
||||
_binary_name="$binary_name"
|
||||
fi
|
||||
$docker_exec "rm -rf /opt/memgraph/ && mkdir -p /opt/memgraph"
|
||||
docker cp "$binary_path" "$jepsen_node_name":/opt/memgraph/"$_binary_name"
|
||||
$docker_exec "ln -s /opt/memgraph/$_binary_name /opt/memgraph/memgraph"
|
||||
$docker_exec "touch /opt/memgraph/memgraph.log"
|
||||
INFO "Copying $binary_name to $jepsen_node_name DONE."
|
||||
done
|
||||
|
||||
|
||||
5
tests/mgbench/.gitignore
vendored
5
tests/mgbench/.gitignore
vendored
@@ -1 +1,6 @@
|
||||
.cache
|
||||
*.zip
|
||||
*.log
|
||||
*.report
|
||||
*.sysinfo
|
||||
*.json
|
||||
|
||||
42
tests/mgbench/Dockerfile.mgbench_client
Normal file
42
tests/mgbench/Dockerfile.mgbench_client
Normal file
@@ -0,0 +1,42 @@
|
||||
FROM ubuntu:22.04 AS mg_bench_client_build_base
|
||||
|
||||
SHELL ["/bin/bash", "-c"]
|
||||
ARG TOOLCHAIN_VERSION
|
||||
ARG TARGETARCH
|
||||
|
||||
ENV DEBIAN_FRONTEND=noninteractive
|
||||
|
||||
USER root
|
||||
|
||||
RUN apt update && apt install -y \
|
||||
ca-certificates wget git
|
||||
|
||||
RUN wget -q https://s3-eu-west-1.amazonaws.com/deps.memgraph.io/${TOOLCHAIN_VERSION}/${TOOLCHAIN_VERSION}-binaries-ubuntu-22.04-${TARGETARCH}.tar.gz \
|
||||
-O ${TOOLCHAIN_VERSION}-binaries-ubuntu-22.04-${TARGETARCH}.tar.gz \
|
||||
&& tar xzvf ${TOOLCHAIN_VERSION}-binaries-ubuntu-22.04-${TARGETARCH}.tar.gz -C /opt
|
||||
|
||||
RUN git clone https://github.com/memgraph/memgraph.git
|
||||
|
||||
WORKDIR memgraph
|
||||
|
||||
RUN if [ ${TARGETARCH} = "amd64" ] ; then ./environment/os/ubuntu-22.04.sh install TOOLCHAIN_RUN_DEPS ; else ./environment/os/ubuntu-22.04-arm.sh install TOOLCHAIN_RUN_DEPS ; fi
|
||||
RUN if [ ${TARGETARCH} = "amd64" ] ; then ./environment/os/ubuntu-22.04.sh install MEMGRAPH_BUILD_DEPS ; else ./environment/os/ubuntu-22.04-arm.sh install MEMGRAPH_BUILD_DEPS ; fi
|
||||
|
||||
RUN source /opt/toolchain-v4/activate && \
|
||||
./init && \
|
||||
rm -r build && \
|
||||
mkdir build && \
|
||||
cd build && \
|
||||
cmake -DCMAKE_BUILD_TYPE=release .. && \
|
||||
make -j$(nproc) memgraph__mgbench__client && \
|
||||
make .
|
||||
|
||||
|
||||
FROM ubuntu:22.04
|
||||
|
||||
RUN apt-get update && apt-get install -y wget libcurl4
|
||||
|
||||
# Copy mgbench client to clean image
|
||||
COPY --from=mg_bench_client_build_base /memgraph/build/tests/mgbench/client /bin/
|
||||
|
||||
ENTRYPOINT ["bin/client"]
|
||||
@@ -1,19 +1,14 @@
|
||||
# :fire: mgBench: Benchmark for graph databases
|
||||
# :fire: Benchgraph: Benchmark for graph databases
|
||||
|
||||
## :clipboard: Benchmark Overview
|
||||
|
||||
mgBench is primarily designed to benchmark graph databases. To test graph database performance, this benchmark executes Cypher queries (write, read, update, aggregate, and analyze) on a given dataset. Queries are general and represent a typical workload that would be used to analyze any graph dataset. [BenchGraph](https://memgraph.com/benchgraph/) platform shows the results of running these queries on supported vendors. It shows the overall performance of each system relative to others.
|
||||
Benchgraph is primarily designed to benchmark graph databases (Currently, Neo4j and Memgraph). To test graph database performance, this benchmark executes Cypher queries that can write, read, update, aggregate, and analyze dataset present in database. There are some predefined queries and dataset in Benchgraph. The present datasets and queries represent a typical workload that would be used to analyze any graph dataset and are pure Cypher based. [BenchGraph](https://memgraph.com/benchgraph/) shows the results of running these queries on specified hardware and under certain conditions. It shows the overall performance of each system under test relative to other, best in test being the baseline.
|
||||
|
||||
Three workload types can be executed:
|
||||
- Isolated - Concurrent execution of a single type of query,
|
||||
- Mixed - Concurrent execution of a single type of query mixed with a certain percentage of queries from a designated query group,
|
||||
- Realistic - Concurrent execution of queries from write, read, update and analyze groups.
|
||||
|
||||
Currently, the benchmark is executed on the social media dataset Pokec, available in different sizes. The full list of queries and their grouping is available as [query list](#query-list).
|
||||
There is also a [tutorial on how to use benchgraph](how_to_use_benchgraph.md) to define your own dataset and queries, and run them on supported vendors. If you are interested in running benchmarks on your data, read tutorial, otherwise if you wish to run and validate results from Benchgraph read on.
|
||||
|
||||
This methodology is designed to be read from top to bottom to understand what is being tested and how, but feel free to jump to parts that interest you.
|
||||
|
||||
- [:fire: mgBench: Benchmark for graph databases](#fire-mgbench-benchmark-for-graph-databases)
|
||||
- [:fire: Benchgraph: Benchmark for graph databases](#fire-benchgraph-benchmark-for-graph-databases)
|
||||
- [:clipboard: Benchmark Overview](#clipboard-benchmark-overview)
|
||||
- [:dart: Design goals](#dart-design-goals)
|
||||
- [Reproducibility and validation](#reproducibility-and-validation)
|
||||
@@ -21,7 +16,7 @@ This methodology is designed to be read from top to bottom to understand what is
|
||||
- [Workloads](#workloads)
|
||||
- [Fine-tuning](#fine-tuning)
|
||||
- [Limitations](#limitations)
|
||||
- [:wrench: mgBench](#wrench-mgbench)
|
||||
- [:wrench: Benchgraph](#wrench-Benchgraph)
|
||||
- [Important files](#important-files)
|
||||
- [Prerequisites](#prerequisites)
|
||||
- [Running the benchmark](#running-the-benchmark)
|
||||
@@ -36,33 +31,34 @@ This methodology is designed to be read from top to bottom to understand what is
|
||||
- [:nut\_and\_bolt: Supported databases](#nut_and_bolt-supported-databases)
|
||||
- [Database notes](#database-notes)
|
||||
- [:raised\_hands: Contributions](#raised_hands-contributions)
|
||||
- [:mega: History and Future of mgBench](#mega-history-and-future-of-mgbench)
|
||||
- [History of mgBench](#history-of-mgbench)
|
||||
- [Future of mgBench](#future-of-mgbench)
|
||||
- [:mega: History and Future of Benchgraph](#mega-history-and-future-of-Benchgraph)
|
||||
- [History of Benchgraph](#history-of-Benchgraph)
|
||||
- [Future of Benchgraph](#future-of-Benchgraph)
|
||||
- [Changelog benchgraph](#changelog-benchgraph)
|
||||
|
||||
## :dart: Design goals
|
||||
|
||||
### Reproducibility and validation
|
||||
|
||||
Running this benchmark is automated, and the code used to run benchmarks is publicly available. You can [run mgBench](#running-the-benchmark) with default settings to validate the results at [BenchGraph platform](https://memgraph.com/benchgraph). The results may differ depending on the hardware, database configuration, and other variables involved in your setup. But if the results you get are significantly different, feel free to [open a GitHub issue](https://github.com/memgraph/memgraph/issues).
|
||||
Running this benchmark is automated, and the code used to run benchmarks is publicly available. You can [run benchgraph](#running-the-benchmark) with settings specified under to validate the results at [BenchGraph](https://memgraph.com/benchgraph). The results may differ depending on the hardware, benchmark run configuration, database configuration, and other variables involved in your setup. But if the results you get are significantly different, feel free to [open a GitHub issue](https://github.com/memgraph/memgraph/issues).
|
||||
|
||||
In the future, the project will be expanded to include more platforms to see how systems perform on different OS and hardware configurations. If you are interested in what will be added and tested, read the section about [the future of mgBench](#future-of-mgbench)
|
||||
In the future, the project will be expanded to include more platforms to see how systems perform on different OS and hardware configurations. If you are interested in what will be added and tested, read the section about [the future of Benchgraph](#future-of-Benchgraph)
|
||||
|
||||
|
||||
### Database compatibility
|
||||
|
||||
At the moment, support for graph databases is limited. To run the benchmarks, the graph database must support Cypher query language and the Bolt protocol.
|
||||
|
||||
Using Cypher ensures that executed queries are identical or similar on every supported system. Possible differences are noted in [database notes](#database-notes). A single C++ client queries all database systems, and it is based on the Bolt protocol. Using a single client ensures minimal performance penalties from the client side and ensures fairness across different vendors.
|
||||
Using Cypher ensures that executed queries are identical or similar as possible on every supported system. A single C++ client queries database systems (Currently, Neo4j and Memgraph), and it is based on the Bolt protocol. Using a single client ensures minimal performance penalties from the client side and ensures fairness across different vendors.
|
||||
|
||||
If your database supports the given requirements, feel free to contribute and add your database to mgBench.
|
||||
If your database supports the given requirements, feel free to contribute and add your database to Benchgraph.
|
||||
If your database does not support the mentioned requirements, follow the project because support for other languages and protocols in graph database space will be added.
|
||||
|
||||
|
||||
### Workloads
|
||||
Running queries as standalone units is simple and relatively easy to measure, but vendors often apply various caching and pre-aggregations that influence the results in these kinds of scenarios. Results from running single queries can hint at the database's general performance, but in real life, a database is queried by multiple clients from multiple sides. That is why the mgBench client supports the consecutive execution of various queries. Concurrently writing, reading, updating and executing aggregational and analytical queries provides a better view of overall system performance than executing and measuring a single query. Queries that the mgBench executes are grouped into 5 groups - write, read, update, aggregate and analytical.
|
||||
Running queries as standalone units is simple and relatively easy to measure, but vendors often apply various caching and pre-aggregations that influence the results in these kinds of scenarios. Results from running single queries can hint at the database's general performance, but in real life, a database is queried by multiple clients from multiple sides. That is why the Benchgraph client supports the consecutive execution of various queries. Concurrently writing, reading, updating and executing aggregational and analytical queries provides a better view of overall system performance than executing and measuring a single query. Queries that the Benchgraph executes are grouped into 5 groups - write, read, update, aggregate and analytical.
|
||||
|
||||
The [BenchGraph platform](https://memgraph.com/benchgraph) shows results made by mgBench by executing three types of workloads:
|
||||
The [BenchGraph platform](https://memgraph.com/benchgraph) shows results made by Benchgraph by executing three types of workloads:
|
||||
- ***Isolated workload***
|
||||
- ***Mixed workload***
|
||||
- ***Realistic workload***
|
||||
@@ -74,7 +70,7 @@ If a query takes arguments, the argument value is changed for each execution. Ar
|
||||
The good thing about isolated workload is that it yields a better picture of single query performance. There is also a negative side, executing the same queries multiple times can trigger strong results caching on the vendor's side, which can result in false query times.
|
||||
|
||||
|
||||
***Mixed*** workload executes a fixed number of queries that read, update, aggregate, or analyze the data concurrently with a certain percentage of write queries because writing from the database can prevent aggressive caching and thus represent a more realistic performance of a single query. The negative side is that there is an added influence of write performance on the results. Currently, mgBench client does not support per-thread performance measurements, but this will be added in future iterations.
|
||||
***Mixed*** workload executes a fixed number of queries that read, update, aggregate, or analyze the data concurrently with a certain percentage of write queries because writing from the database can prevent aggressive caching and thus represent a more realistic performance of a single query. The negative side is that there is an added influence of write performance on the results. Currently, Benchgraph client does not support per-thread performance measurements, but this will be added in future iterations.
|
||||
|
||||
|
||||
***Realistic*** workload represents real-life use cases because queries write, read, update, and perform analytics in a mixed ratio like they would in real projects. The test executes a fixed number of queries, the distribution of which is defined by defining a percentage of queries performing one of four operations. The queries are selected non-randomly, so the workload is identical between different vendors. As with the rest of the workloads, all queries are executed concurrently.
|
||||
@@ -88,45 +84,43 @@ Some configurational changes are necessary for test execution and are not consid
|
||||
### Limitations
|
||||
|
||||
Benchmarking different systems is challenging because the setup, environment, queries, workload, and dataset can benefit specific database vendors. Each vendor may have a particularly strong use-case scenario. This benchmark aims to be neutral and fair to all database vendors. Acknowledging some of the current limitations can help understand the issues you might notice:
|
||||
1. mgBench measures and tracks just a tiny subset of everything that can be tracked and compared during testing. Active benchmarking is strenuous because it requires a lot of time to set up and validate. Passive benchmarking is much faster to iterate on but can have a few bugs.
|
||||
2. Datasets and queries used for testing are simple. Datasets and queries in real-world environments can become quite complex. To avoid Cypher specifics, mgBench uses simple queries of different variates. Future versions will include more complex datasets and queries.
|
||||
3. The scale of the dataset used is miniature for production environments. Production environments can have up to trillions of nodes and edges.
|
||||
Query results are not verified or important. The queries might return different results, but only the performance is measured, not correctness.
|
||||
4. All tests are performed on single-node databases.
|
||||
5. Architecturally different systems can be set up and measured biasedly.
|
||||
1. Benchgraph measures and tracks just a tiny subset of everything that can be tracked and compared during testing. Active benchmarking is strenuous because it requires a lot of time to set up and validate. Passive benchmarking is much faster to iterate on but can have a few bugs.
|
||||
2. The scale of the dataset used is miniature for production environments. Production environments can have up to trillions of nodes and edges.
|
||||
3. All tests are performed on single-node databases.
|
||||
4. Architecturally different systems can be set up and measured biasedly.
|
||||
|
||||
|
||||
## :wrench: mgBench
|
||||
## :wrench: Benchgraph
|
||||
### Important files
|
||||
|
||||
Listed below are the main scripts used to run the benchmarks:
|
||||
|
||||
- `benchmark.py` - Script that runs the queries and workloads.
|
||||
- `datasets.py` - Script that handles datasets and queries for workloads.
|
||||
- `runners.py` - Script holding the configuration for different DB vendors.
|
||||
- `benchmark.py` - The main entry point used for starting and managing the execution of the benchmark. This script initializes all the necessary files, classes, and objects. It starts the database and the benchmark and gathers the results.
|
||||
- `base.py` - This is the base workload class. All other workloads are subclasses located in the workloads directory. For example, ldbc_interactive.py defines ldbc interactive dataset and queries (but this is NOT an official LDBC interactive workload). Each workload class can generate the dataset, use custom import ofthe dataset or provide a CYPHERL file for the import process..
|
||||
- `runners.py` - The script that configures, starts, and stops the database.
|
||||
- `client.cpp` - Client for querying the database.
|
||||
- `graph_bench.py` - Script that starts all predefined and custom-defined workloads.
|
||||
-` compare_results.py` - Script that visually compares benchmark results.
|
||||
- `graph_bench.py` - Script that starts all tests from Benchgraph.
|
||||
- `compare_results.py` - Script that visually compares benchmark results.
|
||||
|
||||
Except for these scripts, the project also includes dataset files and index configuration files. Once the first test is executed, those files can be located in the newly generated .cache folder.
|
||||
Except for these scripts, the project also includes query files, dataset files and index configuration files. Once the first test is executed, those files can be located in the newly generated `.cache` and `.temp` folders.
|
||||
|
||||
### Prerequisites
|
||||
|
||||
To execute a benchmark, you need to download a binary version of supported databases and install Python on your system. Each database vendor can depend on external dependencies, such as Cmake, JVM, etc., so make sure to check specific vendor prerequisites.
|
||||
To execute a Benchgraph benchmark and validate results, you need to compile Memgraph and benchmark C++ bolt client from source, for more details on compilation process, take a look into this [guide](https://www.notion.so/memgraph/Quick-Start-82a99a85e62a4e3d89f6a9fb6d35626d?pvs=4). For Neo4j just download a binary version of Neo4j database you want to benchmark. Python version 3.7 and above is requirement for running benchmarks. Each database vendor can depend on external dependencies, such as Cmake, JVM, etc., so make sure to check specific vendor prerequisites during compilation process or running requirements.
|
||||
|
||||
### Running the benchmark
|
||||
To run benchmarks, use the `graph_bench.py` script, which calls all the other necessary scripts. You can start the benchmarks by executing the following command:
|
||||
To run benchmarks, you can use the `graph_bench.py`, which calls all the other necessary scripts. You can start the benchmarks by executing the following command:
|
||||
|
||||
```
|
||||
```python
|
||||
graph_bench.py
|
||||
--vendor memgraph /home/memgraph/binary
|
||||
--dataset-group basic
|
||||
--dataset-size small
|
||||
--realistic 100 30 70 0 0
|
||||
--realistic 100 50 50 0 0
|
||||
--realistic 100 70 30 0 0
|
||||
--realistic 100 30 40 10 20
|
||||
--mixed 100 30 0 0 0 70
|
||||
--realistic 500 30 70 0 0
|
||||
--realistic 500 50 50 0 0
|
||||
--realistic 500 70 30 0 0
|
||||
--realistic 500 30 40 10 20
|
||||
--mixed 500 30 0 0 0 70
|
||||
```
|
||||
|
||||
|
||||
@@ -134,9 +128,9 @@ Isolated workload are always executed, and this commands calls for the execution
|
||||
|
||||
The distribution of queries from write, read, update and aggregate groups are defined in percentages and stated as arguments following the `--realistic` or `--mixed` flags.
|
||||
|
||||
In the example of `--realistic 100 30 40 10 20` the distribution is as follows:
|
||||
In the example of `--realistic 500 30 40 10 20` the distribution is as follows:
|
||||
|
||||
- 100 - The number of queries to be executed.
|
||||
- 500 - The number of queries to be executed.
|
||||
- 30 - The percentage of write queries to be executed.
|
||||
- 40 - The percentage of read queries to be executed.
|
||||
- 10 - The percentage of update queries to be executed.
|
||||
@@ -147,23 +141,21 @@ For `--mixed` workload argument, the first five parameters are the same, with an
|
||||
|
||||
Feel free to add different configurations if you want. Results from the above benchmark run are visible on [BenchGraph platform](https://memgraph.com/benchgraph)
|
||||
|
||||
The other option is to use `benchgraph.sh` that can execute all benchmarks for each of specified number of workers.
|
||||
|
||||
### Database conditions
|
||||
In a production environment, database query caches are usually warmed from usage or pre-warm procedure to provide the best possible performance. Each workload in mgBench will be executed under the following conditions:
|
||||
In a production environment, database query caches are usually warmed from usage or pre-warm procedure to provide the best possible performance. Each workload in Benchgraph will be executed under the following conditions:
|
||||
- ***Hot run*** - before executing any benchmark query and taking measurements, a set of defined queries is executed to pre-warm the database.
|
||||
- ***Cold run*** - no warm-up was performed on the database before taking benchmark measurements.
|
||||
- ***Vulcanic run*** - The workload is executed twice. The first time is used to pre-warm the database, and the second time is used to take measurements. The workload does not change between the two runs.
|
||||
|
||||
List of queries used for pre-warm up:
|
||||
```
|
||||
CREATE ();
|
||||
CREATE ()-[:TempEdge]->();
|
||||
MATCH (n) RETURN n LIMIT 1;
|
||||
```
|
||||
The details specification of warmup procedure is visible in the `benchmark.py` file, `warmup` function.
|
||||
|
||||
### Comparing results
|
||||
|
||||
Once the benchmark has been run for a single vendor, all the results are saved in appropriately named `.json` files. A summary file is also created for that vendor and it contains all results combined. These summary files are used to compare results against other vendor results via the `compare_results.py` script:
|
||||
|
||||
```
|
||||
```python
|
||||
compare_results.py
|
||||
--compare
|
||||
“path_to/neo4j_summary.json”
|
||||
@@ -177,10 +169,11 @@ The output is an HTML file with the visual representation of the performance dif
|
||||
## :bar_chart: Results
|
||||
Results visible in the HTML file or at [BenchGraph](https://memgraph.com/benchgraph) are throughput, memory, and latency. Database throughput and memory usage directly impact database usability and cost, while the latency of the query shows the base query execution duration.
|
||||
|
||||
***Throughput*** directly defines how performant the database is and how much query traffic it can handle in a fixed time interval. It is expressed in queries per second. In each concurrent workload, execution is split across multiple clients. Each client executes queries concurrently. The duration of total execution is the sum of all concurrent clients' execution duration in seconds. In mgBench, the total count of executed queries and the total duration defines throughput per second across concurrent execution.
|
||||
***Throughput*** directly defines how performant the database is and how much query traffic it can handle in a fixed time interval. It is expressed in queries per second. In each concurrent workload, execution is split across multiple clients. Each client executes queries concurrently. The duration of total execution is the sum of all concurrent clients' execution duration in seconds. In Benchgraph, the total count of executed queries and the total duration defines throughput per second across concurrent execution.
|
||||
|
||||
Here is the code snippet from the client, that calculates ***throughput*** and metadata:
|
||||
```
|
||||
|
||||
```cpp
|
||||
// Create and output summary.
|
||||
Metadata final_metadata;
|
||||
uint64_t final_retries = 0;
|
||||
@@ -190,82 +183,137 @@ Here is the code snippet from the client, that calculates ***throughput*** and m
|
||||
final_retries += worker_retries[i];
|
||||
final_duration += worker_duration[i];
|
||||
}
|
||||
|
||||
auto total_time_end = std::chrono::steady_clock::now();
|
||||
auto total_time = std::chrono::duration_cast<std::chrono::duration<double>>(total_time_end - total_time_start);
|
||||
|
||||
final_duration /= FLAGS_num_workers;
|
||||
nlohmann::json summary = nlohmann::json::object();
|
||||
summary["total_time"] = total_time.count();
|
||||
summary["count"] = queries.size();
|
||||
summary["duration"] = final_duration;
|
||||
summary["throughput"] = static_cast<double>(queries.size()) / final_duration;
|
||||
summary["retries"] = final_retries;
|
||||
summary["metadata"] = final_metadata.Export();
|
||||
summary["num_workers"] = FLAGS_num_workers;
|
||||
summary["latency_stats"] = LatencyStatistics(worker_query_durations);
|
||||
(*stream) << summary.dump() << std::endl;
|
||||
|
||||
```
|
||||
|
||||
***Memory*** usage is calculated as ***peak RES*** (resident size) memory for each query or workload execution within mgBench. The result includes starting the database, executing the query/workload, and stopping the database. The peak RES is extracted from process PID as VmHVM (peak resident set size) before the process is stopped. The peak memory usage defines the worst-case scenario for a given query or workload, while on average, RAM footprint is lower. Measuring RES over time is supported by `runners.py`. For each vendor, it is possible to add RES tracking across workload execution, but it is not reported in the results.
|
||||
***Latency*** is calculated as the serial execution of 100 identical queries on a single thread. Each query has standard query statistics and tail latency data. The result includes query execution times: max, min, mean, p99, p95, p90, p75, and p50 in seconds.
|
||||
***Memory*** usage is calculated as ***peak RES*** (resident size) memory for each query or workload execution within Benchgraph. The result includes starting the database, executing the query/workload, and stopping the database. The peak RES is extracted from process PID as VmHVM (peak resident set size) before the process is stopped. The peak memory usage defines the worst-case scenario for a given query or workload, while on average, RAM footprint is lower. Measuring RES over time is supported by `runners.py`. For each vendor, it is possible to add RES tracking across workload execution, but it is not reported in the results.
|
||||
|
||||
Each workload and all the results are based on concurrent query execution, except ***latency***. As stated in [limitations](#limitations) section, mgBench tracks just a subset of resources, but the chapter on [mgBench future](#future-of-mgbench) explains the expansion plans.
|
||||
***Latency*** is calculated during the execution of each workload. Each query has standard query statistics and tail latency data. The result includes query execution times: max, min, mean, p99, p95, p90, p75, and p50 in seconds.
|
||||
Here is the code snippet that calculates latency:
|
||||
|
||||
```cpp
|
||||
...
|
||||
std::vector<double> query_latency;
|
||||
for (int i = 0; i < FLAGS_num_workers; i++) {
|
||||
for (auto &e : worker_query_latency[i]) {
|
||||
query_latency.push_back(e);
|
||||
}
|
||||
}
|
||||
auto iterations = query_latency.size();
|
||||
const int lower_bound = 10;
|
||||
if (iterations > lower_bound) {
|
||||
std::sort(query_latency.begin(), query_latency.end());
|
||||
statistics["iterations"] = iterations;
|
||||
statistics["min"] = query_latency.front();
|
||||
statistics["max"] = query_latency.back();
|
||||
statistics["mean"] = std::accumulate(query_latency.begin(), query_latency.end(), 0.0) / iterations;
|
||||
statistics["p99"] = query_latency[floor(iterations * 0.99)];
|
||||
statistics["p95"] = query_latency[floor(iterations * 0.95)];
|
||||
statistics["p90"] = query_latency[floor(iterations * 0.90)];
|
||||
statistics["p75"] = query_latency[floor(iterations * 0.75)];
|
||||
statistics["p50"] = query_latency[floor(iterations * 0.50)];
|
||||
}
|
||||
...
|
||||
```
|
||||
|
||||
Each workload and all the results are based on concurrent query execution. As stated in [limitations](#limitations) section, Benchgraph tracks just a subset of resources, but the chapter on [Benchgraph future](#future-of-Benchgraph) explains the expansion plans.
|
||||
|
||||
## :books: Datasets
|
||||
|
||||
Before workload execution, appropriate dataset indexes are set. Each vendor can have a specific syntax for setting up indexes, but those indexes should be schematically as similar as possible.
|
||||
|
||||
|
||||
After each workload is executed, the database is cleaned, and a new dataset is imported to provide a clean start for the following workload run. When executing isolated and mixed workloads, the database is also restarted after executing each query to minimize the impact on the following query execution.
|
||||
|
||||
### Pokec
|
||||
|
||||
Currently, the only available dataset to run the benchmarks on is the Slovenian social network, Pokec. It’s available in three different sizes, small, medium, and large.
|
||||
The Slovenian social network, Pokec is available in three different sizes, small, medium, and large.
|
||||
- [small](https://s3.eu-west-1.amazonaws.com/deps.memgraph.io/dataset/pokec/benchmark/pokec_small_import.cypher) - vertices 10,000, edges 121,716
|
||||
- [medium](https://s3.eu-west-1.amazonaws.com/deps.memgraph.io/dataset/pokec/benchmark/pokec_medium_import.cypher) - vertices 100,000, edges 1,768,515
|
||||
- [large](https://s3.eu-west-1.amazonaws.com/deps.memgraph.io/dataset/pokec/benchmark/pokec_large.setup.cypher.gz) - vertices 1,632,803, edges 30,622,564.
|
||||
|
||||
Dataset is imported as a CYPHERL file of Cypher queries. Feel free to check dataset links for complete Cypher queries.
|
||||
Once the script is started, a single index is configured on (:User{id}). Only then are queries executed.
|
||||
|
||||
Index queries for each supported vendor can be downloaded from “https://s3.eu-west-1.amazonaws.com/deps.memgraph.io/dataset/pokec/benchmark/vendor_name.cypher”, just make sure to use the proper vendor name such as `memgraph.cypher`.
|
||||
|
||||
### LDBC Interactive
|
||||
|
||||
The LDBC interactive dataset is a social network dataset has support for multiple sizes, currently supported are sf01, sf1, sf3 and sf10. Keep in mind that bigger datasets will take longer to import and execute queries. The dataset is available in the following sizes:
|
||||
|
||||
- [sf01](https://s3.eu-west-1.amazonaws.com/deps.memgraph.io/dataset/ldbc/benchmark/interactive/ldbc_interactive_sf0.1.cypher.gz) - vertices 327,588 edges 1,477,965
|
||||
- [sf1](https://s3.eu-west-1.amazonaws.com/deps.memgraph.io/dataset/ldbc/benchmark/ldbc_sf1_import.cypher) - vertices 3,181,724, edges 17,256,038
|
||||
- [sf3](https://s3.eu-west-1.amazonaws.com/deps.memgraph.io/dataset/ldbc/benchmark/ldbc_sf3_import.cypher) - vertices 9,281,922, edges 52,695,735
|
||||
|
||||
Dataset is imported as a CYPHERL file of Cypher queries. Feel free to check dataset links for complete Cypher queries. Keep in mind that the dataset is imported differently for each vendor. For example, Memgraph uses Cypher queries to import the dataset, while Neo4j uses `neo4j-admin import` tool.
|
||||
|
||||
Index queries for each supported vendor can be downloaded from “https://s3.eu-west-1.amazonaws.com/deps.memgraph.io/dataset/ldbc/benchmark/vendor_name.cypher”, just make sure to use the proper vendor name such as `memgraph.cypher`.
|
||||
|
||||
More details about the dataset in the [Interactive workload class](workloads/ldbc_interactive.py)
|
||||
|
||||
DISCLAIMER: This is NOT an official implementation of an LDBC Benchmark.
|
||||
|
||||
### LDBC Bussines Intelligence
|
||||
|
||||
The LDBC business intelligence dataset is a social network dataset has support for multiple sizes, currently supported are sf1, sf3 and sf10. Keep in mind that bigger datasets will take longer to import and execute queries. The dataset is available in the following sizes:
|
||||
|
||||
- [sf1](https://s3.eu-west-1.amazonaws.com/deps.memgraph.io/dataset/ldbc/benchmark/bi/ldbc_bi_sf1.cypher.gz) - vertices 2,997,352 edges 17,196,776
|
||||
- [sf3](https://s3.eu-west-1.amazonaws.com/deps.memgraph.io/dataset/ldbc/benchmark/bi/ldbc_bi_sf3.cypher.gz) - vertices 1 edges 1
|
||||
- [sf10](https://s3.eu-west-1.amazonaws.com/deps.memgraph.io/dataset/ldbc/benchmark/bi/ldbc_bi_sf10.cypher.gz) - vertices 1 edges 1
|
||||
|
||||
Dataset is imported as a CYPHERL file of Cypher queries. Feel free to check dataset links for complete Cypher queries. Keep in mind that the dataset is imported differently for each vendor. For example, Memgraph uses Cypher queries to import the dataset, while Neo4j uses `neo4j-admin import` tool.
|
||||
|
||||
Index queries for each supported vendor can be downloaded from “https://s3.eu-west-1.amazonaws.com/deps.memgraph.io/dataset/ldbc/benchmark/vendor_name.cypher”, just make sure to use the proper vendor name such as `memgraph.cypher`.
|
||||
|
||||
More details about the dataset in the [Business Intelligence workload class](workloads/ldbc_bi.py)
|
||||
|
||||
DISCLAIMER: This is NOT an official implementation of an LDBC Benchmark.
|
||||
|
||||
#### Query list
|
||||
|
||||
| |Name | Group | Query |
|
||||
|-|-----| -- | ------------ |
|
||||
|Q1|aggregate | aggregate | MATCH (n:User) RETURN n.age, COUNT(*)|
|
||||
|Q2|aggregate_count | aggregate | MATCH (n) RETURN count(n), count(n.age)|
|
||||
|Q3|aggregate_with_filter | aggregate | MATCH (n:User) WHERE n.age >= 18 RETURN n.age, COUNT(*)|
|
||||
|Q4|min_max_avg | aggregate | MATCH (n) RETURN min(n.age), max(n.age), avg(n.age)|
|
||||
|Q5|expansion_1 | analytical | MATCH (s:User {id: $id})-->(n:User) RETURN n.id|
|
||||
|Q6|expansion_1_with_filter| analytical | MATCH (s:User {id: $id})-->(n:User) WHERE n.age >= 18 RETURN n.id|
|
||||
|Q7|expansion_2| analytical | MATCH (s:User {id: $id})-->()-->(n:User) RETURN DISTINCT n.id|
|
||||
|Q8|expansion_2_with_filter| analytical | MATCH (s:User {id: $id})-->()-->(n:User) WHERE n.age >= 18 RETURN DISTINCT n.id|
|
||||
|Q9|expansion_3| analytical | MATCH (s:User {id: $id})-->()-->()-->(n:User) RETURN DISTINCT n.id|
|
||||
|Q10|expansion_3_with_filter| analytical | MATCH (s:User {id: $id})-->()-->()-->(n:User) WHERE n.age >= 18 RETURN DISTINCT n.id|
|
||||
|Q11|expansion_4| analytical | MATCH (s:User {id: $id})-->()-->()-->()-->(n:User) RETURN DISTINCT n.id|
|
||||
|Q12|expansion_4_with_filter| analytical | MATCH (s:User {id: $id})-->()-->()-->()-->(n:User) WHERE n.age >= 18 RETURN DISTINCT n.id|
|
||||
|Q13|neighbours_2| analytical | MATCH (s:User {id: $id})-[*1..2]->(n:User) RETURN DISTINCT n.id|
|
||||
|Q14|neighbours_2_with_filter| analytical | MATCH (s:User {id: $id})-[*1..2]->(n:User) WHERE n.age >= 18 RETURN DISTINCT n.id|
|
||||
|Q15|neighbours_2_with_data| analytical | MATCH (s:User {id: $id})-[*1..2]->(n:User) RETURN DISTINCT n.id, n|
|
||||
|Q16|neighbours_2_with_data_and_filter| analytical | MATCH (s:User {id: $id})-[*1..2]->(n:User) WHERE n.age >= 18 RETURN DISTINCT n.id, n|
|
||||
|Q17|pattern_cycle| analytical | MATCH (n:User {id: $id})-[e1]->(m)-[e2]->(n) RETURN e1, m, e2|
|
||||
|Q18|pattern_long| analytical | MATCH (n1:User {id: $id})-[e1]->(n2)-[e2]->(n3)-[e3]->(n4)<-[e4]-(n5) RETURN n5 LIMIT 1|
|
||||
|Q19|pattern_short| analytical | MATCH (n:User {id: $id})-[e]->(m) RETURN m LIMIT 1|
|
||||
|Q20|single_edge_write| write | MATCH (n:User {id: $from}), (m:User {id: $to}) WITH n, m CREATE (n)-[e:Temp]->(m) RETURN e|
|
||||
|Q21|single_vertex_write| write |CREATE (n:UserTemp {id : $id}) RETURN n|
|
||||
|Q22|single_vertex_property_update| update | MATCH (n:User {id: $id}) SET n.property = -1|
|
||||
|Q23|single_vertex_read| read | MATCH (n:User {id : $id}) RETURN n|
|
||||
The queries are executed for each dataset independently and on each dataset size the identical queries are used. Query parameters differ between different dataset sizes. The complete list of queries can be found in the following files for each workload:
|
||||
|
||||
- [Pokec](workloads/pokec.py)
|
||||
- [LDBC Interactive](workloads/ldbc_interactive.py)
|
||||
- [LDBC Business Intelligence](workloads/ldbc_bi.py)
|
||||
|
||||
|
||||
## :computer: Platform
|
||||
|
||||
Testing on different hardware platforms and cloudVMs is essential for validating benchmark results. Currently, the tests are run on two different platforms.
|
||||
|
||||
### Intel - HP
|
||||
### Intel
|
||||
|
||||
- Server: HP DL360 G6
|
||||
- CPU: 2 x Intel Xeon X5650 6C12T @ 2.67GHz
|
||||
- RAM: 144GB
|
||||
- OS: Debian 4.19
|
||||
|
||||
|
||||
### AMD
|
||||
|
||||
- CPU: AMD Ryzen 7 3800X 8-Core Processor
|
||||
- RAM: 64GB
|
||||
|
||||
## :nut_and_bolt: Supported databases
|
||||
|
||||
Due to current [database compatibility](link) requirements, the only supported database systems at the moment are:
|
||||
1. Memgraph v2.4
|
||||
2. Neo4j Community Edition v5.1.
|
||||
1. Memgraph v2.7
|
||||
2. Neo4j Community Edition v5.6.
|
||||
|
||||
### Database notes
|
||||
|
||||
@@ -276,16 +324,45 @@ Running configurations that differ from default configuration:
|
||||
|
||||
## :raised_hands: Contributions
|
||||
|
||||
As previously stated, mgBench will expand, and we will need help adding more datasets, queries, databases, and support for protocols in mgBench. Feel free to contribute to any of those, and throw us a start :star:!
|
||||
As previously stated, Benchgraph will expand, and we will need help adding more datasets, queries, databases, and support for protocols in Benchgraph. Feel free to contribute to any of those, and throw us a start :star:!
|
||||
|
||||
## :mega: History and Future of mgBench
|
||||
### History of mgBench
|
||||
## :mega: History and Future of Benchgraph
|
||||
### History of Benchgraph
|
||||
|
||||
Infrastructure around mgBench was developed to test and maintain Memgraph performance. When critical code is changed, a performance test is run on Memgraph’s CI/CD infrastructure to ensure performance is not impacted. Due to the usage of mgBench for internal testing, some parts of the code are still tightly connected to Memgraph’s CI/CD infrastructure. The remains of that code do not impact benchmark setup or performance in any way.
|
||||
Infrastructure around Benchgraph (previously mgBench) was developed to test and maintain Memgraph performance. When critical code is changed, a performance test is run on Memgraph’s CI/CD infrastructure to ensure performance is not impacted. Due to the usage of Benchgraph for internal testing, some parts of the code are still tightly connected to Memgraph’s CI/CD infrastructure. The remains of that code do not impact benchmark setup or performance in any way.
|
||||
|
||||
### Future of mgBench
|
||||
We have big plans for mgBench infrastructure that refers to the above mentioned [limitations](#limitations). Even though a basic dataset can give a solid indication of performance, adding bigger and more complex datasets is a priority to enable the execution of complex analytical queries.
|
||||
### Future of Benchgraph
|
||||
We have big plans for Benchgraph infrastructure that refers to the above mentioned [limitations](#limitations).
|
||||
|
||||
Also high on the list is expanding the list of vendors and providing support for different protocols and languages. The goal is to use mgBench to see how well Memgraph performs on various benchmarks tasks and publicly commit to improving.
|
||||
Also high on the list is expanding the list of vendors and providing support for different protocols and languages. The goal is to use Benchgraph to see how well Memgraph performs on various benchmarks tasks and publicly commit to improving.
|
||||
|
||||
mgBench is currently a passive benchmark since resource usage and saturation across execution are not tracked. Sanity checks were performed, but these values are needed to get the full picture after each test. mgBench also deserves its own repository, and it will be decoupled from Memgraph’s testing infrastructure.
|
||||
Benchgraph is currently a passive benchmark since resource usage and saturation across execution are not tracked. Sanity checks were performed, but these values are needed to get the full picture after each test. Benchgraph also deserves its own repository, and it will be decoupled from Memgraph’s testing infrastructure.
|
||||
|
||||
## Changelog Benchgraph public benchmark
|
||||
|
||||
Latest version: https://memgraph.com/benchgraph
|
||||
|
||||
### Release v2 (latest) - 2023-25-04
|
||||
|
||||
- Benchmark presets:
|
||||
- single-threaded-runtime = 30 seconds
|
||||
- number-of-workers-for-benchmark = 12, 24, 48
|
||||
- query-count-lower-bound = 300 queries
|
||||
- Mixed and realistic workload queries = 500 queries
|
||||
|
||||
- [Full results](https://github.com/memgraph/benchgraph/blob/main/results/benchmark.json)
|
||||
|
||||
- Memgraph got a label index on :User node for Pokec dataset, Neo4j has that index by default.
|
||||
|
||||
### Release v1 - 2022-30-11
|
||||
|
||||
- Benchmark presets:
|
||||
- single-threaded-runtime = 10 seconds
|
||||
- number-of-workers-for-benchmark = 12
|
||||
- query-count-lower-bound = 30 queries
|
||||
- Mixed and realistic workload queries = 100 queries
|
||||
|
||||
- Results for [Memgraph cold](https://github.com/memgraph/benchgraph/blob/main/results/memgraph_cold.json)
|
||||
- Results for [Memgraph warm](https://github.com/memgraph/benchgraph/blob/main/results/memgraph_hot.json)
|
||||
- Results for [Neo4j cold](https://github.com/memgraph/benchgraph/blob/main/results/neo4j_cold.json)
|
||||
- Results for [Neo4j warm](https://github.com/memgraph/benchgraph/blob/main/results/neo4j_warm.json)
|
||||
|
||||
163
tests/mgbench/benchgraph.sh
Executable file
163
tests/mgbench/benchgraph.sh
Executable file
@@ -0,0 +1,163 @@
|
||||
#!/bin/bash -e
|
||||
|
||||
pushd () { command pushd "$@" > /dev/null; }
|
||||
popd () { command popd "$@" > /dev/null; }
|
||||
SCRIPT_DIR="$( cd "$( dirname "${BASH_SOURCE[0]}" )" && pwd )"
|
||||
pushd "$SCRIPT_DIR"
|
||||
|
||||
print_help () {
|
||||
echo -e "$0\t\t => runs all available benchmarks with the prompt"
|
||||
echo -e "$0 run_all\t => runs all available benchmarks"
|
||||
echo -e "$0 zip\t => packages all result files and info about the system"
|
||||
echo -e "$0 clean\t => removes all result files including the zip"
|
||||
echo -e "$0 -h\t => prints help"
|
||||
echo ""
|
||||
echo " env vars:"
|
||||
echo " MGBENCH_MEMGRAPH_BIN_PATH -> path to the memgraph binary in the release mode"
|
||||
echo " MGBENCH_NEO_BIN_PATH -> path to the neo4j binary"
|
||||
exit 0
|
||||
}
|
||||
|
||||
MG_PATH="${MGBENCH_MEMGRAPH_BIN_PATH:-$SCRIPT_DIR/../../build/memgraph}"
|
||||
NEO_PATH="${MGBENCH_NEO_BIN_PATH:-$SCRIPT_DIR/../../libs/neo4j/bin/neo4j}"
|
||||
# If you want to skip some of the workloads or workers, just comment lines
|
||||
# under the WORKLOADS or WORKERS variables.
|
||||
WORKLOADS=(
|
||||
pokec_small
|
||||
pokec_medium
|
||||
ldbc_interactive_sf0_1
|
||||
ldbc_interactive_sf1
|
||||
ldbc_bi_sf1
|
||||
ldbc_interactive_sf3
|
||||
ldbc_bi_sf3
|
||||
)
|
||||
WORKERS=(
|
||||
24
|
||||
48
|
||||
)
|
||||
|
||||
check_binary () {
|
||||
binary_path=$1
|
||||
if [ -f "$binary_path" ]; then
|
||||
echo "$binary_path found."
|
||||
else
|
||||
echo "Failed to find $binary_path exiting..."
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
check_all_binaries () {
|
||||
check_binary "$MG_PATH"
|
||||
check_binary "$NEO_PATH"
|
||||
}
|
||||
|
||||
pokec_small () {
|
||||
workers=$1
|
||||
echo "running ${FUNCNAME[0]} with $workers client workers"
|
||||
python3 graph_bench.py --vendor memgraph "$MG_PATH" --vendor neo4j "$NEO_PATH" \
|
||||
--dataset-name pokec --dataset-group basic --dataset-size small \
|
||||
--realistic 500 30 70 0 0 \
|
||||
--realistic 500 30 70 0 0 \
|
||||
--realistic 500 50 50 0 0 \
|
||||
--realistic 500 70 30 0 0 \
|
||||
--realistic 500 30 40 10 20 \
|
||||
--mixed 500 30 0 0 0 70 \
|
||||
--num-workers-for-benchmark "$workers"
|
||||
}
|
||||
|
||||
pokec_medium () {
|
||||
workers=$1
|
||||
echo "running ${FUNCNAME[0]} with $workers client workers"
|
||||
python3 graph_bench.py --vendor memgraph "$MG_PATH" --vendor neo4j "$NEO_PATH" \
|
||||
--dataset-name pokec --dataset-group basic --dataset-size medium \
|
||||
--realistic 500 30 70 0 0 \
|
||||
--realistic 500 30 70 0 0 \
|
||||
--realistic 500 50 50 0 0 \
|
||||
--realistic 500 70 30 0 0 \
|
||||
--realistic 500 30 40 10 20 \
|
||||
--mixed 500 30 0 0 0 70 \
|
||||
--num-workers-for-benchmark "$workers"
|
||||
}
|
||||
|
||||
ldbc_interactive_sf0_1 () {
|
||||
workers=$1
|
||||
echo "running ${FUNCNAME[0]} with $workers client workers"
|
||||
python3 graph_bench.py --vendor memgraph "$MG_PATH" --vendor neo4j "$NEO_PATH" \
|
||||
--dataset-name ldbc_interactive --dataset-group interactive --dataset-size sf0.1 \
|
||||
--num-workers-for-benchmark "$workers"
|
||||
}
|
||||
|
||||
ldbc_interactive_sf1 () {
|
||||
workers=$1
|
||||
echo "running ${FUNCNAME[0]} with $workers client workers"
|
||||
python3 graph_bench.py --vendor memgraph "$MG_PATH" --vendor neo4j "$NEO_PATH" \
|
||||
--dataset-name ldbc_interactive --dataset-group interactive --dataset-size sf1 \
|
||||
--num-workers-for-benchmark "$workers"
|
||||
}
|
||||
|
||||
ldbc_bi_sf1 () {
|
||||
workers=$1
|
||||
echo "running ${FUNCNAME[0]} with $workers client workers"
|
||||
python3 graph_bench.py --vendor memgraph "$MG_PATH" --vendor neo4j "$NEO_PATH" \
|
||||
--dataset-name ldbc_bi --dataset-group bi --dataset-size sf1 \
|
||||
--num-workers-for-benchmark "$workers"
|
||||
}
|
||||
|
||||
ldbc_interactive_sf3 () {
|
||||
workers=$1
|
||||
echo "running ${FUNCNAME[0]} with $workers client workers"
|
||||
python3 graph_bench.py --vendor memgraph "$MG_PATH" --vendor neo4j "$NEO_PATH" \
|
||||
--dataset-name ldbc_interactive --dataset-group interactive --dataset-size sf3 \
|
||||
--num-workers-for-benchmark "$workers"
|
||||
}
|
||||
|
||||
ldbc_bi_sf3 () {
|
||||
workers=$1
|
||||
echo "running ${FUNCNAME[0]} with $workers client workers"
|
||||
python3 graph_bench.py --vendor memgraph "$MG_PATH" --vendor neo4j "$NEO_PATH" \
|
||||
--dataset-name ldbc_bi --dataset-group bi --dataset-size sf3 \
|
||||
--num-workers-for-benchmark "$workers"
|
||||
}
|
||||
|
||||
run_all () {
|
||||
for workload in "${WORKLOADS[@]}"; do
|
||||
for workers in "${WORKERS[@]}"; do
|
||||
$workload "$workers"
|
||||
sleep 1
|
||||
done
|
||||
done
|
||||
}
|
||||
|
||||
package_all_results () {
|
||||
cat /proc/cpuinfo > cpu.sysinfo
|
||||
cat /proc/meminfo > mem.sysinfo
|
||||
zip data.zip ./*.json ./*.report ./*.log ./*.sysinfo
|
||||
}
|
||||
|
||||
clean_all_results () {
|
||||
rm data.zip ./*.json ./*.report ./*.log ./*.sysinfo
|
||||
}
|
||||
|
||||
if [ "$#" -eq 0 ]; then
|
||||
check_all_binaries
|
||||
read -p "Run all benchmarks? y|Y for YES, anything else NO " -n 1 -r
|
||||
if [[ $REPLY =~ ^[Yy]$ ]]; then
|
||||
run_all
|
||||
fi
|
||||
elif [ "$#" -eq 1 ]; then
|
||||
case $1 in
|
||||
run_all)
|
||||
run_all
|
||||
;;
|
||||
zip)
|
||||
package_all_results
|
||||
;;
|
||||
clean)
|
||||
clean_all_results
|
||||
;;
|
||||
*)
|
||||
print_help
|
||||
;;
|
||||
esac
|
||||
else
|
||||
print_help
|
||||
fi
|
||||
@@ -14,27 +14,28 @@
|
||||
import argparse
|
||||
import json
|
||||
import multiprocessing
|
||||
import pathlib
|
||||
import platform
|
||||
import random
|
||||
import sys
|
||||
|
||||
import helpers
|
||||
import log
|
||||
import runners
|
||||
import setup
|
||||
from benchmark_context import BenchmarkContext
|
||||
from workloads import *
|
||||
|
||||
WITH_FINE_GRAINED_AUTHORIZATION = "with_fine_grained_authorization"
|
||||
WITHOUT_FINE_GRAINED_AUTHORIZATION = "without_fine_grained_authorization"
|
||||
QUERY_COUNT_LOWER_BOUND = 30
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = argparse.ArgumentParser(description="Main parser.", add_help=False)
|
||||
|
||||
parser = argparse.ArgumentParser(
|
||||
description="Memgraph benchmark executor.",
|
||||
formatter_class=argparse.ArgumentDefaultsHelpFormatter,
|
||||
)
|
||||
parser.add_argument(
|
||||
benchmark_parser = argparse.ArgumentParser(description="Benchmark arguments parser", add_help=False)
|
||||
|
||||
benchmark_parser.add_argument(
|
||||
"benchmarks",
|
||||
nargs="*",
|
||||
default=None,
|
||||
@@ -48,80 +49,65 @@ def parse_args():
|
||||
"the default group is '*' which selects all groups; the"
|
||||
"default query is '*' which selects all queries",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--vendor-binary",
|
||||
help="Vendor binary used for benchmarking, by default it is memgraph",
|
||||
default=helpers.get_binary_path("memgraph"),
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
"--vendor-name",
|
||||
default="memgraph",
|
||||
choices=["memgraph", "neo4j"],
|
||||
help="Input vendor binary name (memgraph, neo4j)",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--client-binary",
|
||||
default=helpers.get_binary_path("tests/mgbench/client"),
|
||||
help="Client binary used for benchmarking",
|
||||
)
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--num-workers-for-import",
|
||||
type=int,
|
||||
default=multiprocessing.cpu_count() // 2,
|
||||
help="number of workers used to import the dataset",
|
||||
)
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--num-workers-for-benchmark",
|
||||
type=int,
|
||||
default=1,
|
||||
help="number of workers used to execute the benchmark",
|
||||
)
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--single-threaded-runtime-sec",
|
||||
type=int,
|
||||
default=10,
|
||||
help="single threaded duration of each query",
|
||||
)
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--query-count-lower-bound",
|
||||
type=int,
|
||||
default=30,
|
||||
help="Lower bound for query count, minimum number of queries that will be executed. If approximated --single-threaded-runtime-sec query count is lower than this value, lower bound is used.",
|
||||
)
|
||||
benchmark_parser.add_argument(
|
||||
"--no-load-query-counts",
|
||||
action="store_true",
|
||||
default=False,
|
||||
help="disable loading of cached query counts",
|
||||
)
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--no-save-query-counts",
|
||||
action="store_true",
|
||||
default=False,
|
||||
help="disable storing of cached query counts",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--export-results",
|
||||
default=None,
|
||||
help="file path into which results should be exported",
|
||||
)
|
||||
parser.add_argument(
|
||||
"--temporary-directory",
|
||||
default="/tmp",
|
||||
help="directory path where temporary data should be stored",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--no-authorization",
|
||||
action="store_false",
|
||||
default=True,
|
||||
help="Run each query with authorization",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--warm-up",
|
||||
default="cold",
|
||||
choices=["cold", "hot", "vulcanic"],
|
||||
help="Run different warmups before benchmarks sample starts",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--workload-realistic",
|
||||
nargs="*",
|
||||
type=int,
|
||||
@@ -134,7 +120,7 @@ def parse_args():
|
||||
70% read, 10% update and 0% analytical.""",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--workload-mixed",
|
||||
nargs="*",
|
||||
type=int,
|
||||
@@ -147,29 +133,66 @@ def parse_args():
|
||||
with the presence of 300 write queries from write type or 30%""",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--time-depended-execution",
|
||||
type=int,
|
||||
default=0,
|
||||
help="Execute defined number of queries (based on single-threaded-runtime-sec) for a defined duration in of wall-clock time",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--performance-tracking",
|
||||
action="store_true",
|
||||
default=False,
|
||||
help="Flag for runners performance tracking, this logs RES through time and vendor specific performance tracking.",
|
||||
)
|
||||
|
||||
parser.add_argument("--customer-workloads", default=None, help="Path to customers workloads")
|
||||
benchmark_parser.add_argument("--customer-workloads", default=None, help="Path to customers workloads")
|
||||
|
||||
parser.add_argument(
|
||||
benchmark_parser.add_argument(
|
||||
"--vendor-specific",
|
||||
nargs="*",
|
||||
default=[],
|
||||
help="Vendor specific arguments that can be applied to each vendor, format: [key=value, key=value ...]",
|
||||
)
|
||||
|
||||
subparsers = parser.add_subparsers(help="Subparsers", dest="run_option")
|
||||
|
||||
# Vendor native parser starts here
|
||||
parser_vendor_native = subparsers.add_parser(
|
||||
"vendor-native",
|
||||
help="Running database in binary native form",
|
||||
parents=[benchmark_parser],
|
||||
)
|
||||
parser_vendor_native.add_argument(
|
||||
"--vendor-name",
|
||||
default="memgraph",
|
||||
choices=["memgraph", "neo4j"],
|
||||
help="Input vendor binary name (memgraph, neo4j)",
|
||||
)
|
||||
parser_vendor_native.add_argument(
|
||||
"--vendor-binary",
|
||||
help="Vendor binary used for benchmarking, by default it is memgraph",
|
||||
default=helpers.get_binary_path("memgraph"),
|
||||
)
|
||||
|
||||
parser_vendor_native.add_argument(
|
||||
"--client-binary",
|
||||
default=helpers.get_binary_path("tests/mgbench/client"),
|
||||
help="Client binary used for benchmarking",
|
||||
)
|
||||
|
||||
# Vendor docker parsers starts here
|
||||
parser_vendor_docker = subparsers.add_parser(
|
||||
"vendor-docker", help="Running database in docker", parents=[benchmark_parser]
|
||||
)
|
||||
parser_vendor_docker.add_argument(
|
||||
"--vendor-name",
|
||||
default="memgraph",
|
||||
choices=["memgraph-docker", "neo4j-docker"],
|
||||
help="Input vendor name to run in docker (memgraph-docker, neo4j-docker)",
|
||||
)
|
||||
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
@@ -184,28 +207,28 @@ def get_queries(gen, count):
|
||||
|
||||
|
||||
def warmup(condition: str, client: runners.BaseRunner, queries: list = None):
|
||||
log.log("Database condition {} ".format(condition))
|
||||
log.init("Started warm-up procedure to match database condition: {} ".format(condition))
|
||||
if condition == "hot":
|
||||
log.log("Execute warm-up to match condition {} ".format(condition))
|
||||
log.log("Execute warm-up to match condition: {} ".format(condition))
|
||||
client.execute(
|
||||
queries=[
|
||||
("CREATE ();", {}),
|
||||
("CREATE ()-[:TempEdge]->();", {}),
|
||||
("MATCH (n) RETURN n LIMIT 1;", {}),
|
||||
("MATCH (n) RETURN count(n.prop) LIMIT 1;", {}),
|
||||
],
|
||||
num_workers=1,
|
||||
)
|
||||
elif condition == "vulcanic":
|
||||
log.log("Execute warm-up to match condition {} ".format(condition))
|
||||
log.log("Execute warm-up to match condition: {} ".format(condition))
|
||||
client.execute(queries=queries)
|
||||
else:
|
||||
log.log("No warm-up on condition {} ".format(condition))
|
||||
log.log("No warm-up on condition: {} ".format(condition))
|
||||
log.log("Finished warm-up procedure to match database condition: {} ".format(condition))
|
||||
|
||||
|
||||
def mixed_workload(
|
||||
vendor: runners.BaseRunner, client: runners.BaseClient, dataset, group, queries, benchmark_context: BenchmarkContext
|
||||
):
|
||||
|
||||
num_of_queries = benchmark_context.mode_config[0]
|
||||
percentage_distribution = benchmark_context.mode_config[1:]
|
||||
if sum(percentage_distribution) != 100:
|
||||
@@ -233,7 +256,7 @@ def mixed_workload(
|
||||
"analytical": [],
|
||||
}
|
||||
|
||||
for (_, funcname) in queries[group]:
|
||||
for _, funcname in queries[group]:
|
||||
for key in queries_by_type.keys():
|
||||
if key in funcname:
|
||||
queries_by_type[key].append(funcname)
|
||||
@@ -252,8 +275,7 @@ def mixed_workload(
|
||||
full_workload = []
|
||||
|
||||
log.info(
|
||||
"Running query in mixed workload:",
|
||||
"{}/{}/{}".format(
|
||||
"Running query in mixed workload: {}/{}/{}".format(
|
||||
group,
|
||||
query,
|
||||
funcname,
|
||||
@@ -278,7 +300,7 @@ def mixed_workload(
|
||||
additional_query = getattr(dataset, funcname)
|
||||
full_workload.append(additional_query())
|
||||
|
||||
vendor.start_benchmark(
|
||||
vendor.start_db(
|
||||
dataset.NAME + dataset.get_variant() + "_" + "mixed" + "_" + query + "_" + config_distribution
|
||||
)
|
||||
warmup(benchmark_context.warm_up, client=client)
|
||||
@@ -286,7 +308,7 @@ def mixed_workload(
|
||||
queries=full_workload,
|
||||
num_workers=benchmark_context.num_workers_for_benchmark,
|
||||
)[0]
|
||||
usage_workload = vendor.stop(
|
||||
usage_workload = vendor.stop_db(
|
||||
dataset.NAME + dataset.get_variant() + "_" + "mixed" + "_" + query + "_" + config_distribution
|
||||
)
|
||||
|
||||
@@ -313,13 +335,13 @@ def mixed_workload(
|
||||
additional_query = getattr(dataset, funcname)
|
||||
full_workload.append(additional_query())
|
||||
|
||||
vendor.start_benchmark(dataset.NAME + dataset.get_variant() + "_" + "realistic" + "_" + config_distribution)
|
||||
vendor.start_db(dataset.NAME + dataset.get_variant() + "_" + "realistic" + "_" + config_distribution)
|
||||
warmup(benchmark_context.warm_up, client=client)
|
||||
ret = client.execute(
|
||||
queries=full_workload,
|
||||
num_workers=benchmark_context.num_workers_for_benchmark,
|
||||
)[0]
|
||||
usage_workload = vendor.stop(
|
||||
usage_workload = vendor.stop_db(
|
||||
dataset.NAME + dataset.get_variant() + "_" + "realistic" + "_" + config_distribution
|
||||
)
|
||||
mixed_workload = {
|
||||
@@ -349,7 +371,6 @@ def get_query_cache_count(
|
||||
config_key: list,
|
||||
benchmark_context: BenchmarkContext,
|
||||
):
|
||||
|
||||
cached_count = config.get_value(*config_key)
|
||||
if cached_count is None:
|
||||
log.info(
|
||||
@@ -357,8 +378,9 @@ def get_query_cache_count(
|
||||
benchmark_context.single_threaded_runtime_sec
|
||||
)
|
||||
)
|
||||
log.log("Running query to prime the query cache...")
|
||||
# First run to prime the query caches.
|
||||
vendor.start_benchmark("cache")
|
||||
vendor.start_db("cache")
|
||||
client.execute(queries=queries, num_workers=1)
|
||||
# Get a sense of the runtime.
|
||||
count = 1
|
||||
@@ -379,11 +401,10 @@ def get_query_cache_count(
|
||||
break
|
||||
else:
|
||||
count = count * 10
|
||||
vendor.stop("cache")
|
||||
vendor.stop_db("cache")
|
||||
|
||||
QUERY_COUNT_LOWER_BOUND = 30
|
||||
if count < QUERY_COUNT_LOWER_BOUND:
|
||||
count = QUERY_COUNT_LOWER_BOUND
|
||||
if count < benchmark_context.query_count_lower_bound:
|
||||
count = benchmark_context.query_count_lower_bound
|
||||
|
||||
config.set_value(
|
||||
*config_key,
|
||||
@@ -394,7 +415,7 @@ def get_query_cache_count(
|
||||
)
|
||||
else:
|
||||
log.log(
|
||||
"Using cached query count of {} queries for {} seconds of single-threaded runtime.".format(
|
||||
"Using cached query count of {} queries for {} seconds of single-threaded runtime to extrapolate .".format(
|
||||
cached_count["count"], cached_count["duration"]
|
||||
),
|
||||
)
|
||||
@@ -403,14 +424,10 @@ def get_query_cache_count(
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
args = parse_args()
|
||||
vendor_specific_args = helpers.parse_kwargs(args.vendor_specific)
|
||||
|
||||
assert args.benchmarks != None, helpers.list_available_workloads()
|
||||
assert args.vendor_name == "memgraph" or args.vendor_name == "neo4j", "Unsupported vendors"
|
||||
assert args.vendor_binary != None, "Pass database binary for runner"
|
||||
assert args.client_binary != None, "Pass client binary for benchmark client "
|
||||
assert args.num_workers_for_import > 0
|
||||
assert args.num_workers_for_benchmark > 0
|
||||
assert args.export_results != None, "Pass where will results be saved"
|
||||
@@ -421,17 +438,21 @@ if __name__ == "__main__":
|
||||
args.workload_realistic == None or args.workload_mixed == None
|
||||
), "Cannot run both realistic and mixed workload, only one mode run at the time"
|
||||
|
||||
temp_dir = pathlib.Path.cwd() / ".temp"
|
||||
temp_dir.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
benchmark_context = BenchmarkContext(
|
||||
benchmark_target_workload=args.benchmarks,
|
||||
vendor_binary=args.vendor_binary,
|
||||
vendor_name=args.vendor_name,
|
||||
client_binary=args.client_binary,
|
||||
vendor_binary=args.vendor_binary if args.run_option == "vendor-native" else None,
|
||||
vendor_name=args.vendor_name.replace("-", ""),
|
||||
client_binary=args.client_binary if args.run_option == "vendor-native" else None,
|
||||
num_workers_for_import=args.num_workers_for_import,
|
||||
num_workers_for_benchmark=args.num_workers_for_benchmark,
|
||||
single_threaded_runtime_sec=args.single_threaded_runtime_sec,
|
||||
query_count_lower_bound=args.query_count_lower_bound,
|
||||
no_load_query_counts=args.no_load_query_counts,
|
||||
export_results=args.export_results,
|
||||
temporary_directory=args.temporary_directory,
|
||||
temporary_directory=temp_dir.absolute(),
|
||||
workload_mixed=args.workload_mixed,
|
||||
workload_realistic=args.workload_realistic,
|
||||
time_dependent_execution=args.time_depended_execution,
|
||||
@@ -444,10 +465,16 @@ if __name__ == "__main__":
|
||||
|
||||
log.init("Executing benchmark with following arguments: ")
|
||||
for key, value in benchmark_context.__dict__.items():
|
||||
log.log(str(key) + " : " + str(value))
|
||||
log.log("{:<30} : {:<30}".format(str(key), str(value)))
|
||||
|
||||
log.init("Check requirements for running benchmark")
|
||||
if setup.check_requirements(benchmark_context=benchmark_context):
|
||||
log.success("Requirements satisfied... ")
|
||||
else:
|
||||
log.warning("Requirements not satisfied...")
|
||||
sys.exit(1)
|
||||
|
||||
log.log("Creating cache folder for: dataset, configurations, indexes, results etc. ")
|
||||
# Create cache, config and results objects.
|
||||
cache = helpers.Cache()
|
||||
log.init("Folder in use: " + cache.get_default_cache_directory())
|
||||
if not benchmark_context.no_load_query_counts:
|
||||
@@ -457,15 +484,11 @@ if __name__ == "__main__":
|
||||
config = helpers.RecursiveDict()
|
||||
results = helpers.RecursiveDict()
|
||||
|
||||
log.init("Creating vendor runner for DB: " + benchmark_context.vendor_name)
|
||||
vendor_runner = runners.BaseRunner.create(
|
||||
benchmark_context=benchmark_context,
|
||||
)
|
||||
log.log("Class in use: " + str(vendor_runner))
|
||||
|
||||
run_config = {
|
||||
"vendor": benchmark_context.vendor_name,
|
||||
"condition": benchmark_context.warm_up,
|
||||
"num_workers_for_benchmark": benchmark_context.num_workers_for_benchmark,
|
||||
"single_threaded_runtime_sec": benchmark_context.single_threaded_runtime_sec,
|
||||
"benchmark_mode": benchmark_context.mode,
|
||||
"benchmark_mode_config": benchmark_context.mode_config,
|
||||
"platform": platform.platform(),
|
||||
@@ -475,53 +498,78 @@ if __name__ == "__main__":
|
||||
|
||||
available_workloads = helpers.get_available_workloads(benchmark_context.customer_workloads)
|
||||
|
||||
log.init("Currently available workloads: ")
|
||||
log.log(helpers.list_available_workloads(benchmark_context.customer_workloads))
|
||||
|
||||
# Filter out the workloads based on the pattern
|
||||
target_workloads = helpers.filter_workloads(
|
||||
available_workloads=available_workloads, benchmark_context=benchmark_context
|
||||
)
|
||||
|
||||
if len(target_workloads) == 0:
|
||||
log.error("No workloads matched the pattern: " + str(benchmark_context.benchmark_target_workload))
|
||||
log.error("Please check the pattern and workload NAME property, query group and query name.")
|
||||
log.info("Currently available workloads: ")
|
||||
log.log(helpers.list_available_workloads(benchmark_context.customer_workloads))
|
||||
sys.exit(1)
|
||||
|
||||
# Run all target workloads.
|
||||
for workload, queries in target_workloads:
|
||||
log.info("Started running following workload: " + str(workload.NAME))
|
||||
|
||||
benchmark_context.set_active_workload(workload.NAME)
|
||||
benchmark_context.set_active_variant(workload.get_variant())
|
||||
|
||||
log.init("Creating vendor runner for DB: " + benchmark_context.vendor_name)
|
||||
vendor_runner = runners.BaseRunner.create(
|
||||
benchmark_context=benchmark_context,
|
||||
)
|
||||
log.log("Class in use: " + str(vendor_runner.__class__.__name__))
|
||||
|
||||
log.info("Cleaning the database from any previous data")
|
||||
vendor_runner.clean_db()
|
||||
|
||||
client = vendor_runner.fetch_client()
|
||||
log.log("Get appropriate client for vendor " + str(client))
|
||||
log.log("Get appropriate client for vendor " + str(client.__class__.__name__))
|
||||
|
||||
ret = None
|
||||
usage = None
|
||||
|
||||
log.init("Preparing workload: " + workload.NAME + "/" + workload.get_variant())
|
||||
workload.prepare(cache.cache_directory("datasets", workload.NAME, workload.get_variant()))
|
||||
generated_queries = workload.dataset_generator()
|
||||
if generated_queries:
|
||||
vendor_runner.start_preparation("import")
|
||||
print("\n")
|
||||
log.info("Using workload as dataset generator...")
|
||||
if workload.get_index():
|
||||
log.info("Using index from specified file: {}".format(workload.get_index()))
|
||||
client.execute(file_path=workload.get_index(), num_workers=benchmark_context.num_workers_for_import)
|
||||
else:
|
||||
log.warning("Make sure proper indexes/constraints are created in generated queries!")
|
||||
|
||||
vendor_runner.start_db_init("import")
|
||||
|
||||
log.warning("Using following indexes...")
|
||||
log.info(workload.indexes_generator())
|
||||
log.info("Executing database index setup...")
|
||||
ret = client.execute(queries=workload.indexes_generator(), num_workers=1)
|
||||
log.log("Finished setting up indexes...")
|
||||
for row in ret:
|
||||
log.success(
|
||||
"Executed {} queries in {} seconds using {} workers with a total throughput of {} Q/S.".format(
|
||||
row["count"], row["duration"], row["num_workers"], row["throughput"]
|
||||
)
|
||||
)
|
||||
|
||||
log.info("Importing dataset...")
|
||||
ret = client.execute(queries=generated_queries, num_workers=benchmark_context.num_workers_for_import)
|
||||
usage = vendor_runner.stop("import")
|
||||
log.log("Finished importing dataset...")
|
||||
usage = vendor_runner.stop_db_init("import")
|
||||
else:
|
||||
log.init("Preparing workload: " + workload.NAME + "/" + workload.get_variant())
|
||||
workload.prepare(cache.cache_directory("datasets", workload.NAME, workload.get_variant()))
|
||||
log.info("Using workload dataset information for import...")
|
||||
imported = workload.custom_import()
|
||||
if not imported:
|
||||
log.log("Basic import execution")
|
||||
vendor_runner.start_preparation("import")
|
||||
vendor_runner.start_db_init("import")
|
||||
log.log("Executing database index setup...")
|
||||
client.execute(file_path=workload.get_index(), num_workers=benchmark_context.num_workers_for_import)
|
||||
client.execute(file_path=workload.get_index(), num_workers=1)
|
||||
log.log("Importing dataset...")
|
||||
ret = client.execute(
|
||||
file_path=workload.get_file(), num_workers=benchmark_context.num_workers_for_import
|
||||
)
|
||||
usage = vendor_runner.stop("import")
|
||||
usage = vendor_runner.stop_db_init("import")
|
||||
else:
|
||||
log.info("Custom import executed...")
|
||||
|
||||
@@ -531,7 +579,7 @@ if __name__ == "__main__":
|
||||
# Display import statistics.
|
||||
for row in ret:
|
||||
log.success(
|
||||
"Executed {} queries in {} seconds using {} workers with a total throughput of {} + Q/S.".format(
|
||||
"Executed {} queries in {} seconds using {} workers with a total throughput of {} Q/S.".format(
|
||||
row["count"], row["duration"], row["num_workers"], row["throughput"]
|
||||
)
|
||||
)
|
||||
@@ -539,7 +587,7 @@ if __name__ == "__main__":
|
||||
log.success(
|
||||
"The database used {} seconds of CPU time and peaked at {} MiB of RAM".format(
|
||||
usage["cpu"], usage["memory"] / 1024 / 1024
|
||||
),
|
||||
)
|
||||
)
|
||||
|
||||
results.set_value(*import_key, value={"client": ret, "database": usage})
|
||||
@@ -548,6 +596,7 @@ if __name__ == "__main__":
|
||||
|
||||
# Run all benchmarks in all available groups.
|
||||
for group in sorted(queries.keys()):
|
||||
print("\n")
|
||||
log.init("Running benchmark in " + benchmark_context.mode)
|
||||
if benchmark_context.mode == "Mixed":
|
||||
mixed_workload(vendor_runner, client, workload, group, queries, benchmark_context)
|
||||
@@ -568,12 +617,13 @@ if __name__ == "__main__":
|
||||
group,
|
||||
query,
|
||||
]
|
||||
log.init("Determining query count for benchmark based on --single-threaded-runtime argument")
|
||||
count = get_query_cache_count(
|
||||
vendor_runner, client, get_queries(func, 1), config_key, benchmark_context
|
||||
)
|
||||
|
||||
# Benchmark run.
|
||||
log.info("Sample query:{}".format(get_queries(func, 1)[0][0]))
|
||||
sample_query = get_queries(func, 1)[0][0]
|
||||
log.info("Sample query:{}".format(sample_query))
|
||||
log.log(
|
||||
"Executing benchmark with {} queries that should yield a single-threaded runtime of {} seconds.".format(
|
||||
count, benchmark_context.single_threaded_runtime_sec
|
||||
@@ -584,11 +634,11 @@ if __name__ == "__main__":
|
||||
benchmark_context.num_workers_for_benchmark
|
||||
)
|
||||
)
|
||||
vendor_runner.start_benchmark(
|
||||
vendor_runner.start_db(
|
||||
workload.NAME + workload.get_variant() + "_" + "_" + benchmark_context.mode + "_" + query
|
||||
)
|
||||
|
||||
warmup(condition=benchmark_context.warm_up, client=client, queries=get_queries(func, count))
|
||||
log.init("Executing benchmark queries...")
|
||||
if benchmark_context.time_dependent_execution != 0:
|
||||
ret = client.execute(
|
||||
queries=get_queries(func, count),
|
||||
@@ -600,8 +650,8 @@ if __name__ == "__main__":
|
||||
queries=get_queries(func, count),
|
||||
num_workers=benchmark_context.num_workers_for_benchmark,
|
||||
)[0]
|
||||
|
||||
usage = vendor_runner.stop(
|
||||
log.info("Benchmark execution finished...")
|
||||
usage = vendor_runner.stop_db(
|
||||
workload.NAME + workload.get_variant() + "_" + benchmark_context.mode + "_" + query
|
||||
)
|
||||
ret["database"] = usage
|
||||
@@ -610,15 +660,28 @@ if __name__ == "__main__":
|
||||
log.log("Executed {} queries in {} seconds.".format(ret["count"], ret["duration"]))
|
||||
log.log("Queries have been retried {} times".format(ret["retries"]))
|
||||
log.log("Database used {:.3f} seconds of CPU time.".format(usage["cpu"]))
|
||||
log.log("Database peaked at {:.3f} MiB of memory.".format(usage["memory"] / 1024.0 / 1024.0))
|
||||
log.log("{:<31} {:>20} {:>20} {:>20}".format("Metadata:", "min", "avg", "max"))
|
||||
metadata = ret["metadata"]
|
||||
for key in sorted(metadata.keys()):
|
||||
log.log(
|
||||
"{name:>30}: {minimum:>20.06f} {average:>20.06f} "
|
||||
"{maximum:>20.06f}".format(name=key, **metadata[key])
|
||||
)
|
||||
log.info("Database peaked at {:.3f} MiB of memory.".format(usage["memory"] / 1024.0 / 1024.0))
|
||||
if "docker" not in benchmark_context.vendor_name:
|
||||
log.log("{:<31} {:>20} {:>20} {:>20}".format("Metadata:", "min", "avg", "max"))
|
||||
metadata = ret["metadata"]
|
||||
for key in sorted(metadata.keys()):
|
||||
log.log(
|
||||
"{name:>30}: {minimum:>20.06f} {average:>20.06f} "
|
||||
"{maximum:>20.06f}".format(name=key, **metadata[key])
|
||||
)
|
||||
print("\n")
|
||||
log.info("Result:")
|
||||
log.info(funcname)
|
||||
log.info(sample_query)
|
||||
log.success("Latency statistics:")
|
||||
for key, value in ret["latency_stats"].items():
|
||||
if key == "iterations":
|
||||
log.success("{:<10} {:>10}".format(key, value))
|
||||
else:
|
||||
log.success("{:<10} {:>10.06f} seconds".format(key, value))
|
||||
|
||||
log.success("Throughput: {:02f} QPS".format(ret["throughput"]))
|
||||
print("\n\n")
|
||||
|
||||
# Save results.
|
||||
results_key = [
|
||||
@@ -632,8 +695,9 @@ if __name__ == "__main__":
|
||||
|
||||
# If there is need for authorization testing.
|
||||
if benchmark_context.no_authorization:
|
||||
log.info("Running queries with authorization...")
|
||||
vendor_runner.start_benchmark("authorization")
|
||||
log.init("Running queries with authorization...")
|
||||
log.info("Setting USER and PRIVILEGES...")
|
||||
vendor_runner.start_db("authorization")
|
||||
client.execute(
|
||||
queries=[
|
||||
("CREATE USER user IDENTIFIED BY 'test';", {}),
|
||||
@@ -644,13 +708,11 @@ if __name__ == "__main__":
|
||||
)
|
||||
|
||||
client.set_credentials(username="user", password="test")
|
||||
vendor_runner.stop("authorization")
|
||||
vendor_runner.stop_db("authorization")
|
||||
|
||||
for query, funcname in queries[group]:
|
||||
|
||||
log.info(
|
||||
"Running query:",
|
||||
"{}/{}/{}/{}".format(group, query, funcname, WITH_FINE_GRAINED_AUTHORIZATION),
|
||||
log.init(
|
||||
"Running query:" + "{}/{}/{}/{}".format(group, query, funcname, WITH_FINE_GRAINED_AUTHORIZATION)
|
||||
)
|
||||
func = getattr(workload, funcname)
|
||||
|
||||
@@ -664,13 +726,14 @@ if __name__ == "__main__":
|
||||
vendor_runner, client, get_queries(func, 1), config_key, benchmark_context
|
||||
)
|
||||
|
||||
vendor_runner.start_benchmark("authorization")
|
||||
vendor_runner.start_db("authorization")
|
||||
warmup(condition=benchmark_context.warm_up, client=client, queries=get_queries(func, count))
|
||||
|
||||
ret = client.execute(
|
||||
queries=get_queries(func, count),
|
||||
num_workers=benchmark_context.num_workers_for_benchmark,
|
||||
)[0]
|
||||
usage = vendor_runner.stop("authorization")
|
||||
usage = vendor_runner.stop_db("authorization")
|
||||
ret["database"] = usage
|
||||
# Output summary.
|
||||
log.log("Executed {} queries in {} seconds.".format(ret["count"], ret["duration"]))
|
||||
@@ -695,8 +758,8 @@ if __name__ == "__main__":
|
||||
]
|
||||
results.set_value(*results_key, value=ret)
|
||||
|
||||
# Clean up database from any roles and users job
|
||||
vendor_runner.start_benchmark("authorizations")
|
||||
log.info("Deleting USER and PRIVILEGES...")
|
||||
vendor_runner.start_db("authorizations")
|
||||
ret = client.execute(
|
||||
queries=[
|
||||
("REVOKE LABELS * FROM user;", {}),
|
||||
@@ -704,7 +767,7 @@ if __name__ == "__main__":
|
||||
("DROP USER user;", {}),
|
||||
]
|
||||
)
|
||||
vendor_runner.stop("authorization")
|
||||
vendor_runner.stop_db("authorization")
|
||||
|
||||
# Save configuration.
|
||||
if not benchmark_context.no_save_query_counts:
|
||||
@@ -714,3 +777,30 @@ if __name__ == "__main__":
|
||||
if benchmark_context.export_results:
|
||||
with open(benchmark_context.export_results, "w") as f:
|
||||
json.dump(results.get_data(), f)
|
||||
|
||||
# Results summary.
|
||||
log.init("~" * 45)
|
||||
log.info("Benchmark finished.")
|
||||
log.init("~" * 45)
|
||||
log.log("\n")
|
||||
log.summary("Benchmark summary")
|
||||
log.log("-" * 90)
|
||||
log.summary("{:<20} {:>30} {:>30}".format("Query name", "Throughput", "Peak Memory usage"))
|
||||
with open(benchmark_context.export_results, "r") as f:
|
||||
results = json.load(f)
|
||||
for dataset, variants in results.items():
|
||||
if dataset == "__run_configuration__":
|
||||
continue
|
||||
for variant, groups in variants.items():
|
||||
for group, queries in groups.items():
|
||||
if group == "__import__":
|
||||
continue
|
||||
for query, auth in queries.items():
|
||||
for key, value in auth.items():
|
||||
log.log("-" * 90)
|
||||
log.summary(
|
||||
"{:<20} {:>26.2f} QPS {:>27.2f} MB".format(
|
||||
query, value["throughput"], value["database"]["memory"] / 1024.0 / 1024.0
|
||||
)
|
||||
)
|
||||
log.log("-" * 90)
|
||||
|
||||
@@ -1,3 +1,15 @@
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
|
||||
# Describes all the information of single benchmark.py run.
|
||||
class BenchmarkContext:
|
||||
"""
|
||||
@@ -7,12 +19,13 @@ class BenchmarkContext:
|
||||
def __init__(
|
||||
self,
|
||||
benchmark_target_workload: str = None, # Workload that needs to be executed (dataset/variant/group/query)
|
||||
vendor_binary: str = None, # Benchmark vendor binary
|
||||
vendor_binary: str = None,
|
||||
vendor_name: str = None,
|
||||
client_binary: str = None,
|
||||
num_workers_for_import: int = None,
|
||||
num_workers_for_benchmark: int = None,
|
||||
single_threaded_runtime_sec: int = 0,
|
||||
query_count_lower_bound: int = 0,
|
||||
no_load_query_counts: bool = False,
|
||||
no_save_query_counts: bool = False,
|
||||
export_results: str = None,
|
||||
@@ -26,7 +39,6 @@ class BenchmarkContext:
|
||||
customer_workloads: str = None,
|
||||
vendor_args: dict = {},
|
||||
) -> None:
|
||||
|
||||
self.benchmark_target_workload = benchmark_target_workload
|
||||
self.vendor_binary = vendor_binary
|
||||
self.vendor_name = vendor_name
|
||||
@@ -34,6 +46,7 @@ class BenchmarkContext:
|
||||
self.num_workers_for_import = num_workers_for_import
|
||||
self.num_workers_for_benchmark = num_workers_for_benchmark
|
||||
self.single_threaded_runtime_sec = single_threaded_runtime_sec
|
||||
self.query_count_lower_bound = query_count_lower_bound
|
||||
self.no_load_query_counts = no_load_query_counts
|
||||
self.no_save_query_counts = no_save_query_counts
|
||||
self.export_results = export_results
|
||||
@@ -55,3 +68,17 @@ class BenchmarkContext:
|
||||
self.no_authorization = no_authorization
|
||||
self.customer_workloads = customer_workloads
|
||||
self.vendor_args = vendor_args
|
||||
self.active_workload = None
|
||||
self.active_variant = None
|
||||
|
||||
def set_active_workload(self, workload: str) -> None:
|
||||
self.active_workload = workload
|
||||
|
||||
def get_active_workload(self) -> str:
|
||||
return self.active_workload
|
||||
|
||||
def set_active_variant(self, variant: str) -> None:
|
||||
self.active_variant = variant
|
||||
|
||||
def get_active_variant(self) -> str:
|
||||
return self.active_variant
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
# Copyright 2022 Memgraph Ltd.
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -97,8 +97,79 @@ def compare_results(results_from, results_to, fields, ignored, different_vendors
|
||||
return ret
|
||||
|
||||
|
||||
def generate_remarkup(fields, data):
|
||||
ret = "==== Benchmark summary: ====\n\n"
|
||||
def generate_remarkup(fields, data, results_from=None, results_to=None):
|
||||
ret = "<html>\n"
|
||||
ret += """
|
||||
<style>
|
||||
table, th, td {
|
||||
border: 1px solid black;
|
||||
}
|
||||
</style>
|
||||
"""
|
||||
ret += "<h1>Benchmark comparison</h1>\n"
|
||||
if results_from and results_to:
|
||||
ret += """
|
||||
<h2>Benchmark configuration</h2>
|
||||
<table>
|
||||
<tr>
|
||||
<th>Configuration</th>
|
||||
<th>Reference vendor</th>
|
||||
<th>Vendor </th>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>Vendor name</td>
|
||||
<td>{}</td>
|
||||
<td>{}</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>Vendor condition</td>
|
||||
<td>{}</td>
|
||||
<td>{}</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>Number of workers</td>
|
||||
<td>{}</td>
|
||||
<td>{}</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>Single threaded runtime</td>
|
||||
<td>{}</td>
|
||||
<td>{}</td>
|
||||
</tr>
|
||||
<tr>
|
||||
<td>Platform</td>
|
||||
<td>{}</td>
|
||||
<td>{}</td>
|
||||
</tr>
|
||||
</table>
|
||||
""".format(
|
||||
results_from["vendor"],
|
||||
results_to["vendor"],
|
||||
results_from["condition"],
|
||||
results_to["condition"],
|
||||
results_from["num_workers_for_benchmark"],
|
||||
results_to["num_workers_for_benchmark"],
|
||||
results_from["single_threaded_runtime_sec"],
|
||||
results_to["single_threaded_runtime_sec"],
|
||||
results_from["platform"],
|
||||
results_to["platform"],
|
||||
)
|
||||
ret += """
|
||||
<h2>How to read benchmark results</h2>
|
||||
<b> Throughput and latency values:</b>
|
||||
<p> If vendor <b> {} </b> is faster than the reference vendor <b> {} </b>, the result for throughput and latency are show in <b style="color:#008000">green </b>, otherwise <b style="color:#FF0000">red </b>. Percentage difference is visible relative to reference vendor {}. </p>
|
||||
<b> Memory usage:</b>
|
||||
<p> If the vendor <b> {} </b> uses less memory then the reference vendor <b> {} </b>, the result is shown in <b style="color:#008000">green </b>, otherwise <b style="color:#FF0000"> red </b>. Percentage difference for memory is visible relative to reference vendor {}.
|
||||
""".format(
|
||||
results_to["vendor"],
|
||||
results_from["vendor"],
|
||||
results_from["vendor"],
|
||||
results_to["vendor"],
|
||||
results_from["vendor"],
|
||||
results_from["vendor"],
|
||||
)
|
||||
|
||||
ret += "<h2>Benchmark results</h2>\n"
|
||||
if len(data) > 0:
|
||||
ret += "<table>\n"
|
||||
ret += " <tr>\n"
|
||||
@@ -135,6 +206,7 @@ def generate_remarkup(fields, data):
|
||||
ret += '<td bgcolor="blue">{:.3f}{} //(new)// </td>\n'.format(value, field["unit"])
|
||||
ret += " </tr>\n"
|
||||
ret += "</table>\n"
|
||||
ret += "</html>\n"
|
||||
else:
|
||||
ret += "No performance change detected.\n"
|
||||
return ret
|
||||
@@ -276,7 +348,11 @@ if __name__ == "__main__":
|
||||
results_to = load_results(file_to)
|
||||
data.update(compare_results(results_from, results_to, fields, ignored, args.different_vendors))
|
||||
|
||||
remarkup = generate_remarkup(fields, data)
|
||||
results_from_config = (
|
||||
results_from["__run_configuration__"] if "__run_configuration__" in results_from.keys() else None
|
||||
)
|
||||
results_to_config = results_to["__run_configuration__"] if "__run_configuration__" in results_to.keys() else None
|
||||
remarkup = generate_remarkup(fields, data, results_from=results_from_config, results_to=results_to_config)
|
||||
if args.output:
|
||||
with open(args.output, "w") as f:
|
||||
f.write(remarkup)
|
||||
|
||||
@@ -1,3 +1,17 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
# --- DISCLAIMER: This is NOT an official implementation of an LDBC Benchmark. ---
|
||||
import argparse
|
||||
import csv
|
||||
import sys
|
||||
@@ -24,7 +38,6 @@ BI_LINK = {
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
parser = argparse.ArgumentParser(
|
||||
prog="LDBC CSV to CYPHERL converter",
|
||||
description="""Converts all LDBC CSV files to CYPHERL transactions, for faster Memgraph load""",
|
||||
@@ -42,7 +55,6 @@ if __name__ == "__main__":
|
||||
output_directory.mkdir(exist_ok=True)
|
||||
|
||||
if args.type == "interactive":
|
||||
|
||||
NODES_INTERACTIVE = [
|
||||
{"filename": "Place", "label": "Place"},
|
||||
{"filename": "Organisation", "label": "Organisation"},
|
||||
@@ -260,7 +272,6 @@ if __name__ == "__main__":
|
||||
raise Exception("Didn't find the file that was needed!")
|
||||
|
||||
elif args.type == "bi":
|
||||
|
||||
NODES_BI = [
|
||||
{"filename": "Place", "label": "Place"},
|
||||
{"filename": "Organisation", "label": "Organisation"},
|
||||
|
||||
@@ -1,3 +1,16 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import subprocess
|
||||
@@ -54,30 +67,61 @@ def parse_arguments():
|
||||
help="Forward config for query",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
"--num-workers-for-benchmark",
|
||||
type=int,
|
||||
default=12,
|
||||
help="number of workers used to execute the benchmark",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
"--query-count-lower-bound",
|
||||
type=int,
|
||||
default=300,
|
||||
help="number of workers used to execute the benchmark (works only for isolated run)",
|
||||
)
|
||||
|
||||
parser.add_argument(
|
||||
"--single-threaded-runtime-sec",
|
||||
type=int,
|
||||
default=30,
|
||||
help="Duration of single threaded benchmark per query (works only for isolated run)",
|
||||
)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
return args
|
||||
|
||||
|
||||
def run_full_benchmarks(vendor, binary, dataset, dataset_size, dataset_group, realistic, mixed):
|
||||
|
||||
def run_full_benchmarks(
|
||||
vendor,
|
||||
binary,
|
||||
dataset,
|
||||
dataset_size,
|
||||
dataset_group,
|
||||
realistic,
|
||||
mixed,
|
||||
workers,
|
||||
query_count_lower_bound,
|
||||
single_threaded_runtime_sec,
|
||||
):
|
||||
configurations = [
|
||||
# Basic isolated test cold
|
||||
[
|
||||
"--export-results",
|
||||
vendor + "_" + dataset + "_" + dataset_size + "_cold_isolated.json",
|
||||
vendor + "_" + str(workers) + "_" + dataset + "_" + dataset_size + "_cold_isolated.json",
|
||||
],
|
||||
# Basic isolated test hot
|
||||
[
|
||||
"--export-results",
|
||||
vendor + "_" + dataset + "_" + dataset_size + "_hot_isolated.json",
|
||||
vendor + "_" + str(workers) + "_" + dataset + "_" + dataset_size + "_hot_isolated.json",
|
||||
"--warm-up",
|
||||
"hot",
|
||||
],
|
||||
# Basic isolated test vulcanic
|
||||
[
|
||||
"--export-results",
|
||||
vendor + "_" + dataset + "_" + dataset_size + "_vulcanic_isolated.json",
|
||||
vendor + "_" + str(workers) + "_" + dataset + "_" + dataset_size + "_vulcanic_isolated.json",
|
||||
"--warm-up",
|
||||
"vulcanic",
|
||||
],
|
||||
@@ -90,6 +134,8 @@ def run_full_benchmarks(vendor, binary, dataset, dataset_size, dataset_group, re
|
||||
"--export-results",
|
||||
vendor
|
||||
+ "_"
|
||||
+ str(workers)
|
||||
+ "_"
|
||||
+ dataset
|
||||
+ "_"
|
||||
+ dataset_size
|
||||
@@ -106,6 +152,8 @@ def run_full_benchmarks(vendor, binary, dataset, dataset_size, dataset_group, re
|
||||
"--export-results",
|
||||
vendor
|
||||
+ "_"
|
||||
+ str(workers)
|
||||
+ "_"
|
||||
+ dataset
|
||||
+ "_"
|
||||
+ dataset_size
|
||||
@@ -130,6 +178,8 @@ def run_full_benchmarks(vendor, binary, dataset, dataset_size, dataset_group, re
|
||||
"--export-results",
|
||||
vendor
|
||||
+ "_"
|
||||
+ str(workers)
|
||||
+ "_"
|
||||
+ dataset
|
||||
+ "_"
|
||||
+ dataset_size
|
||||
@@ -146,6 +196,8 @@ def run_full_benchmarks(vendor, binary, dataset, dataset_size, dataset_group, re
|
||||
"--export-results",
|
||||
vendor
|
||||
+ "_"
|
||||
+ str(workers)
|
||||
+ "_"
|
||||
+ dataset
|
||||
+ "_"
|
||||
+ dataset_size
|
||||
@@ -167,12 +219,17 @@ def run_full_benchmarks(vendor, binary, dataset, dataset_size, dataset_group, re
|
||||
default_args = [
|
||||
"python3",
|
||||
"benchmark.py",
|
||||
"vendor-native",
|
||||
"--vendor-binary",
|
||||
binary,
|
||||
"--vendor-name",
|
||||
vendor,
|
||||
"--num-workers-for-benchmark",
|
||||
"12",
|
||||
str(workers),
|
||||
"--single-threaded-runtime-sec",
|
||||
str(single_threaded_runtime_sec),
|
||||
"--query-count-lower-bound",
|
||||
str(query_count_lower_bound),
|
||||
"--no-authorization",
|
||||
dataset + "/" + dataset_size + "/" + dataset_group + "/*",
|
||||
]
|
||||
@@ -183,10 +240,12 @@ def run_full_benchmarks(vendor, binary, dataset, dataset_size, dataset_group, re
|
||||
subprocess.run(args=full_config, check=True)
|
||||
|
||||
|
||||
def collect_all_results(vendor_name, dataset, dataset_size, dataset_group):
|
||||
def collect_all_results(vendor_name, dataset, dataset_size, dataset_group, workers):
|
||||
working_directory = Path().absolute()
|
||||
print(working_directory)
|
||||
results = sorted(working_directory.glob(vendor_name + "_" + dataset + "_" + dataset_size + "_*.json"))
|
||||
results = sorted(
|
||||
working_directory.glob(vendor_name + "_" + str(workers) + "_" + dataset + "_" + dataset_size + "_*.json")
|
||||
)
|
||||
summary = {dataset: {dataset_size: {dataset_group: {}}}}
|
||||
|
||||
for file in results:
|
||||
@@ -210,7 +269,7 @@ def collect_all_results(vendor_name, dataset, dataset_size, dataset_group):
|
||||
|
||||
json_object = json.dumps(summary, indent=4)
|
||||
print(json_object)
|
||||
with open(vendor_name + "_" + dataset + "_" + dataset_size + "_summary.json", "w") as f:
|
||||
with open(vendor_name + "_" + str(workers) + "_" + dataset + "_" + dataset_size + "_summary.json", "w") as f:
|
||||
json.dump(summary, f)
|
||||
|
||||
|
||||
@@ -232,8 +291,13 @@ if __name__ == "__main__":
|
||||
args.dataset_group,
|
||||
realistic,
|
||||
mixed,
|
||||
args.num_workers_for_benchmark,
|
||||
args.query_count_lower_bound,
|
||||
args.single_threaded_runtime_sec,
|
||||
)
|
||||
collect_all_results(
|
||||
vendor_name, args.dataset_name, args.dataset_size, args.dataset_group, args.num_workers_for_benchmark
|
||||
)
|
||||
collect_all_results(vendor_name, args.dataset_name, args.dataset_size, args.dataset_group)
|
||||
else:
|
||||
raise Exception(
|
||||
"Check that vendor: {} is supported and you are passing right path: {} to binary.".format(
|
||||
|
||||
219
tests/mgbench/how_to_use_benchgraph.md
Normal file
219
tests/mgbench/how_to_use_benchgraph.md
Normal file
@@ -0,0 +1,219 @@
|
||||
# How to use mgBench
|
||||
|
||||
Running your workloads that include custom queries and the dataset is the best way to evaluate system performance on your use case. Each workload has unique requirements that are imposed from the use case. Since your use-case queries and dataset will be used in production, it is best to use those.
|
||||
We worked on cleaning MgBench architecture so it is easier for users to add their custom workloads and queries to evaluate performance on supported systems.
|
||||
|
||||
This tutorial contains the following content:
|
||||
|
||||
- [How to add your custom workload](#how-to-add-your-custom-workload)
|
||||
- [How to run benchmarks on your custom workload](#how-to-run-benchmarks-on-your-custom-workload)
|
||||
- [How to configure benchmark run](#how-to-configure-benchmark-run)
|
||||
- [How to compare results](#how-to-compare-results)
|
||||
- [Customizing workload generator](#customizing-workload-generator)
|
||||
|
||||
|
||||
## How to add your custom workload
|
||||
|
||||
If you want to run your custom workload on supported systems (Currently, Memgraph and Neo4j), you can start by writing a simple Python class. The idea is to specify a simple class that contains your dataset generation queries, index generation queries and queries used for running a benchmark.
|
||||
|
||||
Here are 5 steps you need to do to specify your **workload**:
|
||||
|
||||
1. [Inherit the workload class](#1-inherit-the-workload-class)
|
||||
2. [Define a workload name](#2-define-the-workload-name)
|
||||
3. [Implement dataset generator method](#3-implement-dataset-generator-method)
|
||||
4. [Implement index generator method](#4-implement-the-index-generator)
|
||||
5. [Define the queries you want to benchmark](#4-define-the-queries-you-want-to-benchmark)
|
||||
|
||||
Here is the simplified version of [demo.py](https://github.com/memgraph/memgraph/blob/master/tests/mgbench/workloads/demo.py) example:
|
||||
|
||||
```python
|
||||
import random
|
||||
from workloads.base import Workload
|
||||
|
||||
class Demo(Workload):
|
||||
|
||||
NAME = "demo"
|
||||
|
||||
def indexes_generator(self):
|
||||
indexes = [
|
||||
("CREATE INDEX ON :NodeA(id);", {}),
|
||||
("CREATE INDEX ON :NodeB(id);", {}),
|
||||
]
|
||||
return indexes
|
||||
|
||||
def dataset_generator(self):
|
||||
|
||||
queries = []
|
||||
for i in range(0, 100):
|
||||
queries.append(("CREATE (:NodeA {id: $id});", {"id": i}))
|
||||
queries.append(("CREATE (:NodeB {id: $id});", {"id": i}))
|
||||
for i in range(0, 300):
|
||||
a = random.randint(0, 99)
|
||||
b = random.randint(0, 99)
|
||||
queries.append(
|
||||
(("MATCH(a:NodeA {id: $A_id}),(b:NodeB{id: $B_id}) CREATE (a)-[:EDGE]->(b)"), {"A_id": a, "B_id": b})
|
||||
)
|
||||
|
||||
return queries
|
||||
|
||||
def benchmark__test__get_nodes(self):
|
||||
return ("MATCH (n) RETURN n;", {})
|
||||
|
||||
def benchmark__test__get_node_by_id(self):
|
||||
return ("MATCH (n:NodeA{id: $id}) RETURN n;", {"id": random.randint(0, 99)})
|
||||
|
||||
|
||||
```
|
||||
|
||||
Let's break this script down into smaller important elements:
|
||||
|
||||
### 1. Inherit the `workload` class
|
||||
The `Demo` script class has a parent class `Workload`. Each custom workload should inherit from the base `Workload` class.
|
||||
|
||||
```python
|
||||
from workloads.base import Workload
|
||||
|
||||
class Demo(Workload):
|
||||
```
|
||||
|
||||
### 2. Define the workload name
|
||||
The class should specify the `NAME` property. This is used to describe what workload class you want to execute. When calling `benchmark.py`, this property will be used to differentiate different workloads.
|
||||
|
||||
```python
|
||||
NAME = "demo"
|
||||
```
|
||||
|
||||
### 3. Implement dataset generator method
|
||||
The class should implement the `dataset_generator()` method. The method generates a dataset that returns the ***list of tuples***. Each tuple contains a string of the Cypher query and dictionary that contains optional arguments, so the structure is following [(str, dict), (str, dict)...]. Let's take a look at how the example list could look like what it could method return:
|
||||
|
||||
```python
|
||||
queries = [
|
||||
("CREATE (:NodeA {id: 23});", {}),
|
||||
("CREATE (:NodeB {id: $id, foo: $property});", {"id" : 123, "property": "foo" }),
|
||||
...
|
||||
]
|
||||
```
|
||||
As you can see, you can pass just a Cypher query as a pure string without any values in the dictionary.
|
||||
|
||||
```python
|
||||
("CREATE (:NodeA {id: 23});", {}),
|
||||
```
|
||||
|
||||
Or you can specify parameters inside a dictionary. The variables next to `$` sign in the query string will be replaced by the appropriate values behind the key from the dictionary. In this case `$id` is replaced by `123` and `$property` is replaced by `foo`. The dictionary key names and variable names need to match.
|
||||
|
||||
```python
|
||||
("CREATE (:NodeB {id: $id, foo: $property});", {"id" : 123, "property": "foo" })
|
||||
```
|
||||
|
||||
Back to our `demo.py` example, in the `dataset_generator()` method, here you specify queries for generating a dataset. In the first for loop the queries for creating 100 nodes with the label `NodeA` and 100 nodes with the label `NodeB` are prepared. Each node has `id` between 0 and 99. In the second for loop, queries for connecting nodes randomly are generated. There is a total of 300 edges, each connected to random `NodeA` and `NodeB`.
|
||||
|
||||
```python
|
||||
def dataset_generator(self):
|
||||
|
||||
for i in range(0, 100):
|
||||
queries.append(("CREATE (:NodeA {id: $id});", {"id" : i}))
|
||||
queries.append(("CREATE (:NodeB {id: $id});", {"id" : i}))
|
||||
for i in range(0, 300):
|
||||
a = random.randint(0, 99)
|
||||
b = random.randint(0, 99)
|
||||
queries.append((("MATCH(a:NodeA {id: $A_id}),(b:NodeB{id: $B_id}) CREATE (a)-[:EDGE]->(b)"), {"A_id": a, "B_id" : b}))
|
||||
|
||||
return queries
|
||||
```
|
||||
|
||||
### 4. Implement the index generator method
|
||||
|
||||
The class should also implement the `indexes_generator()` method. This is implemented the same way as the `dataset_generator()` method, instead of queries for the dataset, `indexes_generator()` should return the list of indexes that will be used. The returning structure again is the list of tuples that contains query string and dictionary of parameters. Here is an example:
|
||||
|
||||
```python
|
||||
def indexes_generator(self):
|
||||
indexes = [
|
||||
("CREATE INDEX ON :NodeA(id);", {}),
|
||||
("CREATE INDEX ON :NodeB(id);", {}),
|
||||
]
|
||||
return indexes
|
||||
```
|
||||
|
||||
### 5. Define the queries you want to benchmark
|
||||
|
||||
Now that your dataset will be imported from dataset generator queries, you can specify what queries you wish to benchmark on the given dataset. Here are two queries that `demo.py` workload defines. They are written as Python methods that return a single tuple with query and dictionary, as in the data generator method.
|
||||
|
||||
```python
|
||||
def benchmark__test__get_nodes(self):
|
||||
return ("MATCH (n) RETURN n;", {})
|
||||
|
||||
def benchmark__test__get_node_by_id(self):
|
||||
return ("MATCH (n:NodeA{id: $id}) RETURN n;", {"id": random.randint(0, 99)})
|
||||
|
||||
```
|
||||
|
||||
The necessary details here are that each of the methods you wish to use in the benchmark test needs to start with `benchmark__` in the name, otherwise, it will be ignored. The complete method name has the following structure `benchmark__group__name`. The group can be used to execute specific tests, but more on that later.
|
||||
|
||||
From the workload setup, this is all you need to do. Next step is how to run your workload. If you wish to improve the workload generator, look at [customizing workload generator](#customizing-workload-generator).
|
||||
|
||||
## How to run benchmarks on your custom workload
|
||||
|
||||
When running benchmarks, duration, query arguments, number of workers, and database condition play an important role on the results of the benchmark. MgBench provides several options for the configuration of how the benchmark is executed. Let's start with the most straightforward run of the demo workload from the example above.
|
||||
|
||||
The main script that manages benchmark execution is `benchmark.py`.
|
||||
|
||||
To start the benchmark, you need to run the following command with your paths and options:
|
||||
|
||||
```python3 benchmark.py vendor-docker --vendor-name (memgraph-docker||neo4j-docker) benchmarks demo/*/*/* --export-results result.json --no-authorization```
|
||||
|
||||
To run this on memgraph, the command looks like this:
|
||||
|
||||
```python3 benchmark.py vendor-docker --vendor-name memgraph-docker benchmarks demo/*/*/* --export-results results.json --no-authorization```
|
||||
|
||||
## How to configure benchmark run
|
||||
|
||||
Hopefully, you should get logs from `benchmark.py` process managing the benchmark and execution from the command above. The script takes a lot of arguments. Some used in the run above are self-explanatory. But let's break down the most important ones:
|
||||
|
||||
- `NAME/VARIANT/GROUP/QUERY ` - The argument `demo/*/*/*` says to execute the workload named `demo`, and all of its variants, group's and queries. This flag is used for direct control of what workload you wish to execute. The `NAME` here is the name of the workload defined in the Workload class. `VARIANT` is an additional workload configuration, which will be explained a bit later. `GROUP` is defined in the query method name, and the `QUERY` is query name you wish to execute. If you want to execute a specific query from `demo.py`, it would look like this: `demo/*/test/get_nodes`. This will run `demo` workload on all `variants`, in `test` query group and query `get_nodes`.
|
||||
|
||||
- `--single-threaded-runtime-sec` - The question at hand is how many of each specific queries you wish to execute as a sample for a database benchmark. Each query can take a different time to execute, so fixating a number could yield some queries finishing in 1 second and others running for a minute. This flag defines the duration in seconds that will be used to approximate how many queries you wish to execute. The default value is 10 seconds, this means the `benchmark.py` will generate predetermined numbers of queries to approximate single treaded runtime of 10 seconds. Increasing this will yield a longer running test.
|
||||
Each specific query will get a different count that specifies how many queries will be generated. This can be inspected after the test. For example, for 10 seconds of single-threaded runtime, the queries from demo workload `get_node_by_id` got 64230 different queries, while `get_nodes` got 5061 because of different time complexity of queries.
|
||||
|
||||
- `--num-workers-for-benchmark` - The flag defines how many concurrent clients will open and query the database. With this flag, you can simulate different database users connecting to the database and executing queries. Each of the clients is independent and executes queries as fast as possible. They share a total pool of queries that were generated by the `--single-threaded-runtime-sec`. This means the total number of queries that need to be executed is shared between a specified number of workers.
|
||||
|
||||
- `--warm-up` - The warm-up flag can take a three different arguments, `cold`, `hot` and `vulcanic`. Cold is the default. There is no warm-up being executed, `hot` will execute some predefined queries before the benchmark, while `vulcanic` will run the whole workload first before taking measurements. Here is the implementation of [warm-up](https://github.com/memgraph/memgraph/blob/master/tests/mgbench/benchmark.py#L186)
|
||||
|
||||
|
||||
## How to compare results
|
||||
|
||||
|
||||
Once the benchmark has been run, the results are saved in a file specified by `--export-results` argument. You can use the results files and compare them against other vendor results via the `compare_results.py` script:
|
||||
|
||||
```python compare_results.py --compare path_to/run_1.json path_to/run_2.json --output run_1_vs_run_2.html --different-vendors```
|
||||
|
||||
The output is an HTML file with a visual representation of the performance differences between two compared vendors. The first passed summary JSON file is the reference point. Feel free to open an HTML file in any browser at hand.
|
||||
|
||||
## Customizing workload generator
|
||||
|
||||
### How to run the same workload on the different vendors
|
||||
|
||||
The base [Workload class](#1-inherit-the-workload-class) has benchmarking context information that contains all benchmark arguments used in this run. Some are mentioned above. The key argument here is the `--vendor-name`, which defines what database is being used in this benchmark.
|
||||
|
||||
During the creation of your workload, you can access the parent class property by using `self.benchmark_context.vendor_name`. For example, if you want to specify special index creation for each vendor, the `indexes_generator()` could look like this:
|
||||
|
||||
```python
|
||||
def indexes_generator(self):
|
||||
indexes = []
|
||||
if "neo4j" in self.benchmark_context.vendor_name:
|
||||
indexes.extend(
|
||||
[
|
||||
("CREATE INDEX FOR (n:NodeA) ON (n.id);", {}),
|
||||
("CREATE INDEX FOR (n:NodeB) ON (n.id);", {}),
|
||||
]
|
||||
)
|
||||
else:
|
||||
indexes.extend(
|
||||
[
|
||||
("CREATE INDEX ON :NodeA(id);", {}),
|
||||
("CREATE INDEX ON :NodeB(id);", {}),
|
||||
]
|
||||
)
|
||||
return indexes
|
||||
```
|
||||
|
||||
The same applies to the `dataset_generator()`. During the generation of the dataset, you can use special types of queries for different vendors.
|
||||
@@ -1,4 +1,4 @@
|
||||
# Copyright 2021 Memgraph Ltd.
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -33,7 +33,7 @@ def _log(color, *args):
|
||||
|
||||
|
||||
def log(msg):
|
||||
print(msg)
|
||||
print(str(msg))
|
||||
logger.info(msg=msg)
|
||||
|
||||
|
||||
@@ -60,3 +60,8 @@ def warning(*args):
|
||||
def error(*args):
|
||||
_log(COLOR_RED, *args)
|
||||
logger.critical(*args)
|
||||
|
||||
|
||||
def summary(*args):
|
||||
_log(COLOR_CYAN, *args)
|
||||
logger.info(*args)
|
||||
|
||||
@@ -13,6 +13,7 @@ import atexit
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import socket
|
||||
import subprocess
|
||||
import tempfile
|
||||
import threading
|
||||
@@ -20,12 +21,15 @@ import time
|
||||
from abc import ABC, abstractmethod
|
||||
from pathlib import Path
|
||||
|
||||
import log
|
||||
from benchmark_context import BenchmarkContext
|
||||
|
||||
DOCKER_NETWORK_NAME = "mgbench_network"
|
||||
|
||||
def _wait_for_server(port, delay=0.1):
|
||||
cmd = ["nc", "-z", "-w", "1", "127.0.0.1", str(port)]
|
||||
while subprocess.call(cmd) != 0:
|
||||
|
||||
def _wait_for_server_socket(port, ip="127.0.0.1", delay=0.1):
|
||||
s = socket.socket(socket.AF_INET, socket.SOCK_STREAM)
|
||||
while s.connect_ex((ip, int(port))) != 0:
|
||||
time.sleep(0.01)
|
||||
time.sleep(delay)
|
||||
|
||||
@@ -65,6 +69,27 @@ def _get_current_usage(pid):
|
||||
return rss / 1024
|
||||
|
||||
|
||||
def _setup_docker_benchmark_network(network_name):
|
||||
command = ["docker", "network", "ls", "--format", "{{.Name}}"]
|
||||
networks = subprocess.run(command, check=True, capture_output=True, text=True).stdout.split("\n")
|
||||
if network_name in networks:
|
||||
return
|
||||
else:
|
||||
command = ["docker", "network", "create", network_name]
|
||||
subprocess.run(command, check=True, capture_output=True, text=True)
|
||||
|
||||
|
||||
def _get_docker_container_ip(container_name):
|
||||
command = [
|
||||
"docker",
|
||||
"inspect",
|
||||
"--format",
|
||||
"{{range .NetworkSettings.Networks}}{{.IPAddress}}{{end}}",
|
||||
container_name,
|
||||
]
|
||||
return subprocess.run(command, check=True, capture_output=True, text=True).stdout.strip()
|
||||
|
||||
|
||||
class BaseClient(ABC):
|
||||
@abstractmethod
|
||||
def __init__(self, benchmark_context: BenchmarkContext):
|
||||
@@ -97,26 +122,56 @@ class BoltClient(BaseClient):
|
||||
queries=None,
|
||||
file_path=None,
|
||||
num_workers=1,
|
||||
max_retries: int = 50,
|
||||
max_retries: int = 10000,
|
||||
validation: bool = False,
|
||||
time_dependent_execution: int = 0,
|
||||
):
|
||||
check_db_query = Path(self._directory.name) / "check_db_query.json"
|
||||
with open(check_db_query, "w") as f:
|
||||
query = ["RETURN 0;", {}]
|
||||
json.dump(query, f)
|
||||
f.write("\n")
|
||||
|
||||
check_db_args = self._get_args(
|
||||
input=check_db_query,
|
||||
num_workers=1,
|
||||
max_retries=max_retries,
|
||||
queries_json=True,
|
||||
username=self._username,
|
||||
password=self._password,
|
||||
port=self._bolt_port,
|
||||
validation=False,
|
||||
time_dependent_execution=time_dependent_execution,
|
||||
)
|
||||
|
||||
while True:
|
||||
try:
|
||||
subprocess.run(check_db_args, capture_output=True, text=True, check=True)
|
||||
break
|
||||
except subprocess.CalledProcessError as e:
|
||||
log.log("Checking if database is up and running failed...")
|
||||
log.warning("Reported errors from client:")
|
||||
log.warning("Error: {}".format(e.stderr))
|
||||
log.warning("Database is not up yet, waiting 3 seconds...")
|
||||
time.sleep(3)
|
||||
|
||||
if (queries is None and file_path is None) or (queries is not None and file_path is not None):
|
||||
raise ValueError("Either queries or input_path must be specified!")
|
||||
|
||||
queries_json = False
|
||||
queries_and_args_json = False
|
||||
if queries is not None:
|
||||
queries_json = True
|
||||
file_path = os.path.join(self._directory.name, "queries.json")
|
||||
queries_and_args_json = True
|
||||
file_path = os.path.join(self._directory.name, "queries_and_args_json.json")
|
||||
with open(file_path, "w") as f:
|
||||
for query in queries:
|
||||
json.dump(query, f)
|
||||
f.write("\n")
|
||||
|
||||
args = self._get_args(
|
||||
input=file_path,
|
||||
num_workers=num_workers,
|
||||
max_retries=max_retries,
|
||||
queries_json=queries_json,
|
||||
queries_json=queries_and_args_json,
|
||||
username=self._username,
|
||||
password=self._password,
|
||||
port=self._bolt_port,
|
||||
@@ -131,12 +186,186 @@ class BoltClient(BaseClient):
|
||||
error = ret.stderr.decode("utf-8").strip().split("\n")
|
||||
data = ret.stdout.decode("utf-8").strip().split("\n")
|
||||
if error and error[0] != "":
|
||||
print("Reported errros from client")
|
||||
print(error)
|
||||
log.warning("Reported errors from client:")
|
||||
log.warning("There is a possibility that query from: {} is not executed properly".format(file_path))
|
||||
log.error(error)
|
||||
log.error("Results for this query or benchmark run are probably invalid!")
|
||||
data = [x for x in data if not x.startswith("[")]
|
||||
return list(map(json.loads, data))
|
||||
|
||||
|
||||
class BoltClientDocker(BaseClient):
|
||||
def __init__(self, benchmark_context: BenchmarkContext):
|
||||
self._client_binary = benchmark_context.client_binary
|
||||
self._directory = tempfile.TemporaryDirectory(dir=benchmark_context.temporary_directory)
|
||||
self._username = ""
|
||||
self._password = ""
|
||||
self._bolt_port = (
|
||||
benchmark_context.vendor_args["bolt-port"] if "bolt-port" in benchmark_context.vendor_args.keys() else 7687
|
||||
)
|
||||
self._container_name = "mgbench-bolt-client"
|
||||
self._target_db_container = (
|
||||
"memgraph_benchmark" if "memgraph" in benchmark_context.vendor_name else "neo4j_benchmark"
|
||||
)
|
||||
|
||||
def _remove_container(self):
|
||||
command = ["docker", "rm", "-f", self._container_name]
|
||||
self._run_command(command)
|
||||
|
||||
def _create_container(self, *args):
|
||||
command = [
|
||||
"docker",
|
||||
"create",
|
||||
"--name",
|
||||
self._container_name,
|
||||
"--network",
|
||||
DOCKER_NETWORK_NAME,
|
||||
"memgraph/mgbench-client",
|
||||
*args,
|
||||
]
|
||||
self._run_command(command)
|
||||
|
||||
def _get_logs(self):
|
||||
command = [
|
||||
"docker",
|
||||
"logs",
|
||||
self._container_name,
|
||||
]
|
||||
ret = self._run_command(command)
|
||||
return ret
|
||||
|
||||
def _get_args(self, **kwargs):
|
||||
return _convert_args_to_flags(**kwargs)
|
||||
|
||||
def execute(
|
||||
self,
|
||||
queries=None,
|
||||
file_path=None,
|
||||
num_workers=1,
|
||||
max_retries: int = 50,
|
||||
validation: bool = False,
|
||||
time_dependent_execution: int = 0,
|
||||
):
|
||||
if (queries is None and file_path is None) or (queries is not None and file_path is not None):
|
||||
raise ValueError("Either queries or input_path must be specified!")
|
||||
|
||||
self._remove_container()
|
||||
ip = _get_docker_container_ip(self._target_db_container)
|
||||
|
||||
# Perform a check to make sure the database is up and running
|
||||
args = self._get_args(
|
||||
address=ip,
|
||||
input="/bin/check.json",
|
||||
num_workers=1,
|
||||
max_retries=max_retries,
|
||||
queries_json=True,
|
||||
username=self._username,
|
||||
password=self._password,
|
||||
port=self._bolt_port,
|
||||
validation=False,
|
||||
time_dependent_execution=0,
|
||||
)
|
||||
|
||||
self._create_container(*args)
|
||||
|
||||
check_file = Path(self._directory.name) / "check.json"
|
||||
with open(check_file, "w") as f:
|
||||
query = ["RETURN 0;", {}]
|
||||
json.dump(query, f)
|
||||
f.write("\n")
|
||||
|
||||
command = [
|
||||
"docker",
|
||||
"cp",
|
||||
check_file.resolve().as_posix(),
|
||||
self._container_name + ":/bin/" + check_file.name,
|
||||
]
|
||||
self._run_command(command)
|
||||
|
||||
command = [
|
||||
"docker",
|
||||
"start",
|
||||
"-i",
|
||||
self._container_name,
|
||||
]
|
||||
while True:
|
||||
try:
|
||||
self._run_command(command)
|
||||
break
|
||||
except subprocess.CalledProcessError as e:
|
||||
log.log("Checking if database is up and running failed!")
|
||||
log.warning("Reported errors from client:")
|
||||
log.warning("Error: {}".format(e.stderr))
|
||||
log.warning("Database is not up yet, waiting 3 second")
|
||||
time.sleep(3)
|
||||
|
||||
self._remove_container()
|
||||
|
||||
queries_and_args_json = False
|
||||
if queries is not None:
|
||||
queries_and_args_json = True
|
||||
file_path = os.path.join(self._directory.name, "queries.json")
|
||||
with open(file_path, "w") as f:
|
||||
for query in queries:
|
||||
json.dump(query, f)
|
||||
f.write("\n")
|
||||
|
||||
self._remove_container()
|
||||
ip = _get_docker_container_ip(self._target_db_container)
|
||||
|
||||
# Query file JSON or Cypher
|
||||
file = Path(file_path)
|
||||
|
||||
args = self._get_args(
|
||||
address=ip,
|
||||
input="/bin/" + file.name,
|
||||
num_workers=num_workers,
|
||||
max_retries=max_retries,
|
||||
queries_json=queries_and_args_json,
|
||||
username=self._username,
|
||||
password=self._password,
|
||||
port=self._bolt_port,
|
||||
validation=validation,
|
||||
time_dependent_execution=time_dependent_execution,
|
||||
)
|
||||
|
||||
self._create_container(*args)
|
||||
|
||||
command = [
|
||||
"docker",
|
||||
"cp",
|
||||
file.resolve().as_posix(),
|
||||
self._container_name + ":/bin/" + file.name,
|
||||
]
|
||||
self._run_command(command)
|
||||
log.log("Starting query execution...")
|
||||
try:
|
||||
command = [
|
||||
"docker",
|
||||
"start",
|
||||
"-i",
|
||||
self._container_name,
|
||||
]
|
||||
self._run_command(command)
|
||||
except subprocess.CalledProcessError as e:
|
||||
log.warning("Reported errors from client:")
|
||||
log.warning("Error: {}".format(e.stderr))
|
||||
|
||||
ret = self._get_logs()
|
||||
error = ret.stderr.strip().split("\n")
|
||||
if error and error[0] != "":
|
||||
log.warning("There is a possibility that query from: {} is not executed properly".format(file_path))
|
||||
log.warning(*error)
|
||||
data = ret.stdout.strip().split("\n")
|
||||
data = [x for x in data if not x.startswith("[")]
|
||||
return list(map(json.loads, data))
|
||||
|
||||
def _run_command(self, command):
|
||||
ret = subprocess.run(command, capture_output=True, check=True, text=True)
|
||||
time.sleep(0.2)
|
||||
return ret
|
||||
|
||||
|
||||
class BaseRunner(ABC):
|
||||
subclasses = {}
|
||||
|
||||
@@ -159,15 +388,19 @@ class BaseRunner(ABC):
|
||||
self.benchmark_context = benchmark_context
|
||||
|
||||
@abstractmethod
|
||||
def start_benchmark(self):
|
||||
def start_db_init(self):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def start_preparation(self):
|
||||
def stop_db_init(self):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def stop(self):
|
||||
def start_db(self):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
def stop_db(self):
|
||||
pass
|
||||
|
||||
@abstractmethod
|
||||
@@ -186,11 +419,6 @@ class Memgraph(BaseRunner):
|
||||
self._performance_tracking = benchmark_context.performance_tracking
|
||||
self._directory = tempfile.TemporaryDirectory(dir=benchmark_context.temporary_directory)
|
||||
self._vendor_args = benchmark_context.vendor_args
|
||||
self._properties_on_edges = (
|
||||
self._vendor_args["no-properties-on-edges"]
|
||||
if "no-properties-on-edges" in self._vendor_args.keys()
|
||||
else False
|
||||
)
|
||||
self._bolt_port = self._vendor_args["bolt-port"] if "bolt-port" in self._vendor_args.keys() else 7687
|
||||
self._proc_mg = None
|
||||
self._stop_event = threading.Event()
|
||||
@@ -211,7 +439,9 @@ class Memgraph(BaseRunner):
|
||||
data_directory = os.path.join(self._directory.name, "memgraph")
|
||||
kwargs["bolt_port"] = self._bolt_port
|
||||
kwargs["data_directory"] = data_directory
|
||||
kwargs["storage_properties_on_edges"] = self._properties_on_edges
|
||||
kwargs["storage_properties_on_edges"] = True
|
||||
for key, value in self._vendor_args.items():
|
||||
kwargs[key] = value
|
||||
return _convert_args_to_flags(self._memgraph_binary, **kwargs)
|
||||
|
||||
def _start(self, **kwargs):
|
||||
@@ -223,7 +453,7 @@ class Memgraph(BaseRunner):
|
||||
if self._proc_mg.poll() is not None:
|
||||
self._proc_mg = None
|
||||
raise Exception("The database process died prematurely!")
|
||||
_wait_for_server(self._bolt_port)
|
||||
_wait_for_server_socket(self._bolt_port)
|
||||
ret = self._proc_mg.poll()
|
||||
assert ret is None, "The database process died prematurely " "({})!".format(ret)
|
||||
|
||||
@@ -236,21 +466,37 @@ class Memgraph(BaseRunner):
|
||||
self._proc_mg = None
|
||||
return ret, usage
|
||||
|
||||
def start_preparation(self, workload):
|
||||
def start_db_init(self, workload):
|
||||
if self._performance_tracking:
|
||||
p = threading.Thread(target=self.res_background_tracking, args=(self._rss, self._stop_event))
|
||||
self._stop_event.clear()
|
||||
self._rss.clear()
|
||||
p.start()
|
||||
self._start(storage_snapshot_on_exit=True)
|
||||
self._start(storage_snapshot_on_exit=True, **self._vendor_args)
|
||||
|
||||
def start_benchmark(self, workload):
|
||||
def stop_db_init(self, workload):
|
||||
if self._performance_tracking:
|
||||
self._stop_event.set()
|
||||
self.dump_rss(workload)
|
||||
ret, usage = self._cleanup()
|
||||
assert ret == 0, "The database process exited with a non-zero " "status ({})!".format(ret)
|
||||
return usage
|
||||
|
||||
def start_db(self, workload):
|
||||
if self._performance_tracking:
|
||||
p = threading.Thread(target=self.res_background_tracking, args=(self._rss, self._stop_event))
|
||||
self._stop_event.clear()
|
||||
self._rss.clear()
|
||||
p.start()
|
||||
self._start(storage_recover_on_startup=True)
|
||||
self._start(storage_recover_on_startup=True, **self._vendor_args)
|
||||
|
||||
def stop_db(self, workload):
|
||||
if self._performance_tracking:
|
||||
self._stop_event.set()
|
||||
self.dump_rss(workload)
|
||||
ret, usage = self._cleanup()
|
||||
assert ret == 0, "The database process exited with a non-zero " "status ({})!".format(ret)
|
||||
return usage
|
||||
|
||||
def clean_db(self):
|
||||
if self._proc_mg is not None:
|
||||
@@ -284,14 +530,6 @@ class Memgraph(BaseRunner):
|
||||
f.write("\n")
|
||||
f.close()
|
||||
|
||||
def stop(self, workload):
|
||||
if self._performance_tracking:
|
||||
self._stop_event.set()
|
||||
self.dump_rss(workload)
|
||||
ret, usage = self._cleanup()
|
||||
assert ret == 0, "The database process exited with a non-zero " "status ({})!".format(ret)
|
||||
return usage
|
||||
|
||||
def fetch_client(self) -> BoltClient:
|
||||
return BoltClient(benchmark_context=self.benchmark_context)
|
||||
|
||||
@@ -304,6 +542,14 @@ class Neo4j(BaseRunner):
|
||||
self._neo4j_config = self._neo4j_path / "conf" / "neo4j.conf"
|
||||
self._neo4j_pid = self._neo4j_path / "run" / "neo4j.pid"
|
||||
self._neo4j_admin = self._neo4j_path / "bin" / "neo4j-admin"
|
||||
self._neo4j_dump = (
|
||||
Path()
|
||||
/ ".cache"
|
||||
/ "datasets"
|
||||
/ self.benchmark_context.get_active_workload()
|
||||
/ self.benchmark_context.get_active_variant()
|
||||
/ "neo4j.dump"
|
||||
)
|
||||
self._performance_tracking = benchmark_context.performance_tracking
|
||||
self._vendor_args = benchmark_context.vendor_args
|
||||
self._stop_event = threading.Event()
|
||||
@@ -372,13 +618,13 @@ class Neo4j(BaseRunner):
|
||||
raise Exception("The database process is already running!")
|
||||
args = _convert_args_to_flags(self._neo4j_binary, "start", **kwargs)
|
||||
start_proc = subprocess.run(args, check=True)
|
||||
time.sleep(5)
|
||||
time.sleep(0.5)
|
||||
if self._neo4j_pid.exists():
|
||||
print("Neo4j started!")
|
||||
else:
|
||||
raise Exception("The database process died prematurely!")
|
||||
print("Run server check:")
|
||||
_wait_for_server(self._bolt_port)
|
||||
_wait_for_server_socket(self._bolt_port)
|
||||
|
||||
def _cleanup(self):
|
||||
if self._neo4j_pid.exists():
|
||||
@@ -391,7 +637,7 @@ class Neo4j(BaseRunner):
|
||||
else:
|
||||
return 0
|
||||
|
||||
def start_preparation(self, workload):
|
||||
def start_db_init(self, workload):
|
||||
if self._performance_tracking:
|
||||
p = threading.Thread(target=self.res_background_tracking, args=(self._rss, self._stop_event))
|
||||
self._stop_event.clear()
|
||||
@@ -404,18 +650,48 @@ class Neo4j(BaseRunner):
|
||||
if self._performance_tracking:
|
||||
self.get_memory_usage("start_" + workload)
|
||||
|
||||
def start_benchmark(self, workload):
|
||||
def stop_db_init(self, workload):
|
||||
if self._performance_tracking:
|
||||
self._stop_event.set()
|
||||
self.get_memory_usage("stop_" + workload)
|
||||
self.dump_rss(workload)
|
||||
ret, usage = self._cleanup()
|
||||
self.dump_db(path=self._neo4j_dump.parent)
|
||||
assert ret == 0, "The database process exited with a non-zero " "status ({})!".format(ret)
|
||||
return usage
|
||||
|
||||
def start_db(self, workload):
|
||||
if self._performance_tracking:
|
||||
p = threading.Thread(target=self.res_background_tracking, args=(self._rss, self._stop_event))
|
||||
self._stop_event.clear()
|
||||
self._rss.clear()
|
||||
p.start()
|
||||
|
||||
neo4j_dump = (
|
||||
Path()
|
||||
/ ".cache"
|
||||
/ "datasets"
|
||||
/ self.benchmark_context.get_active_workload()
|
||||
/ self.benchmark_context.get_active_variant()
|
||||
/ "neo4j.dump"
|
||||
)
|
||||
if neo4j_dump.exists():
|
||||
self.load_db_from_dump(path=neo4j_dump.parent)
|
||||
# Start DB
|
||||
self._start()
|
||||
|
||||
if self._performance_tracking:
|
||||
self.get_memory_usage("start_" + workload)
|
||||
|
||||
def stop_db(self, workload):
|
||||
if self._performance_tracking:
|
||||
self._stop_event.set()
|
||||
self.get_memory_usage("stop_" + workload)
|
||||
self.dump_rss(workload)
|
||||
ret, usage = self._cleanup()
|
||||
assert ret == 0, "The database process exited with a non-zero " "status ({})!".format(ret)
|
||||
return usage
|
||||
|
||||
def dump_db(self, path):
|
||||
print("Dumping the neo4j database...")
|
||||
if self._neo4j_pid.exists():
|
||||
@@ -426,7 +702,7 @@ class Neo4j(BaseRunner):
|
||||
self._neo4j_admin,
|
||||
"database",
|
||||
"dump",
|
||||
"--overwrite-destination=false",
|
||||
"--overwrite-destination=true",
|
||||
"--to-path",
|
||||
path,
|
||||
"neo4j",
|
||||
@@ -478,20 +754,10 @@ class Neo4j(BaseRunner):
|
||||
def is_stopped(self):
|
||||
pid_file = self._neo4j_path / "run" / "neo4j.pid"
|
||||
if pid_file.exists():
|
||||
|
||||
return False
|
||||
else:
|
||||
return True
|
||||
|
||||
def stop(self, workload):
|
||||
if self._performance_tracking:
|
||||
self._stop_event.set()
|
||||
self.get_memory_usage("stop_" + workload)
|
||||
self.dump_rss(workload)
|
||||
ret, usage = self._cleanup()
|
||||
assert ret == 0, "The database process exited with a non-zero " "status ({})!".format(ret)
|
||||
return usage
|
||||
|
||||
def dump_rss(self, workload):
|
||||
file_name = workload + "_rss"
|
||||
Path.mkdir(Path().cwd() / "neo4j_memory", exist_ok=True)
|
||||
@@ -521,3 +787,311 @@ class Neo4j(BaseRunner):
|
||||
|
||||
def fetch_client(self) -> BoltClient:
|
||||
return BoltClient(benchmark_context=self.benchmark_context)
|
||||
|
||||
|
||||
class MemgraphDocker(BaseRunner):
|
||||
def __init__(self, benchmark_context: BenchmarkContext):
|
||||
super().__init__(benchmark_context=benchmark_context)
|
||||
self._directory = tempfile.TemporaryDirectory(dir=benchmark_context.temporary_directory)
|
||||
self._vendor_args = benchmark_context.vendor_args
|
||||
self._bolt_port = self._vendor_args["bolt-port"] if "bolt-port" in self._vendor_args.keys() else "7687"
|
||||
self._container_name = "memgraph_benchmark"
|
||||
self._container_ip = None
|
||||
self._config_file = None
|
||||
_setup_docker_benchmark_network(network_name=DOCKER_NETWORK_NAME)
|
||||
|
||||
def _set_args(self, **kwargs):
|
||||
return _convert_args_to_flags(**kwargs)
|
||||
|
||||
def start_db_init(self, message):
|
||||
log.init("Starting database for import...")
|
||||
try:
|
||||
command = [
|
||||
"docker",
|
||||
"run",
|
||||
"--detach",
|
||||
"--network",
|
||||
DOCKER_NETWORK_NAME,
|
||||
"--name",
|
||||
self._container_name,
|
||||
"-it",
|
||||
"-p",
|
||||
self._bolt_port + ":" + self._bolt_port,
|
||||
"memgraph/memgraph:2.7.0",
|
||||
"--storage_wal_enabled=false",
|
||||
"--storage_recover_on_startup=true",
|
||||
"--storage_snapshot_interval_sec",
|
||||
"0",
|
||||
]
|
||||
command.extend(self._set_args(**self._vendor_args))
|
||||
ret = self._run_command(command)
|
||||
except subprocess.CalledProcessError as e:
|
||||
log.error("Failed to start Memgraph docker container.")
|
||||
log.error(
|
||||
"There is probably a database running on that port, please stop the running container and try again."
|
||||
)
|
||||
raise e
|
||||
|
||||
command = [
|
||||
"docker",
|
||||
"cp",
|
||||
self._container_name + ":/etc/memgraph/memgraph.conf",
|
||||
self._directory.name + "/memgraph.conf",
|
||||
]
|
||||
self._run_command(command)
|
||||
self._config_file = Path(self._directory.name + "/memgraph.conf")
|
||||
_wait_for_server_socket(self._bolt_port, delay=0.5)
|
||||
log.log("Database started.")
|
||||
|
||||
def stop_db_init(self, message):
|
||||
log.init("Stopping database...")
|
||||
usage = self._get_cpu_memory_usage()
|
||||
|
||||
# Stop to save the snapshot
|
||||
command = ["docker", "stop", self._container_name]
|
||||
self._run_command(command)
|
||||
|
||||
# Change config back to default
|
||||
argument = "--storage-snapshot-on-exit=false"
|
||||
self._replace_config_args(argument)
|
||||
command = [
|
||||
"docker",
|
||||
"cp",
|
||||
self._config_file.resolve(),
|
||||
self._container_name + ":/etc/memgraph/memgraph.conf",
|
||||
]
|
||||
self._run_command(command)
|
||||
log.log("Database stopped.")
|
||||
return usage
|
||||
|
||||
def start_db(self, message):
|
||||
log.init("Starting database for benchmark...")
|
||||
command = ["docker", "start", self._container_name]
|
||||
self._run_command(command)
|
||||
ip_address = _get_docker_container_ip(self._container_name)
|
||||
_wait_for_server_socket(self._bolt_port, delay=0.5)
|
||||
log.log("Database started.")
|
||||
|
||||
def stop_db(self, message):
|
||||
log.init("Stopping database...")
|
||||
usage = self._get_cpu_memory_usage()
|
||||
command = ["docker", "stop", self._container_name]
|
||||
self._run_command(command)
|
||||
log.log("Database stopped.")
|
||||
return usage
|
||||
|
||||
def clean_db(self):
|
||||
self.remove_container(self._container_name)
|
||||
|
||||
def fetch_client(self) -> BaseClient:
|
||||
return BoltClientDocker(benchmark_context=self.benchmark_context)
|
||||
|
||||
def remove_container(self, containerName):
|
||||
command = ["docker", "rm", "-f", containerName]
|
||||
self._run_command(command)
|
||||
|
||||
def _replace_config_args(self, argument):
|
||||
config_lines = []
|
||||
with self._config_file.open("r") as file:
|
||||
lines = file.readlines()
|
||||
file.close()
|
||||
key, value = argument.split("=")
|
||||
for line in lines:
|
||||
if line[0] == "#" or line.strip("\n") == "":
|
||||
config_lines.append(line)
|
||||
else:
|
||||
key_file, value_file = line.split("=")
|
||||
if key_file == key and value != value_file:
|
||||
line = argument + "\n"
|
||||
config_lines.append(line)
|
||||
|
||||
with self._config_file.open("w") as file:
|
||||
file.writelines(config_lines)
|
||||
file.close()
|
||||
|
||||
def _get_cpu_memory_usage(self):
|
||||
command = [
|
||||
"docker",
|
||||
"exec",
|
||||
"-it",
|
||||
self._container_name,
|
||||
"bash",
|
||||
"-c",
|
||||
"grep ^VmPeak /proc/1/status",
|
||||
]
|
||||
usage = {"cpu": 0, "memory": 0}
|
||||
ret = self._run_command(command)
|
||||
memory = ret.stdout.split()
|
||||
usage["memory"] = int(memory[1]) * 1024
|
||||
|
||||
command = [
|
||||
"docker",
|
||||
"exec",
|
||||
"-it",
|
||||
self._container_name,
|
||||
"bash",
|
||||
"-c",
|
||||
"cat /proc/1/stat",
|
||||
]
|
||||
stat = self._run_command(command).stdout.strip("\n")
|
||||
|
||||
command = [
|
||||
"docker",
|
||||
"exec",
|
||||
"-it",
|
||||
self._container_name,
|
||||
"bash",
|
||||
"-c",
|
||||
"getconf CLK_TCK",
|
||||
]
|
||||
CLK_TCK = int(self._run_command(command).stdout.strip("\n"))
|
||||
|
||||
cpu_time = sum(map(int, stat.split(")")[1].split()[11:15])) / CLK_TCK
|
||||
usage["cpu"] = cpu_time
|
||||
|
||||
return usage
|
||||
|
||||
def _run_command(self, command):
|
||||
ret = subprocess.run(command, check=True, capture_output=True, text=True)
|
||||
|
||||
time.sleep(0.2)
|
||||
return ret
|
||||
|
||||
|
||||
class Neo4jDocker(BaseRunner):
|
||||
def __init__(self, benchmark_context: BenchmarkContext):
|
||||
super().__init__(benchmark_context=benchmark_context)
|
||||
self._directory = tempfile.TemporaryDirectory(dir=benchmark_context.temporary_directory)
|
||||
self._vendor_args = benchmark_context.vendor_args
|
||||
self._bolt_port = self._vendor_args["bolt-port"] if "bolt-port" in self._vendor_args.keys() else "7687"
|
||||
self._container_name = "neo4j_benchmark"
|
||||
self._container_ip = None
|
||||
self._config_file = None
|
||||
_setup_docker_benchmark_network(DOCKER_NETWORK_NAME)
|
||||
|
||||
def _set_args(self, **kwargs):
|
||||
return _convert_args_to_flags(**kwargs)
|
||||
|
||||
def start_db_init(self, message):
|
||||
log.init("Starting database for initialization...")
|
||||
try:
|
||||
command = [
|
||||
"docker",
|
||||
"run",
|
||||
"--detach",
|
||||
"--network",
|
||||
DOCKER_NETWORK_NAME,
|
||||
"--name",
|
||||
self._container_name,
|
||||
"-it",
|
||||
"-p",
|
||||
self._bolt_port + ":" + self._bolt_port,
|
||||
"--env",
|
||||
"NEO4J_AUTH=none",
|
||||
"neo4j:5.6.0",
|
||||
]
|
||||
command.extend(self._set_args(**self._vendor_args))
|
||||
ret = self._run_command(command)
|
||||
except subprocess.CalledProcessError as e:
|
||||
log.error("There was an error starting the Neo4j container!")
|
||||
log.error(
|
||||
"There is probably a database running on that port, please stop the running container and try again."
|
||||
)
|
||||
raise e
|
||||
_wait_for_server_socket(self._bolt_port, delay=5)
|
||||
log.log("Database started.")
|
||||
|
||||
def stop_db_init(self, message):
|
||||
log.init("Stopping database...")
|
||||
usage = self._get_cpu_memory_usage()
|
||||
|
||||
command = ["docker", "stop", self._container_name]
|
||||
self._run_command(command)
|
||||
log.log("Database stopped.")
|
||||
|
||||
return usage
|
||||
|
||||
def start_db(self, message):
|
||||
log.init("Starting database...")
|
||||
command = ["docker", "start", self._container_name]
|
||||
self._run_command(command)
|
||||
_wait_for_server_socket(self._bolt_port, delay=5)
|
||||
log.log("Database started.")
|
||||
|
||||
def stop_db(self, message):
|
||||
log.init("Stopping database...")
|
||||
usage = self._get_cpu_memory_usage()
|
||||
|
||||
command = ["docker", "stop", self._container_name]
|
||||
self._run_command(command)
|
||||
log.log("Database stopped.")
|
||||
return usage
|
||||
|
||||
def clean_db(self):
|
||||
self.remove_container(self._container_name)
|
||||
|
||||
def fetch_client(self) -> BaseClient:
|
||||
return BoltClientDocker(benchmark_context=self.benchmark_context)
|
||||
|
||||
def remove_container(self, containerName):
|
||||
command = ["docker", "rm", "-f", containerName]
|
||||
self._run_command(command)
|
||||
|
||||
def _get_cpu_memory_usage(self):
|
||||
command = [
|
||||
"docker",
|
||||
"exec",
|
||||
"-it",
|
||||
self._container_name,
|
||||
"bash",
|
||||
"-c",
|
||||
"cat /var/lib/neo4j/run/neo4j.pid",
|
||||
]
|
||||
ret = self._run_command(command)
|
||||
pid = ret.stdout.split()[0]
|
||||
|
||||
command = [
|
||||
"docker",
|
||||
"exec",
|
||||
"-it",
|
||||
self._container_name,
|
||||
"bash",
|
||||
"-c",
|
||||
"grep ^VmPeak /proc/{}/status".format(pid),
|
||||
]
|
||||
usage = {"cpu": 0, "memory": 0}
|
||||
ret = self._run_command(command)
|
||||
memory = ret.stdout.split()
|
||||
usage["memory"] = int(memory[1]) * 1024
|
||||
|
||||
command = [
|
||||
"docker",
|
||||
"exec",
|
||||
"-it",
|
||||
self._container_name,
|
||||
"bash",
|
||||
"-c",
|
||||
"cat /proc/{}/stat".format(pid),
|
||||
]
|
||||
stat = self._run_command(command).stdout.strip("\n")
|
||||
|
||||
command = [
|
||||
"docker",
|
||||
"exec",
|
||||
"-it",
|
||||
self._container_name,
|
||||
"bash",
|
||||
"-c",
|
||||
"getconf CLK_TCK",
|
||||
]
|
||||
CLK_TCK = int(self._run_command(command).stdout.strip("\n"))
|
||||
|
||||
cpu_time = sum(map(int, stat.split(")")[1].split()[11:15])) / CLK_TCK
|
||||
usage["cpu"] = cpu_time
|
||||
|
||||
return usage
|
||||
|
||||
def _run_command(self, command):
|
||||
ret = subprocess.run(command, capture_output=True, check=True, text=True)
|
||||
time.sleep(0.2)
|
||||
return ret
|
||||
|
||||
34
tests/mgbench/setup.py
Normal file
34
tests/mgbench/setup.py
Normal file
@@ -0,0 +1,34 @@
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
import subprocess
|
||||
import sys
|
||||
from subprocess import CalledProcessError
|
||||
|
||||
import log
|
||||
from benchmark_context import BenchmarkContext
|
||||
|
||||
|
||||
def check_requirements(benchmark_context: BenchmarkContext):
|
||||
if "docker" in benchmark_context.vendor_name:
|
||||
log.info("Checking requirements ... ")
|
||||
command = ["docker", "info"]
|
||||
try:
|
||||
subprocess.run(command, check=True, capture_output=True, text=True)
|
||||
except CalledProcessError:
|
||||
log.error("Docker is not installed or not running")
|
||||
return False
|
||||
|
||||
if sys.version_info.major < 3 or sys.version_info.minor < 6:
|
||||
log.error("Python version 3.6 or higher is required")
|
||||
return False
|
||||
|
||||
return True
|
||||
@@ -1,3 +1,16 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
import argparse
|
||||
import copy
|
||||
import multiprocessing
|
||||
@@ -11,7 +24,6 @@ from workloads import base
|
||||
|
||||
|
||||
def pars_args():
|
||||
|
||||
parser = argparse.ArgumentParser(
|
||||
prog="Validator for individual query checking",
|
||||
description="""Validates that query is running, and validates output between different vendors""",
|
||||
@@ -90,7 +102,6 @@ def get_queries(gen, count):
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
|
||||
args = pars_args()
|
||||
|
||||
benchmark_context_db_1 = BenchmarkContext(
|
||||
@@ -120,19 +131,18 @@ if __name__ == "__main__":
|
||||
results_db_1 = {}
|
||||
|
||||
for workload, queries in workloads:
|
||||
|
||||
vendor_runner.clean_db()
|
||||
|
||||
generated_queries = workload.dataset_generator()
|
||||
if generated_queries:
|
||||
vendor_runner.start_preparation("import")
|
||||
vendor_runner.start_db_init("import")
|
||||
client.execute(queries=generated_queries, num_workers=benchmark_context_db_1.num_workers_for_import)
|
||||
vendor_runner.stop("import")
|
||||
vendor_runner.stop_db_init("import")
|
||||
else:
|
||||
workload.prepare(cache.cache_directory("datasets", workload.NAME, workload.get_variant()))
|
||||
imported = workload.custom_import()
|
||||
if not imported:
|
||||
vendor_runner.start_preparation("import")
|
||||
vendor_runner.start_db_init("import")
|
||||
print("Executing database cleanup and index setup...")
|
||||
client.execute(
|
||||
file_path=workload.get_index(), num_workers=benchmark_context_db_1.num_workers_for_import
|
||||
@@ -141,14 +151,14 @@ if __name__ == "__main__":
|
||||
ret = client.execute(
|
||||
file_path=workload.get_file(), num_workers=benchmark_context_db_1.num_workers_for_import
|
||||
)
|
||||
usage = vendor_runner.stop("import")
|
||||
usage = vendor_runner.stop_db_init("import")
|
||||
|
||||
for group in sorted(queries.keys()):
|
||||
for query, funcname in queries[group]:
|
||||
print("Running query:{}/{}/{}".format(group, query, funcname))
|
||||
func = getattr(workload, funcname)
|
||||
count = 1
|
||||
vendor_runner.start_benchmark("validation")
|
||||
vendor_runner.start_db("validation")
|
||||
try:
|
||||
ret = client.execute(queries=get_queries(func, count), num_workers=1, validation=True)[0]
|
||||
results_db_1[funcname] = ret["results"].items()
|
||||
@@ -157,7 +167,7 @@ if __name__ == "__main__":
|
||||
print(e)
|
||||
results_db_1[funcname] = "Query not executed properly"
|
||||
finally:
|
||||
usage = vendor_runner.stop("validation")
|
||||
usage = vendor_runner.stop_db("validation")
|
||||
print("Database used {:.3f} seconds of CPU time.".format(usage["cpu"]))
|
||||
print("Database peaked at {:.3f} MiB of memory.".format(usage["memory"] / 1024.0 / 1024.0))
|
||||
|
||||
@@ -182,19 +192,18 @@ if __name__ == "__main__":
|
||||
results_db_2 = {}
|
||||
|
||||
for workload, queries in workloads:
|
||||
|
||||
vendor_runner.clean_db()
|
||||
|
||||
generated_queries = workload.dataset_generator()
|
||||
if generated_queries:
|
||||
vendor_runner.start_preparation("import")
|
||||
vendor_runner.start_db_init("import")
|
||||
client.execute(queries=generated_queries, num_workers=benchmark_context_db_2.num_workers_for_import)
|
||||
vendor_runner.stop("import")
|
||||
else:
|
||||
workload.prepare(cache.cache_directory("datasets", workload.NAME, workload.get_variant()))
|
||||
imported = workload.custom_import()
|
||||
if not imported:
|
||||
vendor_runner.start_preparation("import")
|
||||
vendor_runner.start_db_init("import")
|
||||
print("Executing database cleanup and index setup...")
|
||||
client.execute(
|
||||
file_path=workload.get_index(), num_workers=benchmark_context_db_2.num_workers_for_import
|
||||
@@ -203,14 +212,14 @@ if __name__ == "__main__":
|
||||
ret = client.execute(
|
||||
file_path=workload.get_file(), num_workers=benchmark_context_db_2.num_workers_for_import
|
||||
)
|
||||
usage = vendor_runner.stop("import")
|
||||
usage = vendor_runner.stop_db_init("import")
|
||||
|
||||
for group in sorted(queries.keys()):
|
||||
for query, funcname in queries[group]:
|
||||
print("Running query:{}/{}/{}".format(group, query, funcname))
|
||||
func = getattr(workload, funcname)
|
||||
count = 1
|
||||
vendor_runner.start_benchmark("validation")
|
||||
vendor_runner.start_db("validation")
|
||||
try:
|
||||
ret = client.execute(queries=get_queries(func, count), num_workers=1, validation=True)[0]
|
||||
results_db_2[funcname] = ret["results"].items()
|
||||
@@ -219,7 +228,7 @@ if __name__ == "__main__":
|
||||
print(e)
|
||||
results_db_2[funcname] = "Query not executed properly"
|
||||
finally:
|
||||
usage = vendor_runner.stop("validation")
|
||||
usage = vendor_runner.stop_db("validation")
|
||||
print("Database used {:.3f} seconds of CPU time.".format(usage["cpu"]))
|
||||
print("Database peaked at {:.3f} MiB of memory.".format(usage["memory"] / 1024.0 / 1024.0))
|
||||
|
||||
|
||||
@@ -1,3 +1,14 @@
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
modules = Path(__file__).resolve().parent.glob("*.py")
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
# Copyright 2022 Memgraph Ltd.
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -19,7 +19,6 @@ from benchmark_context import BenchmarkContext
|
||||
# Base dataset class used as a template to create each individual dataset. All
|
||||
# common logic is handled here.
|
||||
class Workload(ABC):
|
||||
|
||||
# Name of the workload/dataset.
|
||||
NAME = ""
|
||||
# List of all variants of the workload/dataset that exist.
|
||||
@@ -48,6 +47,7 @@ class Workload(ABC):
|
||||
def __init_subclass__(cls) -> None:
|
||||
name_prerequisite = "NAME" in cls.__dict__
|
||||
generator_prerequisite = "dataset_generator" in cls.__dict__
|
||||
generator_indexes_prerequisite = "indexes_generator" in cls.__dict__
|
||||
custom_import_prerequisite = "custom_import" in cls.__dict__
|
||||
basic_import_prerequisite = ("LOCAL_FILE" in cls.__dict__ or "URL_FILE" in cls.__dict__) and (
|
||||
"LOCAL_INDEX_FILE" in cls.__dict__ or "URL_INDEX_FILE" in cls.__dict__
|
||||
@@ -55,21 +55,20 @@ class Workload(ABC):
|
||||
|
||||
if not name_prerequisite:
|
||||
raise ValueError(
|
||||
"""Can't define a workload class {} without NAME property:
|
||||
NAME = "dataset name"
|
||||
Name property defines the workload you want to execute, for example: "demo/*/*/*"
|
||||
""".format(
|
||||
"""
|
||||
Can't define a workload class {} without NAME property: NAME = "dataset name"
|
||||
Name property defines the workload you want to execute, for example: "demo/*/*/*"
|
||||
""".format(
|
||||
cls.__name__
|
||||
)
|
||||
)
|
||||
|
||||
# Check workload is in generator or dataset mode during interpretation (not both), not runtime
|
||||
if generator_prerequisite and (custom_import_prerequisite or basic_import_prerequisite):
|
||||
raise ValueError(
|
||||
"""
|
||||
The workload class {} cannot have defined dataset import and generate dataset at
|
||||
the same time.
|
||||
""".format(
|
||||
The workload class {} cannot have defined dataset import and generate dataset at
|
||||
the same time.
|
||||
""".format(
|
||||
cls.__name__
|
||||
)
|
||||
)
|
||||
@@ -77,12 +76,15 @@ class Workload(ABC):
|
||||
if not generator_prerequisite and (not custom_import_prerequisite and not basic_import_prerequisite):
|
||||
raise ValueError(
|
||||
"""
|
||||
The workload class {} need to have defined dataset import or dataset generator
|
||||
""".format(
|
||||
The workload class {} need to have defined dataset import or dataset generator
|
||||
""".format(
|
||||
cls.__name__
|
||||
)
|
||||
)
|
||||
|
||||
if generator_prerequisite and not generator_indexes_prerequisite:
|
||||
raise ValueError("The workload class {} need to define indexes_generator for generating a dataset. ")
|
||||
|
||||
return super().__init_subclass__()
|
||||
|
||||
def __init__(self, variant: str = None, benchmark_context: BenchmarkContext = None):
|
||||
|
||||
@@ -1,28 +1,56 @@
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
import random
|
||||
|
||||
from workloads.base import Workload
|
||||
|
||||
|
||||
class Demo(Workload):
|
||||
|
||||
NAME = "demo"
|
||||
|
||||
def indexes_generator(self):
|
||||
indexes = []
|
||||
if "neo4j" in self.benchmark_context.vendor_name:
|
||||
indexes.extend(
|
||||
[
|
||||
("CREATE INDEX FOR (n:NodeA) ON (n.id);", {}),
|
||||
("CREATE INDEX FOR (n:NodeB) ON (n.id);", {}),
|
||||
]
|
||||
)
|
||||
else:
|
||||
indexes.extend(
|
||||
[
|
||||
("CREATE INDEX ON :NodeA(id);", {}),
|
||||
("CREATE INDEX ON :NodeB(id);", {}),
|
||||
]
|
||||
)
|
||||
return indexes
|
||||
|
||||
def dataset_generator(self):
|
||||
|
||||
queries = [("MATCH (n) DETACH DELETE n;", {})]
|
||||
for i in range(0, 100):
|
||||
queries.append(("CREATE (:NodeA{{ id:{}}});".format(i), {}))
|
||||
queries.append(("CREATE (:NodeB{{ id:{}}});".format(i), {}))
|
||||
|
||||
for i in range(0, 100):
|
||||
a = random.randint(0, 99)
|
||||
b = random.randint(0, 99)
|
||||
queries.append(("MATCH(a:NodeA{{ id: {}}}),(b:NodeB{{id: {}}}) CREATE (a)-[:EDGE]->(b)".format(a, b), {}))
|
||||
queries = []
|
||||
for i in range(0, 10000):
|
||||
queries.append(("CREATE (:NodeA {id: $id});", {"id": i}))
|
||||
queries.append(("CREATE (:NodeB {id: $id});", {"id": i}))
|
||||
for i in range(0, 50000):
|
||||
a = random.randint(0, 9999)
|
||||
b = random.randint(0, 9999)
|
||||
queries.append(
|
||||
(("MATCH(a:NodeA {id: $A_id}),(b:NodeB{id: $B_id}) CREATE (a)-[:EDGE]->(b)"), {"A_id": a, "B_id": b})
|
||||
)
|
||||
|
||||
return queries
|
||||
|
||||
def benchmark__test__sample_query1(self):
|
||||
return ("MATCH (n) RETURN n", {})
|
||||
def benchmark__test__get_nodes(self):
|
||||
return ("MATCH (n) RETURN n;", {})
|
||||
|
||||
def benchmark__test__sample_query2(self):
|
||||
return ("MATCH (n) RETURN n", {})
|
||||
def benchmark__test__get_node_by_id(self):
|
||||
return ("MATCH (n:NodeA{id: $id}) RETURN n;", {"id": random.randint(0, 9999)})
|
||||
|
||||
@@ -1,3 +1,15 @@
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
# --- DISCLAIMER: This is NOT an official implementation of an LDBC Benchmark. ---
|
||||
import csv
|
||||
import subprocess
|
||||
from collections import defaultdict
|
||||
@@ -21,7 +33,6 @@ class ImporterLDBCBI:
|
||||
self._csv_dict = csv_dict
|
||||
|
||||
def execute_import(self):
|
||||
|
||||
vendor_runner = BaseRunner.create(
|
||||
benchmark_context=self._benchmark_context,
|
||||
)
|
||||
@@ -30,8 +41,8 @@ class ImporterLDBCBI:
|
||||
if self._benchmark_context.vendor_name == "neo4j":
|
||||
data_dir = Path() / ".cache" / "datasets" / self._dataset_name / self._variant / "data_neo4j"
|
||||
data_dir.mkdir(parents=True, exist_ok=True)
|
||||
dir_name = self._csv_dict[self._variant].split("/")[-1:][0].removesuffix(".tar.zst")
|
||||
if (data_dir / dir_name).exists():
|
||||
dir_name = self._csv_dict[self._variant].split("/")[-1:][0].replace(".tar.zst", "")
|
||||
if (data_dir / dir_name).exists() and any((data_dir / dir_name).iterdir()):
|
||||
print("Files downloaded")
|
||||
data_dir = data_dir / dir_name
|
||||
else:
|
||||
@@ -42,7 +53,7 @@ class ImporterLDBCBI:
|
||||
|
||||
headers_dir = Path() / ".cache" / "datasets" / self._dataset_name / self._variant / "headers_neo4j"
|
||||
headers_dir.mkdir(parents=True, exist_ok=True)
|
||||
headers = HEADERS_URL.split("/")[-1:][0].removesuffix(".tar.gz")
|
||||
headers = HEADERS_URL.split("/")[-1:][0].replace(".tar.gz", "")
|
||||
if (headers_dir / headers).exists():
|
||||
print("Header files downloaded.")
|
||||
else:
|
||||
@@ -204,10 +215,10 @@ class ImporterLDBCBI:
|
||||
check=True,
|
||||
)
|
||||
|
||||
vendor_runner.start_preparation("Index preparation")
|
||||
vendor_runner.start_db_init("Index preparation")
|
||||
print("Executing database index setup")
|
||||
client.execute(file_path=self._index_file, num_workers=1)
|
||||
vendor_runner.stop("Stop index preparation")
|
||||
vendor_runner.stop_db_init("Stop index preparation")
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
|
||||
@@ -1,3 +1,15 @@
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
# --- DISCLAIMER: This is NOT an official implementation of an LDBC Benchmark. ---
|
||||
import csv
|
||||
import subprocess
|
||||
from pathlib import Path
|
||||
@@ -53,7 +65,6 @@ class ImporterLDBCInteractive:
|
||||
self._csv_dict = csv_dict
|
||||
|
||||
def execute_import(self):
|
||||
|
||||
vendor_runner = BaseRunner.create(
|
||||
benchmark_context=self._benchmark_context,
|
||||
)
|
||||
@@ -63,8 +74,8 @@ class ImporterLDBCInteractive:
|
||||
print("Runnning Neo4j import")
|
||||
dump_dir = Path() / ".cache" / "datasets" / self._dataset_name / self._variant / "dump"
|
||||
dump_dir.mkdir(parents=True, exist_ok=True)
|
||||
dir_name = self._csv_dict[self._variant].split("/")[-1:][0].removesuffix(".tar.zst")
|
||||
if (dump_dir / dir_name).exists():
|
||||
dir_name = self._csv_dict[self._variant].split("/")[-1:][0].replace(".tar.zst", "")
|
||||
if (dump_dir / dir_name).exists() and any((dump_dir / dir_name).iterdir()):
|
||||
print("Files downloaded")
|
||||
dump_dir = dump_dir / dir_name
|
||||
else:
|
||||
@@ -153,10 +164,10 @@ class ImporterLDBCInteractive:
|
||||
check=True,
|
||||
)
|
||||
|
||||
vendor_runner.start_preparation("Index preparation")
|
||||
vendor_runner.start_db_init("Index preparation")
|
||||
print("Executing database index setup")
|
||||
client.execute(file_path=self._index_file, num_workers=1)
|
||||
vendor_runner.stop("Stop index preparation")
|
||||
vendor_runner.stop_db_init("Stop index preparation")
|
||||
|
||||
return True
|
||||
else:
|
||||
|
||||
@@ -1,5 +1,17 @@
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
from pathlib import Path
|
||||
|
||||
import log
|
||||
from benchmark_context import BenchmarkContext
|
||||
from runners import BaseRunner
|
||||
|
||||
@@ -16,26 +28,22 @@ class ImporterPokec:
|
||||
|
||||
def execute_import(self):
|
||||
if self._benchmark_context.vendor_name == "neo4j":
|
||||
|
||||
neo4j_dump = Path() / ".cache" / "datasets" / self._dataset_name / self._variant / "neo4j.dump"
|
||||
vendor_runner = BaseRunner.create(
|
||||
benchmark_context=self._benchmark_context,
|
||||
)
|
||||
client = vendor_runner.fetch_client()
|
||||
vendor_runner.clean_db()
|
||||
vendor_runner.start_preparation("preparation")
|
||||
print("Executing database cleanup and index setup...")
|
||||
client.execute(file_path=self._index_file, num_workers=1)
|
||||
vendor_runner.stop("preparation")
|
||||
neo4j_dump = Path() / ".cache" / "datasets" / self._dataset_name / self._variant / "neo4j.dump"
|
||||
if neo4j_dump.exists():
|
||||
log.log("Loading database from existing dump...")
|
||||
vendor_runner.load_db_from_dump(path=neo4j_dump.parent)
|
||||
else:
|
||||
vendor_runner.start_preparation("import")
|
||||
client = vendor_runner.fetch_client()
|
||||
vendor_runner.start_db_init("import")
|
||||
print("Executing database index setup...")
|
||||
client.execute(file_path=self._index_file, num_workers=1)
|
||||
print("Importing dataset...")
|
||||
client.execute(file_path=self._dataset_file, num_workers=self._benchmark_context.num_workers_for_import)
|
||||
vendor_runner.stop("import")
|
||||
vendor_runner.dump_db(path=neo4j_dump.parent)
|
||||
|
||||
vendor_runner.stop_db_init("import")
|
||||
return True
|
||||
else:
|
||||
return False
|
||||
|
||||
@@ -1,3 +1,15 @@
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
# --- DISCLAIMER: This is NOT an official implementation of an LDBC Benchmark. ---
|
||||
import inspect
|
||||
import random
|
||||
from pathlib import Path
|
||||
@@ -63,7 +75,7 @@ class LDBC_BI(Workload):
|
||||
print("Downloading files")
|
||||
downloaded_file = helpers.download_file(self.QUERY_PARAMETERS[self._variant], parameters.parent.absolute())
|
||||
print("Unpacking the file..." + downloaded_file)
|
||||
parameters = helpers.unpack_zip(Path(downloaded_file))
|
||||
helpers.unpack_zip(Path(downloaded_file))
|
||||
return parameters / ("parameters-" + self._variant)
|
||||
|
||||
def _get_query_parameters(self) -> dict:
|
||||
@@ -71,7 +83,7 @@ class LDBC_BI(Workload):
|
||||
parameters = {}
|
||||
for file in self._parameters_dir.glob("bi-*.csv"):
|
||||
file_name_query_id = file.name.split("-")[1][0:-4]
|
||||
func_name_id = func_name.split("_")[-1]
|
||||
func_name_id = func_name.split("_")[-2]
|
||||
if file_name_query_id == func_name_id or file_name_query_id == func_name_id + "a":
|
||||
with file.open("r") as input:
|
||||
lines = input.readlines()
|
||||
@@ -103,7 +115,6 @@ class LDBC_BI(Workload):
|
||||
self._parameters_dir = self._prepare_parameters_directory()
|
||||
|
||||
def benchmark__bi__query_1_analytical(self):
|
||||
|
||||
memgraph = (
|
||||
"""
|
||||
MATCH (message:Message)
|
||||
@@ -197,7 +208,6 @@ class LDBC_BI(Workload):
|
||||
return neo4j
|
||||
|
||||
def benchmark__bi__query_2_analytical(self):
|
||||
|
||||
memgraph = (
|
||||
"""
|
||||
MATCH (tag:Tag)-[:HAS_TYPE]->(:TagClass {name: $tagClass})
|
||||
@@ -327,7 +337,6 @@ class LDBC_BI(Workload):
|
||||
)
|
||||
|
||||
def benchmark__bi__query_7_analytical(self):
|
||||
|
||||
memgraph = (
|
||||
"""
|
||||
MATCH
|
||||
@@ -622,59 +631,7 @@ class LDBC_BI(Workload):
|
||||
self._get_query_parameters(),
|
||||
)
|
||||
|
||||
def benchmark__bi__query_17_analytical(self):
|
||||
|
||||
memgraph = (
|
||||
"""
|
||||
MATCH
|
||||
(tag:Tag {name: $tag}),
|
||||
(person1:Person)<-[:HAS_CREATOR]-(message1:Message)-[:REPLY_OF*0..]->(post1:Post)<-[:CONTAINER_OF]-(forum1:Forum),
|
||||
(message1)-[:HAS_TAG]->(tag),
|
||||
(forum1)<-[:HAS_MEMBER]->(person2:Person)<-[:HAS_CREATOR]-(comment:Comment)-[:HAS_TAG]->(tag),
|
||||
(forum1)<-[:HAS_MEMBER]->(person3:Person)<-[:HAS_CREATOR]-(message2:Message),
|
||||
(comment)-[:REPLY_OF]->(message2)-[:REPLY_OF*0..]->(post2:Post)<-[:CONTAINER_OF]-(forum2:Forum)
|
||||
MATCH (comment)-[:HAS_TAG]->(tag)
|
||||
MATCH (message2)-[:HAS_TAG]->(tag)
|
||||
OPTIONAL MATCH (forum2)-[:HAS_MEMBER]->(person1)
|
||||
WHERE forum1 <> forum2 AND message2.creationDate > message1.creationDate + duration({hours: $delta}) AND person1 IS NULL
|
||||
RETURN person1, count(DISTINCT message2) AS messageCount
|
||||
ORDER BY messageCount DESC, person1.id ASC
|
||||
LIMIT 10
|
||||
""".replace(
|
||||
"\n", ""
|
||||
),
|
||||
self._get_query_parameters(),
|
||||
)
|
||||
|
||||
neo4j = (
|
||||
"""
|
||||
MATCH
|
||||
(tag:Tag {name: $tag}),
|
||||
(person1:Person)<-[:HAS_CREATOR]-(message1:Message)-[:REPLY_OF*0..]->(post1:Post)<-[:CONTAINER_OF]-(forum1:Forum),
|
||||
(message1)-[:HAS_TAG]->(tag),
|
||||
(forum1)<-[:HAS_MEMBER]->(person2:Person)<-[:HAS_CREATOR]-(comment:Comment)-[:HAS_TAG]->(tag),
|
||||
(forum1)<-[:HAS_MEMBER]->(person3:Person)<-[:HAS_CREATOR]-(message2:Message),
|
||||
(comment)-[:REPLY_OF]->(message2)-[:REPLY_OF*0..]->(post2:Post)<-[:CONTAINER_OF]-(forum2:Forum)
|
||||
MATCH (comment)-[:HAS_TAG]->(tag)
|
||||
MATCH (message2)-[:HAS_TAG]->(tag)
|
||||
WHERE forum1 <> forum2
|
||||
AND message2.creationDate > message1.creationDate + duration({hours: $delta})
|
||||
AND NOT (forum2)-[:HAS_MEMBER]->(person1)
|
||||
RETURN person1, count(DISTINCT message2) AS messageCount
|
||||
ORDER BY messageCount DESC, person1.id ASC
|
||||
LIMIT 10
|
||||
""".replace(
|
||||
"\n", ""
|
||||
),
|
||||
self._get_query_parameters(),
|
||||
)
|
||||
if self._vendor == "memgraph":
|
||||
return memgraph
|
||||
else:
|
||||
return neo4j
|
||||
|
||||
def benchmark__bi__query_18_analytical(self):
|
||||
|
||||
memgraph = (
|
||||
"""
|
||||
MATCH (tag:Tag {name: $tag})<-[:HAS_INTEREST]-(person1:Person)-[:KNOWS]-(mutualFriend:Person)-[:KNOWS]-(person2:Person)-[:HAS_INTEREST]->(tag)
|
||||
|
||||
@@ -1,3 +1,15 @@
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
# --- DISCLAIMER: This is NOT an official implementation of an LDBC Benchmark. ---
|
||||
import inspect
|
||||
import random
|
||||
from datetime import datetime
|
||||
@@ -10,7 +22,6 @@ from workloads.importers.importer_ldbc_interactive import *
|
||||
|
||||
|
||||
class LDBC_Interactive(Workload):
|
||||
|
||||
NAME = "ldbc_interactive"
|
||||
VARIANTS = ["sf0.1", "sf1", "sf3", "sf10"]
|
||||
DEFAULT_VARIANT = "sf1"
|
||||
@@ -31,7 +42,7 @@ class LDBC_Interactive(Workload):
|
||||
SIZES = {
|
||||
"sf0.1": {"vertices": 327588, "edges": 1477965},
|
||||
"sf1": {"vertices": 3181724, "edges": 17256038},
|
||||
"sf3": {"vertices": 1, "edges": 1},
|
||||
"sf3": {"vertices": 9281922, "edges": 52695735},
|
||||
"sf10": {"vertices": 1, "edges": 1},
|
||||
}
|
||||
|
||||
@@ -44,8 +55,8 @@ class LDBC_Interactive(Workload):
|
||||
|
||||
QUERY_PARAMETERS = {
|
||||
"sf0.1": "https://repository.surfsara.nl/datasets/cwi/snb/files/substitution_parameters/substitution_parameters-sf0.1.tar.zst",
|
||||
"sf1": "https://repository.surfsara.nl/datasets/cwi/snb/files/substitution_parameters/substitution_parameters-sf0.1.tar.zst",
|
||||
"sf3": "https://repository.surfsara.nl/datasets/cwi/snb/files/substitution_parameters/substitution_parameters-sf0.1.tar.zst",
|
||||
"sf1": "https://repository.surfsara.nl/datasets/cwi/snb/files/substitution_parameters/substitution_parameters-sf1.tar.zst",
|
||||
"sf3": "https://repository.surfsara.nl/datasets/cwi/snb/files/substitution_parameters/substitution_parameters-sf3.tar.zst",
|
||||
}
|
||||
|
||||
def custom_import(self) -> bool:
|
||||
@@ -61,7 +72,7 @@ class LDBC_Interactive(Workload):
|
||||
def _prepare_parameters_directory(self):
|
||||
parameters = Path() / ".cache" / "datasets" / self.NAME / self._variant / "parameters"
|
||||
parameters.mkdir(parents=True, exist_ok=True)
|
||||
dir_name = self.QUERY_PARAMETERS[self._variant].split("/")[-1:][0].removesuffix(".tar.zst")
|
||||
dir_name = self.QUERY_PARAMETERS[self._variant].split("/")[-1:][0].replace(".tar.zst", "")
|
||||
if (parameters / dir_name).exists():
|
||||
print("Files downloaded:")
|
||||
parameters = parameters / dir_name
|
||||
@@ -230,7 +241,6 @@ class LDBC_Interactive(Workload):
|
||||
)
|
||||
|
||||
def benchmark__interactive__complex_query_3_analytical(self):
|
||||
|
||||
memgraph = (
|
||||
"""
|
||||
MATCH (countryX:Country {name: $countryXName }),
|
||||
@@ -327,8 +337,9 @@ class LDBC_Interactive(Workload):
|
||||
RETURN tag.name AS tagName, postCount
|
||||
ORDER BY postCount DESC, tagName ASC
|
||||
LIMIT 10
|
||||
|
||||
""",
|
||||
""".replace(
|
||||
"\n", ""
|
||||
),
|
||||
self._get_query_parameters(),
|
||||
)
|
||||
|
||||
@@ -351,8 +362,9 @@ class LDBC_Interactive(Workload):
|
||||
RETURN tag.name AS tagName, postCount
|
||||
ORDER BY postCount DESC, tagName ASC
|
||||
LIMIT 10
|
||||
|
||||
""",
|
||||
""".replace(
|
||||
"\n", ""
|
||||
),
|
||||
self._get_query_parameters(),
|
||||
)
|
||||
|
||||
@@ -528,72 +540,6 @@ class LDBC_Interactive(Workload):
|
||||
self._get_query_parameters(),
|
||||
)
|
||||
|
||||
def benchmark__interactive__complex_query_10_analytical(self):
|
||||
memgraph = (
|
||||
"""
|
||||
MATCH (person:Person {id: $personId})-[:KNOWS*2..2]-(friend),
|
||||
(friend)-[:IS_LOCATED_IN]->(city:City)
|
||||
WHERE NOT friend=person AND
|
||||
NOT (friend)-[:KNOWS]-(person)
|
||||
WITH person, city, friend, datetime({epochMillis: friend.birthday}) as birthday
|
||||
WHERE (birthday.month=$month AND birthday.day>=21) OR
|
||||
(birthday.month=($month%12)+1 AND birthday.day<22)
|
||||
WITH DISTINCT friend, city, person
|
||||
OPTIONAL MATCH (friend)<-[:HAS_CREATOR]-(post:Post)
|
||||
WITH friend, city, collect(post) AS posts, person
|
||||
WITH friend,
|
||||
city,
|
||||
size(posts) AS postCount,
|
||||
size([p IN posts WHERE (p)-[:HAS_TAG]->()<-[:HAS_INTEREST]-(person)]) AS commonPostCount
|
||||
RETURN friend.id AS personId,
|
||||
friend.firstName AS personFirstName,
|
||||
friend.lastName AS personLastName,
|
||||
commonPostCount - (postCount - commonPostCount) AS commonInterestScore,
|
||||
friend.gender AS personGender,
|
||||
city.name AS personCityName
|
||||
ORDER BY commonInterestScore DESC, personId ASC
|
||||
LIMIT 10
|
||||
""".replace(
|
||||
"\n", ""
|
||||
),
|
||||
self._get_query_parameters(),
|
||||
)
|
||||
|
||||
neo4j = (
|
||||
"""
|
||||
MATCH (person:Person {id: $personId})-[:KNOWS*2..2]-(friend),
|
||||
(friend)-[:IS_LOCATED_IN]->(city:City)
|
||||
WHERE NOT friend=person AND
|
||||
NOT (friend)-[:KNOWS]-(person)
|
||||
WITH person, city, friend, datetime({epochMillis: friend.birthday}) as birthday
|
||||
WHERE (birthday.month=$month AND birthday.day>=21) OR
|
||||
(birthday.month=($month%12)+1 AND birthday.day<22)
|
||||
WITH DISTINCT friend, city, person
|
||||
OPTIONAL MATCH (friend)<-[:HAS_CREATOR]-(post:Post)
|
||||
WITH friend, city, collect(post) AS posts, person
|
||||
WITH friend,
|
||||
city,
|
||||
size(posts) AS postCount,
|
||||
size([p IN posts WHERE (p)-[:HAS_TAG]->()<-[:HAS_INTEREST]-(person)]) AS commonPostCount
|
||||
RETURN friend.id AS personId,
|
||||
friend.firstName AS personFirstName,
|
||||
friend.lastName AS personLastName,
|
||||
commonPostCount - (postCount - commonPostCount) AS commonInterestScore,
|
||||
friend.gender AS personGender,
|
||||
city.name AS personCityName
|
||||
ORDER BY commonInterestScore DESC, personId ASC
|
||||
LIMIT 10
|
||||
""".replace(
|
||||
"\n", ""
|
||||
),
|
||||
self._get_query_parameters(),
|
||||
)
|
||||
|
||||
if self._vendor == "memgraph":
|
||||
return memgraph
|
||||
else:
|
||||
return neo4j
|
||||
|
||||
def benchmark__interactive__complex_query_11_analytical(self):
|
||||
return (
|
||||
"""
|
||||
|
||||
@@ -1,3 +1,14 @@
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
import random
|
||||
|
||||
from benchmark_context import BenchmarkContext
|
||||
|
||||
46
tests/mgbench/zip_benchgraph.py
Normal file
46
tests/mgbench/zip_benchgraph.py
Normal file
@@ -0,0 +1,46 @@
|
||||
# Copyright 2023 Memgraph Ltd.
|
||||
#
|
||||
# Use of this software is governed by the Business Source License
|
||||
# included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
# License, and you may not use this file except in compliance with the Business Source License.
|
||||
#
|
||||
# As of the Change Date specified in that file, in accordance with
|
||||
# the Business Source License, use of this software will be governed
|
||||
# by the Apache License, Version 2.0, included in the file
|
||||
# licenses/APL.txt.
|
||||
|
||||
import zipfile
|
||||
from pathlib import Path
|
||||
|
||||
import log
|
||||
|
||||
|
||||
def zip_benchgraph():
|
||||
log.info("Creating benchgraph.zip ...")
|
||||
parent = Path(__file__).resolve().parent
|
||||
zip = zipfile.ZipFile("./benchgraph.zip", "w")
|
||||
zip.write(parent / "benchmark.py", "benchgraph/benchmark.py")
|
||||
zip.write(parent / "setup.py", "benchgraph/setup.py")
|
||||
zip.write(parent / "log.py", "benchgraph/log.py")
|
||||
zip.write(parent / "benchmark_context.py", "benchgraph/benchmark_context.py")
|
||||
zip.write(parent / "validation.py", "benchgraph/validation.py")
|
||||
zip.write(parent / "compare_results.py", "benchgraph/compare_results.py")
|
||||
zip.write(parent / "runners.py", "benchgraph/runners.py")
|
||||
zip.write(parent / "helpers.py", "benchgraph/helpers.py")
|
||||
zip.write(parent / "graph_bench.py", "benchgraph/graph_bench.py")
|
||||
zip.write(parent / "README.md", "benchgraph/README.md")
|
||||
zip.write(parent / "how_to_use_benchgraph.md", "benchgraph/how_to_use_benchgraph.md")
|
||||
zip.write(parent / "workloads/__init__.py", "benchgraph/workloads/__init__.py")
|
||||
zip.write(parent / "workloads/base.py", "benchgraph/workloads/base.py")
|
||||
zip.write(parent / "workloads/demo.py", "benchgraph/workloads/demo.py")
|
||||
|
||||
zip.close()
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
zip_benchgraph()
|
||||
|
||||
if Path("./benchgraph.zip").is_file():
|
||||
log.success("benchgraph.zip created successfully")
|
||||
else:
|
||||
log.error("benchgraph.zip was not created")
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -618,7 +618,7 @@ class DurabilityTest : public ::testing::TestWithParam<bool> {
|
||||
std::vector<memgraph::storage::Gid> extended_edge_gids_;
|
||||
};
|
||||
|
||||
void DestroySnapshot(const std::filesystem::path &path) {
|
||||
void CorruptSnapshot(const std::filesystem::path &path) {
|
||||
auto info = memgraph::storage::durability::ReadSnapshotInfo(path);
|
||||
spdlog::info("Destroying snapshot {}", path);
|
||||
memgraph::utils::OutputFile file;
|
||||
@@ -752,7 +752,7 @@ TEST_P(DurabilityTest, SnapshotFallback) {
|
||||
{
|
||||
auto snapshots = GetSnapshotsList();
|
||||
ASSERT_EQ(snapshots.size(), 2);
|
||||
DestroySnapshot(*snapshots.begin());
|
||||
CorruptSnapshot(*snapshots.begin());
|
||||
}
|
||||
|
||||
// Recover snapshot.
|
||||
@@ -835,7 +835,7 @@ TEST_P(DurabilityTest, SnapshotEverythingCorrupt) {
|
||||
spdlog::info("Skipping snapshot {}", snapshot);
|
||||
continue;
|
||||
}
|
||||
DestroySnapshot(snapshot);
|
||||
CorruptSnapshot(snapshot);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2323,7 +2323,7 @@ TEST_P(DurabilityTest, WalAndSnapshotWalRetention) {
|
||||
}
|
||||
|
||||
// Destroy current snapshot.
|
||||
DestroySnapshot(snapshots[i]);
|
||||
CorruptSnapshot(snapshots[i]);
|
||||
}
|
||||
|
||||
// Recover data after all of the snapshots have been destroyed. The recovery
|
||||
|
||||
@@ -14,6 +14,7 @@
|
||||
|
||||
#include <limits>
|
||||
|
||||
#include "storage/v2/id_types.hpp"
|
||||
#include "storage/v2/property_store.hpp"
|
||||
#include "storage/v2/property_value.hpp"
|
||||
#include "storage/v2/temporal.hpp"
|
||||
@@ -651,24 +652,40 @@ TEST(PropertyStore, IsPropertyEqualTemporalData) {
|
||||
}
|
||||
|
||||
TEST(PropertyStore, SetMultipleProperties) {
|
||||
memgraph::storage::PropertyStore store;
|
||||
std::vector<memgraph::storage::PropertyValue> vec{memgraph::storage::PropertyValue(true),
|
||||
memgraph::storage::PropertyValue(123),
|
||||
memgraph::storage::PropertyValue()};
|
||||
std::map<std::string, memgraph::storage::PropertyValue> map{{"nandare", memgraph::storage::PropertyValue(false)}};
|
||||
const memgraph::storage::TemporalData temporal{memgraph::storage::TemporalType::LocalDateTime, 23};
|
||||
std::map<memgraph::storage::PropertyId, memgraph::storage::PropertyValue> data{
|
||||
// The order of property ids are purposfully not monotonic to test that PropertyStore orders them properly
|
||||
const std::vector<std::pair<memgraph::storage::PropertyId, memgraph::storage::PropertyValue>> data{
|
||||
{memgraph::storage::PropertyId::FromInt(1), memgraph::storage::PropertyValue(true)},
|
||||
{memgraph::storage::PropertyId::FromInt(2), memgraph::storage::PropertyValue(123)},
|
||||
{memgraph::storage::PropertyId::FromInt(10), memgraph::storage::PropertyValue(123)},
|
||||
{memgraph::storage::PropertyId::FromInt(3), memgraph::storage::PropertyValue(123.5)},
|
||||
{memgraph::storage::PropertyId::FromInt(4), memgraph::storage::PropertyValue("nandare")},
|
||||
{memgraph::storage::PropertyId::FromInt(5), memgraph::storage::PropertyValue(vec)},
|
||||
{memgraph::storage::PropertyId::FromInt(12), memgraph::storage::PropertyValue(vec)},
|
||||
{memgraph::storage::PropertyId::FromInt(6), memgraph::storage::PropertyValue(map)},
|
||||
{memgraph::storage::PropertyId::FromInt(7), memgraph::storage::PropertyValue(temporal)}};
|
||||
|
||||
store.InitProperties(data);
|
||||
const std::map<memgraph::storage::PropertyId, memgraph::storage::PropertyValue> data_in_map{data.begin(), data.end()};
|
||||
|
||||
for (auto &[key, value] : data) {
|
||||
ASSERT_TRUE(store.IsPropertyEqual(key, value));
|
||||
auto check_store = [data](const memgraph::storage::PropertyStore &store) {
|
||||
for (auto &[key, value] : data) {
|
||||
ASSERT_TRUE(store.IsPropertyEqual(key, value));
|
||||
}
|
||||
};
|
||||
{
|
||||
memgraph::storage::PropertyStore store;
|
||||
EXPECT_TRUE(store.InitProperties(data));
|
||||
check_store(store);
|
||||
EXPECT_FALSE(store.InitProperties(data));
|
||||
EXPECT_FALSE(store.InitProperties(data_in_map));
|
||||
}
|
||||
{
|
||||
memgraph::storage::PropertyStore store;
|
||||
EXPECT_TRUE(store.InitProperties(data_in_map));
|
||||
check_store(store);
|
||||
EXPECT_FALSE(store.InitProperties(data_in_map));
|
||||
EXPECT_FALSE(store.InitProperties(data));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,4 +1,13 @@
|
||||
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
// License, and you may not use this file except in compliance with the Business Source License.
|
||||
//
|
||||
// As of the Change Date specified in that file, in accordance with
|
||||
// the Business Source License, use of this software will be governed
|
||||
// by the Apache License, Version 2.0, included in the file
|
||||
// licenses/APL.txt.
|
||||
|
||||
#include <gtest/gtest.h>
|
||||
#include <chrono>
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// Copyright 2022 Memgraph Ltd.
|
||||
// Copyright 2023 Memgraph Ltd.
|
||||
//
|
||||
// Use of this software is governed by the Business Source License
|
||||
// included in the file licenses/BSL.txt; by using this file, you agree to be bound by the terms of the Business Source
|
||||
@@ -67,6 +67,7 @@ class DeltaGenerator final {
|
||||
auto gid = memgraph::storage::Gid::FromUint(gen_->vertices_count_++);
|
||||
auto delta = memgraph::storage::CreateDeleteObjectDelta(&transaction_);
|
||||
auto &it = gen_->vertices_.emplace_back(gid, delta);
|
||||
// TODO(gvolfing) This is likely problematic when deleting vertices in ANALYTICAL MODE...
|
||||
if (delta != nullptr) {
|
||||
delta->prev.Set(&it);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user