Compare commits
766 Commits
release/0.
...
optimize_e
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e1c0dc0379 | ||
|
|
5bf199710f | ||
|
|
260660f1dd | ||
|
|
e5350a011c | ||
|
|
7bf827bb1e | ||
|
|
5d13c281fa | ||
|
|
020273f475 | ||
|
|
4a99625287 | ||
|
|
5bbed6ef9a | ||
|
|
f0bac53e7b | ||
|
|
514fed51c4 | ||
|
|
2877c343e8 | ||
|
|
50a1d1abb3 | ||
|
|
e8850549d2 | ||
|
|
fd819cd099 | ||
|
|
60f4ffc6a1 | ||
|
|
bd2ec6374a | ||
|
|
c501f59a09 | ||
|
|
210bea83d4 | ||
|
|
57fe3463f2 | ||
|
|
53fcd8ac4d | ||
|
|
259cba5d43 | ||
|
|
285b409927 | ||
|
|
8ebab84324 | ||
|
|
3fd9ce4a33 | ||
|
|
be4eb95a98 | ||
|
|
903a9f4636 | ||
|
|
18bd02423a | ||
|
|
58c0c4cebb | ||
|
|
110ca3968c | ||
|
|
9072fb7703 | ||
|
|
2b7707a2f1 | ||
|
|
76ca019f31 | ||
|
|
609b9a20f1 | ||
|
|
e489e4f3e7 | ||
|
|
ab4d1efe0b | ||
|
|
036da58d30 | ||
|
|
919f07fae1 | ||
|
|
ca1e98ad94 | ||
|
|
f0aca2d23b | ||
|
|
ec9840fff2 | ||
|
|
992e718a97 | ||
|
|
2ec6b7f40b | ||
|
|
05eca46267 | ||
|
|
fae039c215 | ||
|
|
3b9133fd5a | ||
|
|
9d056e7649 | ||
|
|
aa4f68a37d | ||
|
|
84721f7e0a | ||
|
|
261aa4f49b | ||
|
|
5ce1526995 | ||
|
|
cb843ee664 | ||
|
|
0f1ca745e5 | ||
|
|
3b781bf525 | ||
|
|
d573eda8bb | ||
|
|
00226dee24 | ||
|
|
546bfc0ede | ||
|
|
5b1ba10183 | ||
|
|
b25e9968ee | ||
|
|
bcd23fe3cb | ||
|
|
68e5610566 | ||
|
|
0ea96663ba | ||
|
|
da17fe92d6 | ||
|
|
e73eac77a9 | ||
|
|
d51a61fc5f | ||
|
|
b875649270 | ||
|
|
05cc35bf93 | ||
|
|
63f8298033 | ||
|
|
df95775222 | ||
|
|
eb22edfd35 | ||
|
|
1b85d77e9e | ||
|
|
cf1a86ed13 | ||
|
|
7fb3f62703 | ||
|
|
cb4b71bdbd | ||
|
|
30ec570bb9 | ||
|
|
d917c3f0fd | ||
|
|
d842adbed3 | ||
|
|
cdfcbc106c | ||
|
|
651b6f3a5a | ||
|
|
0d9bd74a8a | ||
|
|
802f8aceda | ||
|
|
7ddce539fa | ||
|
|
c3e4f81026 | ||
|
|
208705f296 | ||
|
|
69634a5354 | ||
|
|
b8f282468d | ||
|
|
ab38161cd2 | ||
|
|
3a5f140c2b | ||
|
|
eead0f79fc | ||
|
|
91017b7f36 | ||
|
|
00f8d54249 | ||
|
|
4fcdd52f88 | ||
|
|
6c947947eb | ||
|
|
64fd281b2e | ||
|
|
97e250129e | ||
|
|
b02b201129 | ||
|
|
2c6a55775d | ||
|
|
940bf6722c | ||
|
|
49b5343238 | ||
|
|
26a0866938 | ||
|
|
6545283dac | ||
|
|
69c735934c | ||
|
|
64e837b355 | ||
|
|
a586f2f98d | ||
|
|
9fc51f74a0 | ||
|
|
128771a6ec | ||
|
|
f5a49ed29f | ||
|
|
398503da7a | ||
|
|
0819b40202 | ||
|
|
029be10f1d | ||
|
|
a9dc344b49 | ||
|
|
8b0dca9eab | ||
|
|
cb813c3070 | ||
|
|
6349fc9501 | ||
|
|
c4167bafdd | ||
|
|
6f51141148 | ||
|
|
97d45ab1d8 | ||
|
|
6abd356d01 | ||
|
|
99a6c72bba | ||
|
|
173f5430aa | ||
|
|
362dc95e27 | ||
|
|
eead24a562 | ||
|
|
d79dd69607 | ||
|
|
024bf0c578 | ||
|
|
d7443a5558 | ||
|
|
2df357b012 | ||
|
|
b2b5a6e2a0 | ||
|
|
5e2ee6c817 | ||
|
|
862a1afdf1 | ||
|
|
bbce21e78f | ||
|
|
15c8662023 | ||
|
|
beaba0fc16 | ||
|
|
8f70c5f2a5 | ||
|
|
14c651d3ba | ||
|
|
04efc7a4a6 | ||
|
|
19832b5838 | ||
|
|
e63fae2d5b | ||
|
|
34dd47ef07 | ||
|
|
d3f275b231 | ||
|
|
aad4bcb7a0 | ||
|
|
034b54cb72 | ||
|
|
8cf51d9f68 | ||
|
|
1cd1da84fd | ||
|
|
128a6cd522 | ||
|
|
d9eeedb9ee | ||
|
|
156e2cd095 | ||
|
|
8b834c702c | ||
|
|
eda5213d95 | ||
|
|
1f2a15e7c8 | ||
|
|
d72e7fa38d | ||
|
|
e5e37bc14a | ||
|
|
3ee068bbf9 | ||
|
|
68e846b182 | ||
|
|
310e305cfb | ||
|
|
9d6a23b6bd | ||
|
|
f2d5ab61c4 | ||
|
|
0f77c85824 | ||
|
|
d6d4153fb7 | ||
|
|
7d6a5e5b9c | ||
|
|
c529d52664 | ||
|
|
45451bae3b | ||
|
|
3e11f38548 | ||
|
|
6e4047a847 | ||
|
|
8febdc12fb | ||
|
|
3f23a10f44 | ||
|
|
11300960de | ||
|
|
1d5f387ddd | ||
|
|
c4c3a254bf | ||
|
|
b4beb0fc86 | ||
|
|
58e6097664 | ||
|
|
ff21c0705c | ||
|
|
2a2b99b02a | ||
|
|
3daab6ce97 | ||
|
|
fbd7274c95 | ||
|
|
6efc84f022 | ||
|
|
287c2e94d1 | ||
|
|
417cf4b30b | ||
|
|
68e7fd3d36 | ||
|
|
5261d82063 | ||
|
|
9eb87bcf3e | ||
|
|
a9491d3e68 | ||
|
|
b42e47b0be | ||
|
|
898c894a48 | ||
|
|
a3c2492672 | ||
|
|
a0b8871b36 | ||
|
|
bb6cf35441 | ||
|
|
5bc301d21d | ||
|
|
1b89e679df | ||
|
|
43e0520bc8 | ||
|
|
2c8e45e889 | ||
|
|
0876a8848d | ||
|
|
fb4641a6be | ||
|
|
201f75e809 | ||
|
|
dc8dad9794 | ||
|
|
aa02745915 | ||
|
|
b2d5a8eeca | ||
|
|
c09b175c76 | ||
|
|
f1fe77adfb | ||
|
|
35f8978560 | ||
|
|
9e8fb2516b | ||
|
|
0a66feccff | ||
|
|
d008a2ad8d | ||
|
|
7478300762 | ||
|
|
0bc298c3ad | ||
|
|
05f120b7d4 | ||
|
|
d73d153978 | ||
|
|
b489ac7cff | ||
|
|
e15576f56c | ||
|
|
a98463b0bd | ||
|
|
705631a35d | ||
|
|
d4f0bb0e38 | ||
|
|
531db2d47c | ||
|
|
116262d9a0 | ||
|
|
bbfef45b37 | ||
|
|
05b00edfd4 | ||
|
|
480df4ed69 | ||
|
|
80e0e439b7 | ||
|
|
351258ace8 | ||
|
|
74d3663821 | ||
|
|
eb0b3141d5 | ||
|
|
ff2f8031a9 | ||
|
|
094d4f282d | ||
|
|
3dd2657320 | ||
|
|
6fe474282a | ||
|
|
7fc0fb6520 | ||
|
|
063e297e1e | ||
|
|
86b1688192 | ||
|
|
f629de7e60 | ||
|
|
b737e53456 | ||
|
|
10ca68bb2a | ||
|
|
bfbd8538d4 | ||
|
|
3e0e17d469 | ||
|
|
b57f91fcfc | ||
|
|
066a96c0ae | ||
|
|
1ae6b71c5f | ||
|
|
cbe15e7f44 | ||
|
|
65a7ba01da | ||
|
|
7a2bbd4bb3 | ||
|
|
589e0e098b | ||
|
|
41d4185156 | ||
|
|
599c0a641f | ||
|
|
1fb49c4865 | ||
|
|
df1485aeec | ||
|
|
b2e1056389 | ||
|
|
e4c9411e63 | ||
|
|
a0bc1371dd | ||
|
|
21ad5d4328 | ||
|
|
8e3ab1ad0f | ||
|
|
cccf32e79d | ||
|
|
22bd60c613 | ||
|
|
8059a3e653 | ||
|
|
a7f4c98bea | ||
|
|
483f4d04bd | ||
|
|
3e7aef432f | ||
|
|
10ea9c773e | ||
|
|
b782271be8 | ||
|
|
a8ffcfa046 | ||
|
|
7b78665cd8 | ||
|
|
4abaf27765 | ||
|
|
ea2806bd57 | ||
|
|
c8dbaf5979 | ||
|
|
17049ada09 | ||
|
|
1b619f51b2 | ||
|
|
537855a0b2 | ||
|
|
5822b44b15 | ||
|
|
bf01c58ed9 | ||
|
|
89a5566f3f | ||
|
|
29452f8774 | ||
|
|
60ad05acff | ||
|
|
4f593c7fca | ||
|
|
770ea1189a | ||
|
|
695bb343f1 | ||
|
|
12b4ec1589 | ||
|
|
b33d2c3940 | ||
|
|
477acad1f6 | ||
|
|
ddca2b40f5 | ||
|
|
4817be0add | ||
|
|
3fb7e5378d | ||
|
|
1d88893715 | ||
|
|
bd2c30fddc | ||
|
|
06e6ead4d2 | ||
|
|
728b37080d | ||
|
|
48a531aac1 | ||
|
|
914fc1a656 | ||
|
|
1d1c182c2d | ||
|
|
c6e19ec09f | ||
|
|
693dab78d2 | ||
|
|
69eca9b043 | ||
|
|
5aeaad198b | ||
|
|
b23f88c607 | ||
|
|
4fd8bdce4c | ||
|
|
265b203b00 | ||
|
|
661e5185d8 | ||
|
|
7348ad6800 | ||
|
|
6c00d146f2 | ||
|
|
ced84e17b6 | ||
|
|
834ce5f51a | ||
|
|
bb1308acc7 | ||
|
|
e1f31d3d02 | ||
|
|
339cd9b84e | ||
|
|
7deac4ac8b | ||
|
|
fec887a67c | ||
|
|
f3a3a4ed87 | ||
|
|
ec65187442 | ||
|
|
079c0495c3 | ||
|
|
90cfed2ada | ||
|
|
8716b8e992 | ||
|
|
a277541354 | ||
|
|
b01564f179 | ||
|
|
d9bb4e2e46 | ||
|
|
05d0aee494 | ||
|
|
aabec99a8e | ||
|
|
d277dd49a3 | ||
|
|
18f9d19b18 | ||
|
|
530eed5c6d | ||
|
|
12f4e0068a | ||
|
|
e8976e0f1c | ||
|
|
6eb52581eb | ||
|
|
8606e69fd6 | ||
|
|
c7b045bffc | ||
|
|
b66cc66503 | ||
|
|
0e4719018a | ||
|
|
0ebd52aac3 | ||
|
|
6c971b856e | ||
|
|
6cb293688d | ||
|
|
16709dff6c | ||
|
|
95dd3481c0 | ||
|
|
47c0c629c7 | ||
|
|
636c551047 | ||
|
|
72384b2b71 | ||
|
|
a9b1ff9bea | ||
|
|
e40bc32624 | ||
|
|
bd21bc82b7 | ||
|
|
9c1680e82c | ||
|
|
1a78c3695d | ||
|
|
10196f3d7d | ||
|
|
2c86fefbb5 | ||
|
|
df689dccd5 | ||
|
|
6033fbe9bd | ||
|
|
519a204424 | ||
|
|
3fc5e57fef | ||
|
|
906933d2b3 | ||
|
|
e9a937ad6d | ||
|
|
6dc2cdfae4 | ||
|
|
69ea410406 | ||
|
|
24a576c8e9 | ||
|
|
d417ffee6e | ||
|
|
f868b51483 | ||
|
|
26dc73c809 | ||
|
|
9119c89eba | ||
|
|
4e7ea34ae9 | ||
|
|
d7d291217b | ||
|
|
42facfacbc | ||
|
|
bd1e848363 | ||
|
|
0ffb58b764 | ||
|
|
9551b6973a | ||
|
|
f7ef5f50a4 | ||
|
|
046cd80054 | ||
|
|
9e678f8cbe | ||
|
|
69e4f1784b | ||
|
|
d04e23805d | ||
|
|
7e82cc6550 | ||
|
|
18b801a722 | ||
|
|
7e81a95b81 | ||
|
|
0a23eb11ae | ||
|
|
8ea281649a | ||
|
|
da68f86fc9 | ||
|
|
738b5fb8d8 | ||
|
|
b45ae403e6 | ||
|
|
d744711e5e | ||
|
|
8d87e38c64 | ||
|
|
086fc47769 | ||
|
|
5abcb5081d | ||
|
|
9534dfb6ed | ||
|
|
e2112faff0 | ||
|
|
6913feacc7 | ||
|
|
4e604de9d7 | ||
|
|
e628b5ba6e | ||
|
|
5da32c1bff | ||
|
|
ccca46370d | ||
|
|
e33fa5f9c0 | ||
|
|
721eefe263 | ||
|
|
be9ed7e879 | ||
|
|
482798295e | ||
|
|
d58a1cbb58 | ||
|
|
8dc3153fde | ||
|
|
b94e50bf1c | ||
|
|
9f855676cb | ||
|
|
f560293657 | ||
|
|
2afc1b99f6 | ||
|
|
5a8bda2531 | ||
|
|
0138d277d4 | ||
|
|
09cfca35f8 | ||
|
|
09c58501f1 | ||
|
|
945bbfdc49 | ||
|
|
45ef069372 | ||
|
|
e51954fc94 | ||
|
|
7776160836 | ||
|
|
ae280fd8db | ||
|
|
fb5a2ed4b6 | ||
|
|
6cfec787dc | ||
|
|
13c9bf76af | ||
|
|
2e1a717dcb | ||
|
|
ad32db5168 | ||
|
|
a37755ce43 | ||
|
|
a928c158da | ||
|
|
ac230d0c2d | ||
|
|
d80ff745eb | ||
|
|
d6a6d280dd | ||
|
|
4004e94ca1 | ||
|
|
36afc6c5f3 | ||
|
|
3c9a46f823 | ||
|
|
f994b68ad5 | ||
|
|
3b336e3e0b | ||
|
|
715162e205 | ||
|
|
e016c74e4b | ||
|
|
15911b64dc | ||
|
|
cbf826e0c3 | ||
|
|
644a3a0b2a | ||
|
|
8cd9f696cf | ||
|
|
90a093bd95 | ||
|
|
542a324c96 | ||
|
|
cd03e13443 | ||
|
|
be91126134 | ||
|
|
03cb007339 | ||
|
|
524acb17a1 | ||
|
|
50f6e348dc | ||
|
|
839d45b3f8 | ||
|
|
560eb04f67 | ||
|
|
a3ecc52429 | ||
|
|
e8a1d15a55 | ||
|
|
1abee1ed3a | ||
|
|
5af3d0ff68 | ||
|
|
62a628c51f | ||
|
|
883f9c7ed3 | ||
|
|
b459639968 | ||
|
|
11c0dde11c | ||
|
|
2f3fa656d9 | ||
|
|
7bf40eb5d2 | ||
|
|
7e44434cdf | ||
|
|
2a0b0d969f | ||
|
|
13ea35af2d | ||
|
|
8a99670301 | ||
|
|
c4555f5448 | ||
|
|
999b3ef79f | ||
|
|
30413a7b4f | ||
|
|
1def0c9104 | ||
|
|
782c377f5d | ||
|
|
cc27a04139 | ||
|
|
b71345655f | ||
|
|
ccdd58b336 | ||
|
|
50b6afd73d | ||
|
|
59105f68bd | ||
|
|
8de31092ad | ||
|
|
5c93f81881 | ||
|
|
6d4fe5cdd5 | ||
|
|
7b5263d300 | ||
|
|
e8a41e4457 | ||
|
|
27f09e1c0a | ||
|
|
6dd9d32721 | ||
|
|
276e09d7d3 | ||
|
|
92dfc93b20 | ||
|
|
06f761bdf9 | ||
|
|
50ddd59450 | ||
|
|
60da033010 | ||
|
|
ee555b0c0d | ||
|
|
e8e4cd7f97 | ||
|
|
ad4c80af13 | ||
|
|
9c6bf4b1b8 | ||
|
|
cc56ac3dd8 | ||
|
|
dee885d69c | ||
|
|
bbed7a2397 | ||
|
|
e8810a4152 | ||
|
|
25eb2c147a | ||
|
|
b3914c6b5d | ||
|
|
d913a67e16 | ||
|
|
f950a91732 | ||
|
|
f6d5f576d5 | ||
|
|
dc5eb4befd | ||
|
|
593f7a3499 | ||
|
|
77a0d7b8fa | ||
|
|
c240b5b564 | ||
|
|
866ed45562 | ||
|
|
1598cb24ea | ||
|
|
35d789c56b | ||
|
|
16715d5005 | ||
|
|
a9f5f45b3d | ||
|
|
2e0dd19bac | ||
|
|
f807b495ab | ||
|
|
3f3c55a4aa | ||
|
|
435af8b833 | ||
|
|
cc1c1513ef | ||
|
|
fae407d3fe | ||
|
|
42c245df8a | ||
|
|
e4852cc5e3 | ||
|
|
2caf0e617e | ||
|
|
90d4ebdb1e | ||
|
|
151812da1a | ||
|
|
1e10b9cb77 | ||
|
|
c44dbdda86 | ||
|
|
46b1783618 | ||
|
|
e866526c7a | ||
|
|
28413fd626 | ||
|
|
c0bd59bb09 | ||
|
|
a84ffe86c1 | ||
|
|
8f5b88f24a | ||
|
|
afbf672915 | ||
|
|
10c8256ec9 | ||
|
|
b19cd4f5d1 | ||
|
|
8fc9298832 | ||
|
|
9966ba1d52 | ||
|
|
b7bbd026de | ||
|
|
855c2ea9ca | ||
|
|
adc355a22a | ||
|
|
bff9cf07de | ||
|
|
60d742a2dc | ||
|
|
a0fb3fc463 | ||
|
|
f23e2e12c4 | ||
|
|
87e00f4fef | ||
|
|
f7b764607d | ||
|
|
200ce5f45e | ||
|
|
a0705746cb | ||
|
|
4e36b646df | ||
|
|
ec909ced57 | ||
|
|
7e9175052a | ||
|
|
3c85319701 | ||
|
|
76f0d5873b | ||
|
|
03cc568e39 | ||
|
|
bc0c944910 | ||
|
|
42f6118c00 | ||
|
|
b10255a12f | ||
|
|
c68ed8d94e | ||
|
|
d5b02eafb1 | ||
|
|
1580c46abc | ||
|
|
367cb44983 | ||
|
|
77bda187d2 | ||
|
|
57806544cd | ||
|
|
7bff678cd9 | ||
|
|
647388792a | ||
|
|
b42ee31c1d | ||
|
|
53fa1482cb | ||
|
|
c5f722b724 | ||
|
|
573b6cb045 | ||
|
|
4afcf1b655 | ||
|
|
4e7c569071 | ||
|
|
ea27ac9391 | ||
|
|
958bc870b3 | ||
|
|
f0382c82cd | ||
|
|
6fab357430 | ||
|
|
f273be63a2 | ||
|
|
9b984ab2f2 | ||
|
|
92bad4f62b | ||
|
|
126c9d697b | ||
|
|
0478e89646 | ||
|
|
814bb66ea6 | ||
|
|
98f83e0c88 | ||
|
|
2e7d3822fe | ||
|
|
0a7d4278b1 | ||
|
|
291158160d | ||
|
|
e525818355 | ||
|
|
0bcc1d67bc | ||
|
|
42ac5d4ea3 | ||
|
|
48587d6d5e | ||
|
|
157590a294 | ||
|
|
7f179dc462 | ||
|
|
594f3ec1b0 | ||
|
|
804a44dd6c | ||
|
|
1b50dc60f9 | ||
|
|
9d6b578237 | ||
|
|
f7f861ca71 | ||
|
|
c12e4a49b1 | ||
|
|
4c0fc11f69 | ||
|
|
e0ffc533b9 | ||
|
|
0dcfdb9b89 | ||
|
|
1d973f7e31 | ||
|
|
4f4903803f | ||
|
|
ff0d5962e1 | ||
|
|
04ceb8d4b1 | ||
|
|
fb7025a716 | ||
|
|
89b6262fa4 | ||
|
|
d2ff465f8e | ||
|
|
c12a87e9ca | ||
|
|
63de0b5db4 | ||
|
|
dd9180da32 | ||
|
|
aaf0c1ca08 | ||
|
|
1513a455de | ||
|
|
4f9e9aeafe | ||
|
|
857de23687 | ||
|
|
fba5d75bf4 | ||
|
|
478acb7934 | ||
|
|
df27b8477d | ||
|
|
02d3a04ffa | ||
|
|
4a53a31fcf | ||
|
|
28590aea53 | ||
|
|
0a2f2bfc90 | ||
|
|
a7b672ebbf | ||
|
|
9831f0396a | ||
|
|
8bfebfca9d | ||
|
|
21cee1eaec | ||
|
|
098333f735 | ||
|
|
3dd393f8eb | ||
|
|
6628d20e5a | ||
|
|
b923d2bc36 | ||
|
|
3932376301 | ||
|
|
d36da12b48 | ||
|
|
a149f5bd81 | ||
|
|
1a8ae7f9c2 | ||
|
|
b510ac1049 | ||
|
|
59bc9d8989 | ||
|
|
b6dfb44b4c | ||
|
|
81ad5110ff | ||
|
|
f92dee6e7b | ||
|
|
f047f55020 | ||
|
|
0c42bedf2f | ||
|
|
7d9f741ceb | ||
|
|
4d2eda398f | ||
|
|
317f7f118c | ||
|
|
f35225f26b | ||
|
|
e04b96a0ac | ||
|
|
21be41995c | ||
|
|
6c69351ea1 | ||
|
|
e9ea8693fa | ||
|
|
9d3f75a34d | ||
|
|
590fc5c9a8 | ||
|
|
1d5952bb9c | ||
|
|
24eb021840 | ||
|
|
ac3560ec8c | ||
|
|
974ce1e9e9 | ||
|
|
66e644551f | ||
|
|
df72723664 | ||
|
|
96bb7a87e3 | ||
|
|
39deba3b9e | ||
|
|
d04993df67 | ||
|
|
7e35798401 | ||
|
|
1d2bb2cda6 | ||
|
|
8fb3a53b78 | ||
|
|
a1b5bdd88f | ||
|
|
3cd89e1fe4 | ||
|
|
90c83de7c2 | ||
|
|
7cd96dc2f5 | ||
|
|
41551bcd36 | ||
|
|
e5b3414335 | ||
|
|
b7738c64b3 | ||
|
|
a8b81fdcde | ||
|
|
9f006c1b57 | ||
|
|
2b8f068ca9 | ||
|
|
d4c2798551 | ||
|
|
4b5068e455 | ||
|
|
44917fdfa6 | ||
|
|
ba55372610 | ||
|
|
008efaf243 | ||
|
|
68a1a2da23 | ||
|
|
7b824ed622 | ||
|
|
5632852890 | ||
|
|
d63eb191f9 | ||
|
|
f1327c52ef | ||
|
|
f48ad62647 | ||
|
|
eaabcd8d0d | ||
|
|
fb4ce6c5ed | ||
|
|
dd5d29fa1d | ||
|
|
963d779d48 | ||
|
|
5695774251 | ||
|
|
181f937c15 | ||
|
|
7ea2d1b638 | ||
|
|
c4a1a6c0b4 | ||
|
|
d0411510fa | ||
|
|
fbdcad1106 | ||
|
|
6f83fff171 | ||
|
|
8363991575 | ||
|
|
1d30a2e5cb | ||
|
|
31a4c55e76 | ||
|
|
456267249e | ||
|
|
32b35f6f88 | ||
|
|
000d6dba55 | ||
|
|
6e9404c89a | ||
|
|
d531e44bef | ||
|
|
74d9dd0b0a | ||
|
|
a456d6cdc0 | ||
|
|
ec0caad44c | ||
|
|
14c4c6d060 | ||
|
|
b922ca75d0 | ||
|
|
f5a94d6e29 | ||
|
|
eed83a210e | ||
|
|
0e95934719 | ||
|
|
565927631f | ||
|
|
64da28ca83 | ||
|
|
4049b9d10a | ||
|
|
bb4e747fd3 | ||
|
|
b6b6403b06 | ||
|
|
ad892f2db3 | ||
|
|
32e56684db | ||
|
|
bfbace8168 | ||
|
|
4e5a91e7fb | ||
|
|
2562070b06 | ||
|
|
e7f363ecbe | ||
|
|
34db077cbd | ||
|
|
47ce444c02 | ||
|
|
e8903ef477 | ||
|
|
d1be5ea136 | ||
|
|
c50b518fad | ||
|
|
dfc546183d | ||
|
|
5953f07be3 | ||
|
|
ad740e4ae2 | ||
|
|
bd0fd2619c | ||
|
|
95c9755e13 | ||
|
|
4669e0ae7d | ||
|
|
6a7b983a58 | ||
|
|
beec3e3f5f | ||
|
|
ffbe5b449d | ||
|
|
0a7de969f3 | ||
|
|
2aa960403a | ||
|
|
d17d1497d1 | ||
|
|
e2e7823ec4 | ||
|
|
ab53cbb0c3 | ||
|
|
90286b8166 | ||
|
|
7128320b36 | ||
|
|
87511460f1 | ||
|
|
3e20afde6e | ||
|
|
29aa9b9e92 | ||
|
|
4281a57923 | ||
|
|
6c422f3208 | ||
|
|
a3229881bf | ||
|
|
c96421424f | ||
|
|
d156d5b41c | ||
|
|
84a6ab75cb | ||
|
|
fd87facf65 | ||
|
|
ec4d8c7bb1 | ||
|
|
cfbfeb1dfc | ||
|
|
015573b5af | ||
|
|
de48548164 | ||
|
|
7d0590a753 | ||
|
|
fd81ebdfe3 | ||
|
|
0c7313bb5f | ||
|
|
c425624ba6 | ||
|
|
0683ed4134 | ||
|
|
6d10d90d98 | ||
|
|
4f16b814c7 | ||
|
|
029a36eab1 | ||
|
|
a1fa7de115 | ||
|
|
98dc7e2849 | ||
|
|
1a50165c22 | ||
|
|
c11d391e62 | ||
|
|
b7a5532cc2 | ||
|
|
1fb5d14751 | ||
|
|
3121f7d89d | ||
|
|
3a88668ace | ||
|
|
cc2160d397 | ||
|
|
c7f69fd861 | ||
|
|
5fefd9d82f | ||
|
|
c3bfd3004b | ||
|
|
9889b9421d | ||
|
|
4776bea221 | ||
|
|
5906258de0 | ||
|
|
79947c376b | ||
|
|
0c7f384fd2 | ||
|
|
2011aea165 | ||
|
|
ba6632a00d | ||
|
|
551c6c5ad6 | ||
|
|
3a0dda976a | ||
|
|
b5e255b896 | ||
|
|
b3d1cd8257 | ||
|
|
d968370c3e | ||
|
|
d910813955 | ||
|
|
591eadad20 | ||
|
|
ec67e71d39 | ||
|
|
0c111d52dc |
@@ -1,7 +0,0 @@
|
||||
{
|
||||
"project_id" : "memgraph",
|
||||
"conduit_uri" : "https://phabricator.memgraph.io",
|
||||
"phabricator_uri" : "https://phabricator.memgraph.io",
|
||||
"git.default-relative-commit": "origin/master",
|
||||
"arc.land.onto.default": "master"
|
||||
}
|
||||
9
.arclint
9
.arclint
@@ -1,9 +0,0 @@
|
||||
{
|
||||
"linters": {
|
||||
"clang-tidy": {
|
||||
"type": "script-and-regex",
|
||||
"script-and-regex.script": "./tools/arc-clang-tidy",
|
||||
"script-and-regex.regex": "/^(?P<file>.*):(?P<line>\\d+):(?P<char>\\d+): (?P<severity>warning|error): (?P<message>.*)$/m"
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,9 +1,11 @@
|
||||
---
|
||||
BasedOnStyle: Google
|
||||
---
|
||||
Language: Cpp
|
||||
BasedOnStyle: Google
|
||||
Standard: "C++11"
|
||||
Standard: "c++20"
|
||||
UseTab: Never
|
||||
DerivePointerAlignment: false
|
||||
PointerAlignment: Right
|
||||
ColumnLimit : 80
|
||||
ColumnLimit : 120
|
||||
IncludeBlocks: Preserve
|
||||
...
|
||||
|
||||
21
.clang-tidy
21
.clang-tidy
@@ -1,5 +1,9 @@
|
||||
---
|
||||
Checks: '*,
|
||||
-abseil-string-find-str-contains,
|
||||
-altera-id-dependent-backward-branch,
|
||||
-altera-struct-pack-align,
|
||||
-altera-unroll-loops,
|
||||
-android-*,
|
||||
-cert-err58-cpp,
|
||||
-cppcoreguidelines-avoid-c-arrays,
|
||||
@@ -26,6 +30,7 @@ Checks: '*,
|
||||
-fuchsia-virtual-inheritance,
|
||||
-google-explicit-constructor,
|
||||
-google-readability-*,
|
||||
-google-runtime-references,
|
||||
-hicpp-avoid-c-arrays,
|
||||
-hicpp-avoid-goto,
|
||||
-hicpp-braces-around-statements,
|
||||
@@ -34,10 +39,14 @@ Checks: '*,
|
||||
-hicpp-no-assembler,
|
||||
-hicpp-no-malloc,
|
||||
-hicpp-use-equals-default,
|
||||
-hicpp-use-nullptr,
|
||||
-hicpp-vararg,
|
||||
-llvm-header-guard,
|
||||
-llvm-include-order,
|
||||
-llvmlibc-callee-namespace,
|
||||
-llvmlibc-implementation-in-namespace,
|
||||
-llvmlibc-restrict-system-libc-headers,
|
||||
-misc-non-private-member-variables-in-classes,
|
||||
-misc-unused-parameters,
|
||||
-modernize-avoid-c-arrays,
|
||||
-modernize-concat-nested-namespaces,
|
||||
-modernize-pass-by-value,
|
||||
@@ -47,11 +56,16 @@ Checks: '*,
|
||||
-performance-unnecessary-value-param,
|
||||
-readability-braces-around-statements,
|
||||
-readability-else-after-return,
|
||||
-readability-function-cognitive-complexity,
|
||||
-readability-implicit-bool-conversion,
|
||||
-readability-magic-numbers,
|
||||
-readability-named-parameter'
|
||||
-readability-named-parameter,
|
||||
-misc-no-recursion,
|
||||
-concurrency-mt-unsafe,
|
||||
-bugprone-easily-swappable-parameters'
|
||||
|
||||
WarningsAsErrors: ''
|
||||
HeaderFilterRegex: ''
|
||||
HeaderFilterRegex: 'src/.*'
|
||||
AnalyzeTemporaryDtors: false
|
||||
FormatStyle: none
|
||||
CheckOptions:
|
||||
@@ -76,4 +90,3 @@ CheckOptions:
|
||||
- key: modernize-use-nullptr.NullMacros
|
||||
value: 'NULL'
|
||||
...
|
||||
|
||||
|
||||
36
.githooks/pre-commit
Executable file
36
.githooks/pre-commit
Executable file
@@ -0,0 +1,36 @@
|
||||
#!/bin/sh
|
||||
|
||||
project_folder=$(git rev-parse --show-toplevel)
|
||||
if git rev-parse --verify HEAD >/dev/null 2>&1
|
||||
then
|
||||
against=HEAD
|
||||
else
|
||||
# Initial commit: diff against an empty tree object
|
||||
against=$(git hash-object -t tree /dev/null)
|
||||
fi
|
||||
|
||||
# Redirect output to stderr.
|
||||
exec 1>&2
|
||||
|
||||
tmpdir=$(mktemp -d repo-XXXXXXXX)
|
||||
trap "rm -rf $tmpdir" EXIT INT
|
||||
|
||||
modified_files=$(git diff --cached --name-only --diff-filter=AM $against | sed -nE "/.*\.(cpp|cc|cxx|c|h|hpp)$/p")
|
||||
FAIL=0
|
||||
for file in $modified_files; do
|
||||
echo "Checking $file..."
|
||||
|
||||
cp $project_folder/.clang-format $project_folder/.clang-tidy $tmpdir
|
||||
|
||||
git checkout-index --prefix="$tmpdir/" -- $file
|
||||
|
||||
# Do not break header checker
|
||||
echo "Running header checker..."
|
||||
$project_folder/tools/header-checker.py $tmpdir/$file $file --amend-year
|
||||
CODE=$?
|
||||
if [ $CODE -ne 0 ]; then
|
||||
FAIL=1
|
||||
fi
|
||||
done;
|
||||
|
||||
return ${FAIL}
|
||||
34
.github/ISSUE_TEMPLATE/bug_report.md
vendored
Normal file
34
.github/ISSUE_TEMPLATE/bug_report.md
vendored
Normal file
@@ -0,0 +1,34 @@
|
||||
---
|
||||
name: Bug report
|
||||
about: Create a report to help us improve
|
||||
title: "[BUG] "
|
||||
labels: bug
|
||||
assignees: gitbuda, antonio2368
|
||||
|
||||
---
|
||||
|
||||
|
||||
**Memgraph version**
|
||||
Which version did you use?
|
||||
|
||||
**Environment**
|
||||
Some information about the environment you are using Memgraph on: operating
|
||||
system, how do you connect, with or without docker, which driver etc.
|
||||
|
||||
**Describe the bug**
|
||||
A clear and concise description of what the bug is.
|
||||
|
||||
**To Reproduce**
|
||||
Steps to reproduce the behavior:
|
||||
1. Run the following query '...'
|
||||
2. Click on '....'
|
||||
|
||||
**Expected behavior**
|
||||
A clear and concise description of what you expected to happen.
|
||||
|
||||
**Logs**
|
||||
If applicable, add logs of Memgraph, CLI output or screenshots to help explain
|
||||
your problem.
|
||||
|
||||
**Additional context**
|
||||
Add any other context about the problem here.
|
||||
14
.github/pull_request_template.md
vendored
Normal file
14
.github/pull_request_template.md
vendored
Normal file
@@ -0,0 +1,14 @@
|
||||
[master < Epic] PR
|
||||
- [ ] Check, and update documentation if necessary
|
||||
- [ ] Write E2E tests
|
||||
- [ ] Compare the [benchmarking results](https://bench-graph.memgraph.com/) between the master branch and the Epic branch
|
||||
- [ ] Provide the full content or a guide for the final git message
|
||||
|
||||
[master < Task] PR
|
||||
- [ ] Check, and update documentation if necessary
|
||||
- [ ] Provide the full content or a guide for the final git message
|
||||
|
||||
|
||||
To keep docs changelog up to date, one more thing to do:
|
||||
- [ ] Write a release note here, including added/changed clauses
|
||||
- [ ] Tag someone from docs team in the comments
|
||||
82
.github/workflows/daily_benchmark.yaml
vendored
Normal file
82
.github/workflows/daily_benchmark.yaml
vendored
Normal file
@@ -0,0 +1,82 @@
|
||||
name: Daily Benchmark
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
schedule:
|
||||
- cron: "0 22 * * *"
|
||||
|
||||
jobs:
|
||||
release_benchmarks:
|
||||
name: "Release benchmarks"
|
||||
runs-on: [self-hosted, Linux, X64, Diff, Gen7]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build release binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build only memgraph release binaries.
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=release ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Run macro benchmarks
|
||||
run: |
|
||||
cd tests/macro_benchmark
|
||||
./harness QuerySuite MemgraphRunner \
|
||||
--groups aggregation 1000_create unwind_create dense_expand match \
|
||||
--no-strict
|
||||
|
||||
- name: Get branch name (merge)
|
||||
if: github.event_name != 'pull_request'
|
||||
shell: bash
|
||||
run: echo "BRANCH_NAME=$(echo ${GITHUB_REF#refs/heads/} | tr / -)" >> $GITHUB_ENV
|
||||
|
||||
- name: Get branch name (pull request)
|
||||
if: github.event_name == 'pull_request'
|
||||
shell: bash
|
||||
run: echo "BRANCH_NAME=$(echo ${GITHUB_HEAD_REF} | tr / -)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload macro benchmark results
|
||||
run: |
|
||||
cd tools/bench-graph-client
|
||||
virtualenv -p python3 ve3
|
||||
source ve3/bin/activate
|
||||
pip install -r requirements.txt
|
||||
./main.py --benchmark-name "macro_benchmark" \
|
||||
--benchmark-results-path "../../tests/macro_benchmark/.harness_summary" \
|
||||
--github-run-id "${{ github.run_id }}" \
|
||||
--github-run-number "${{ github.run_number }}" \
|
||||
--head-branch-name "${{ env.BRANCH_NAME }}"
|
||||
|
||||
- name: Run mgbench
|
||||
run: |
|
||||
cd tests/mgbench
|
||||
./benchmark.py vendor-native --num-workers-for-benchmark 12 --export-results benchmark_result.json pokec/medium/*/*
|
||||
|
||||
- name: Upload mgbench results
|
||||
run: |
|
||||
cd tools/bench-graph-client
|
||||
virtualenv -p python3 ve3
|
||||
source ve3/bin/activate
|
||||
pip install -r requirements.txt
|
||||
./main.py --benchmark-name "mgbench" \
|
||||
--benchmark-results-path "../../tests/mgbench/benchmark_result.json" \
|
||||
--github-run-id "${{ github.run_id }}" \
|
||||
--github-run-number "${{ github.run_number }}" \
|
||||
--head-branch-name "${{ env.BRANCH_NAME }}"
|
||||
433
.github/workflows/diff.yaml
vendored
Normal file
433
.github/workflows/diff.yaml
vendored
Normal file
@@ -0,0 +1,433 @@
|
||||
name: Diff
|
||||
concurrency:
|
||||
group: ${{ github.head_ref || github.sha }}
|
||||
cancel-in-progress: true
|
||||
|
||||
on:
|
||||
push:
|
||||
branches:
|
||||
- master
|
||||
workflow_dispatch:
|
||||
pull_request:
|
||||
paths-ignore:
|
||||
- "docs/**"
|
||||
- "**/*.md"
|
||||
- ".clang-format"
|
||||
- "CODEOWNERS"
|
||||
- "licenses/*"
|
||||
|
||||
jobs:
|
||||
community_build:
|
||||
name: "Community build"
|
||||
runs-on: [self-hosted, Linux, X64, Diff]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build community binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build community binaries.
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=release -DMG_ENTERPRISE=OFF ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Run unit tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run unit tests.
|
||||
cd build
|
||||
ctest -R memgraph__unit --output-on-failure -j$THREADS
|
||||
|
||||
code_analysis:
|
||||
name: "Code analysis"
|
||||
runs-on: [self-hosted, Linux, X64, Diff]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
# This is also needed if we want do to comparison against other branches
|
||||
# See https://github.community/t/checkout-code-fails-when-it-runs-lerna-run-test-since-master/17920
|
||||
- name: Fetch all history for all tags and branches
|
||||
run: git fetch
|
||||
|
||||
- name: Initialize deps
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
- name: Set base branch
|
||||
if: ${{ github.event_name == 'pull_request' }}
|
||||
run: |
|
||||
echo "BASE_BRANCH=origin/${{ github.base_ref }}" >> $GITHUB_ENV
|
||||
|
||||
- name: Set base branch # if we manually dispatch or push to master
|
||||
if: ${{ github.event_name != 'pull_request' }}
|
||||
run: |
|
||||
echo "BASE_BRANCH=origin/master" >> $GITHUB_ENV
|
||||
|
||||
- name: Python code analysis
|
||||
run: |
|
||||
CHANGED_FILES=$(git diff -U0 ${{ env.BASE_BRANCH }}... --name-only)
|
||||
for file in ${CHANGED_FILES}; do
|
||||
echo ${file}
|
||||
if [[ ${file} == *.py ]]; then
|
||||
python3 -m black --check --diff ${file}
|
||||
python3 -m isort --check-only --diff ${file}
|
||||
fi
|
||||
done
|
||||
|
||||
- name: Build combined ASAN, UBSAN and coverage binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
cd build
|
||||
cmake -DTEST_COVERAGE=ON -DASAN=ON -DUBSAN=ON ..
|
||||
make -j$THREADS memgraph__unit
|
||||
|
||||
- name: Run unit tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run unit tests. It is restricted to 2 threads intentionally, because higher concurrency makes the timing related tests unstable.
|
||||
cd build
|
||||
LSAN_OPTIONS=suppressions=$PWD/../tools/lsan.supp UBSAN_OPTIONS=halt_on_error=1 ctest -R memgraph__unit --output-on-failure -j2
|
||||
|
||||
- name: Compute code coverage
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Compute code coverage.
|
||||
cd tools/github
|
||||
./coverage_convert
|
||||
|
||||
# Package code coverage.
|
||||
cd generated
|
||||
tar -czf code_coverage.tar.gz coverage.json html report.json summary.rmu
|
||||
|
||||
- name: Save code coverage
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "Code coverage"
|
||||
path: tools/github/generated/code_coverage.tar.gz
|
||||
|
||||
- name: Run clang-tidy
|
||||
run: |
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Restrict clang-tidy results only to the modified parts
|
||||
git diff -U0 ${{ env.BASE_BRANCH }}... -- src | ./tools/github/clang-tidy/clang-tidy-diff.py -p 1 -j $THREADS -path build -regex ".+\.cpp" | tee ./build/clang_tidy_output.txt
|
||||
|
||||
# Fail if any warning is reported
|
||||
! cat ./build/clang_tidy_output.txt | ./tools/github/clang-tidy/grep_error_lines.sh > /dev/null
|
||||
|
||||
debug_build:
|
||||
name: "Debug build"
|
||||
runs-on: [self-hosted, Linux, X64, Diff]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build debug binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build debug binaries.
|
||||
cd build
|
||||
cmake ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Run leftover CTest tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run leftover CTest tests (all except unit and benchmark tests).
|
||||
cd build
|
||||
ctest -E "(memgraph__unit|memgraph__benchmark)" --output-on-failure
|
||||
|
||||
- name: Run drivers tests
|
||||
run: |
|
||||
./tests/drivers/run.sh
|
||||
|
||||
- name: Run integration tests
|
||||
run: |
|
||||
tests/integration/run.sh
|
||||
|
||||
- name: Run cppcheck and clang-format
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run cppcheck and clang-format.
|
||||
cd tools/github
|
||||
./cppcheck_and_clang_format diff
|
||||
|
||||
- name: Save cppcheck and clang-format errors
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "Code coverage"
|
||||
path: tools/github/cppcheck_and_clang_format.txt
|
||||
|
||||
release_build:
|
||||
name: "Release build"
|
||||
runs-on: [self-hosted, Linux, X64, Diff]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build release binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build release binaries.
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=release ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Run GQL Behave tests
|
||||
run: |
|
||||
cd tests/gql_behave
|
||||
./continuous_integration
|
||||
|
||||
- name: Save quality assurance status
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "GQL Behave Status"
|
||||
path: |
|
||||
tests/gql_behave/gql_behave_status.csv
|
||||
tests/gql_behave/gql_behave_status.html
|
||||
|
||||
- name: Run unit tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run unit tests.
|
||||
cd build
|
||||
ctest -R memgraph__unit --output-on-failure -j$THREADS
|
||||
|
||||
- name: Run e2e tests
|
||||
run: |
|
||||
cd tests
|
||||
./setup.sh /opt/toolchain-v4/activate
|
||||
source ve3/bin/activate_e2e
|
||||
cd e2e
|
||||
./run.sh
|
||||
|
||||
- name: Run stress test (plain)
|
||||
run: |
|
||||
cd tests/stress
|
||||
./continuous_integration
|
||||
|
||||
- name: Run stress test (SSL)
|
||||
run: |
|
||||
cd tests/stress
|
||||
./continuous_integration --use-ssl
|
||||
|
||||
- name: Run durability test
|
||||
run: |
|
||||
cd tests/stress
|
||||
source ve3/bin/activate
|
||||
python3 durability --num-steps 5
|
||||
|
||||
- name: Create enterprise DEB package
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
cd build
|
||||
|
||||
# create mgconsole
|
||||
# we use the -B to force the build
|
||||
make -j$THREADS -B mgconsole
|
||||
|
||||
# Create enterprise DEB package.
|
||||
mkdir output && cd output
|
||||
cpack -G DEB --config ../CPackConfig.cmake
|
||||
|
||||
- name: Save enterprise DEB package
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "Enterprise DEB package"
|
||||
path: build/output/memgraph*.deb
|
||||
|
||||
- name: Save test data
|
||||
uses: actions/upload-artifact@v3
|
||||
if: always()
|
||||
with:
|
||||
name: "Test data"
|
||||
path: |
|
||||
# multiple paths could be defined
|
||||
build/logs
|
||||
|
||||
release_jepsen_test:
|
||||
name: "Release Jepsen Test"
|
||||
runs-on: [self-hosted, Linux, X64, Debian10, JepsenControl]
|
||||
#continue-on-error: true
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build release binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
# Build only memgraph release binarie.
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=release ..
|
||||
make -j$THREADS memgraph
|
||||
|
||||
- name: Run Jepsen tests
|
||||
run: |
|
||||
cd tests/jepsen
|
||||
./run.sh test-all-individually --binary ../../build/memgraph --ignore-run-stdout-logs --ignore-run-stderr-logs
|
||||
|
||||
- name: Save Jepsen report
|
||||
uses: actions/upload-artifact@v3
|
||||
if: ${{ always() }}
|
||||
with:
|
||||
name: "Jepsen Report"
|
||||
path: tests/jepsen/Jepsen.tar.gz
|
||||
|
||||
release_benchmarks:
|
||||
name: "Release benchmarks"
|
||||
runs-on: [self-hosted, Linux, X64, Diff, Gen7]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build release binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build only memgraph release binaries.
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=release ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Run macro benchmarks
|
||||
run: |
|
||||
cd tests/macro_benchmark
|
||||
./harness QuerySuite MemgraphRunner \
|
||||
--groups aggregation 1000_create unwind_create dense_expand match \
|
||||
--no-strict
|
||||
|
||||
- name: Get branch name (merge)
|
||||
if: github.event_name != 'pull_request'
|
||||
shell: bash
|
||||
run: echo "BRANCH_NAME=$(echo ${GITHUB_REF#refs/heads/} | tr / -)" >> $GITHUB_ENV
|
||||
|
||||
- name: Get branch name (pull request)
|
||||
if: github.event_name == 'pull_request'
|
||||
shell: bash
|
||||
run: echo "BRANCH_NAME=$(echo ${GITHUB_HEAD_REF} | tr / -)" >> $GITHUB_ENV
|
||||
|
||||
- name: Upload macro benchmark results
|
||||
run: |
|
||||
cd tools/bench-graph-client
|
||||
virtualenv -p python3 ve3
|
||||
source ve3/bin/activate
|
||||
pip install -r requirements.txt
|
||||
./main.py --benchmark-name "macro_benchmark" \
|
||||
--benchmark-results-path "../../tests/macro_benchmark/.harness_summary" \
|
||||
--github-run-id "${{ github.run_id }}" \
|
||||
--github-run-number "${{ github.run_number }}" \
|
||||
--head-branch-name "${{ env.BRANCH_NAME }}"
|
||||
|
||||
- name: Run mgbench
|
||||
run: |
|
||||
cd tests/mgbench
|
||||
./benchmark.py vendor-native --num-workers-for-benchmark 12 --export-results benchmark_result.json pokec/medium/*/*
|
||||
|
||||
- name: Upload mgbench results
|
||||
run: |
|
||||
cd tools/bench-graph-client
|
||||
virtualenv -p python3 ve3
|
||||
source ve3/bin/activate
|
||||
pip install -r requirements.txt
|
||||
./main.py --benchmark-name "mgbench" \
|
||||
--benchmark-results-path "../../tests/mgbench/benchmark_result.json" \
|
||||
--github-run-id "${{ github.run_id }}" \
|
||||
--github-run-number "${{ github.run_number }}" \
|
||||
--head-branch-name "${{ env.BRANCH_NAME }}"
|
||||
46
.github/workflows/full_clang_tidy.yaml
vendored
Normal file
46
.github/workflows/full_clang_tidy.yaml
vendored
Normal file
@@ -0,0 +1,46 @@
|
||||
name: Run clang-tidy on the full codebase
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
|
||||
jobs:
|
||||
clang_tidy_check:
|
||||
name: "Clang-tidy check"
|
||||
runs-on: [self-hosted, Linux, X64, Ubuntu20.04]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build debug binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build debug binaries.
|
||||
|
||||
cd build
|
||||
cmake ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Run clang-tidy
|
||||
run: |
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# The results are also written to standard output in order to retain them in the logs
|
||||
./tools/github/clang-tidy/run-clang-tidy.py -p build -j $THREADS -clang-tidy-binary=/opt/toolchain-v4/bin/clang-tidy "$PWD/src/*" |
|
||||
tee ./build/full_clang_tidy_output.txt
|
||||
|
||||
- name: Summarize clang-tidy results
|
||||
run: cat ./build/full_clang_tidy_output.txt | ./tools/github/clang-tidy/count_errors.sh
|
||||
256
.github/workflows/package_all.yaml
vendored
Normal file
256
.github/workflows/package_all.yaml
vendored
Normal file
@@ -0,0 +1,256 @@
|
||||
name: Package All
|
||||
|
||||
# TODO(gitbuda): Cleanup docker container if GHA job was canceled.
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
memgraph_version:
|
||||
description: "Memgraph version to upload as. If empty upload is skipped. Format: 'X.Y.Z'"
|
||||
required: false
|
||||
|
||||
jobs:
|
||||
centos-7:
|
||||
runs-on: [self-hosted, DockerMgBuild, X64]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
./release/package/run.sh package centos-7
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: centos-7
|
||||
path: build/output/centos-7/memgraph*.rpm
|
||||
|
||||
centos-9:
|
||||
runs-on: [self-hosted, DockerMgBuild, X64]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
./release/package/run.sh package centos-9
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: centos-9
|
||||
path: build/output/centos-9/memgraph*.rpm
|
||||
|
||||
debian-10:
|
||||
runs-on: [self-hosted, DockerMgBuild, X64]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
./release/package/run.sh package debian-10
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: debian-10
|
||||
path: build/output/debian-10/memgraph*.deb
|
||||
|
||||
debian-11:
|
||||
runs-on: [self-hosted, DockerMgBuild, X64]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
./release/package/run.sh package debian-11
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: debian-11
|
||||
path: build/output/debian-11/memgraph*.deb
|
||||
|
||||
docker:
|
||||
runs-on: [self-hosted, DockerMgBuild, X64]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
cd release/package
|
||||
./run.sh package debian-11 --for-docker
|
||||
./run.sh docker
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: docker
|
||||
path: build/output/docker/memgraph*.tar.gz
|
||||
|
||||
ubuntu-1804:
|
||||
runs-on: [self-hosted, DockerMgBuild, X64]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
./release/package/run.sh package ubuntu-18.04
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: ubuntu-18.04
|
||||
path: build/output/ubuntu-18.04/memgraph*.deb
|
||||
|
||||
ubuntu-2004:
|
||||
runs-on: [self-hosted, DockerMgBuild, X64]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
./release/package/run.sh package ubuntu-20.04
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: ubuntu-20.04
|
||||
path: build/output/ubuntu-20.04/memgraph*.deb
|
||||
|
||||
ubuntu-2204:
|
||||
runs-on: [self-hosted, DockerMgBuild, X64]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
./release/package/run.sh package ubuntu-22.04
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: ubuntu-22.04
|
||||
path: build/output/ubuntu-22.04/memgraph*.deb
|
||||
|
||||
debian-11-platform:
|
||||
runs-on: [self-hosted, DockerMgBuild, X64]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
./release/package/run.sh package debian-11 --for-platform
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: debian-11-platform
|
||||
path: build/output/debian-11/memgraph*.deb
|
||||
|
||||
fedora-36:
|
||||
runs-on: [self-hosted, DockerMgBuild, X64]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
./release/package/run.sh package fedora-36
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: fedora-36
|
||||
path: build/output/fedora-36/memgraph*.rpm
|
||||
|
||||
amzn-2:
|
||||
runs-on: [self-hosted, DockerMgBuild, X64]
|
||||
timeout-minutes: 60
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
./release/package/run.sh package amzn-2
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: amzn-2
|
||||
path: build/output/amzn-2/memgraph*.rpm
|
||||
|
||||
debian-11-arm:
|
||||
runs-on: [self-hosted, DockerMgBuild, ARM64, strange]
|
||||
timeout-minutes: 120
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
./release/package/run.sh package debian-11-arm
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: debian-11-aarch64
|
||||
path: build/output/debian-11-arm/memgraph*.deb
|
||||
|
||||
ubuntu-2204-arm:
|
||||
runs-on: [self-hosted, DockerMgBuild, ARM64, strange]
|
||||
timeout-minutes: 120
|
||||
steps:
|
||||
- name: "Set up repository"
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
fetch-depth: 0 # Required because of release/get_version.py
|
||||
- name: "Build package"
|
||||
run: |
|
||||
./release/package/run.sh package ubuntu-22.04-arm
|
||||
- name: "Upload package"
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: ubuntu-22.04-aarch64
|
||||
path: build/output/ubuntu-22.04-arm/memgraph*.deb
|
||||
|
||||
upload-to-s3:
|
||||
# only run upload if we specified version. Allows for runs without upload
|
||||
if: "${{ github.event.inputs.memgraph_version != '' }}"
|
||||
needs: [centos-7, centos-9, debian-10, debian-11, docker, ubuntu-1804, ubuntu-2004, ubuntu-2204, debian-11-platform, fedora-36, amzn-2, debian-11-arm, ubuntu-2204-arm]
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Download artifacts
|
||||
uses: actions/download-artifact@v3
|
||||
with:
|
||||
# name: # if name input parameter is not provided, all artifacts are downloaded
|
||||
# and put in directories named after each one.
|
||||
path: build/output/release
|
||||
- name: Upload to S3
|
||||
uses: jakejarvis/s3-sync-action@v0.5.1
|
||||
env:
|
||||
AWS_S3_BUCKET: "download.memgraph.com"
|
||||
AWS_ACCESS_KEY_ID: ${{ secrets.S3_AWS_ACCESS_KEY_ID }}
|
||||
AWS_SECRET_ACCESS_KEY: ${{ secrets.S3_AWS_SECRET_ACCESS_KEY }}
|
||||
AWS_REGION: "eu-west-1"
|
||||
SOURCE_DIR: "build/output/release"
|
||||
DEST_DIR: "memgraph/v${{ github.event.inputs.memgraph_version }}/"
|
||||
299
.github/workflows/release_centos8.yaml
vendored
Normal file
299
.github/workflows/release_centos8.yaml
vendored
Normal file
@@ -0,0 +1,299 @@
|
||||
name: Release CentOS 8
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
schedule:
|
||||
- cron: "0 22 * * *"
|
||||
|
||||
jobs:
|
||||
community_build:
|
||||
name: "Community build"
|
||||
runs-on: [self-hosted, Linux, X64, CentOS8]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
timeout-minutes: 960
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build community binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build community binaries.
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=release -DMG_ENTERPRISE=OFF ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Run unit tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run unit tests.
|
||||
cd build
|
||||
ctest -R memgraph__unit --output-on-failure
|
||||
|
||||
coverage_build:
|
||||
name: "Coverage build"
|
||||
runs-on: [self-hosted, Linux, X64, CentOS8]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build coverage binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build coverage binaries.
|
||||
cd build
|
||||
cmake -DTEST_COVERAGE=ON ..
|
||||
make -j$THREADS memgraph__unit
|
||||
|
||||
- name: Run unit tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run unit tests.
|
||||
cd build
|
||||
ctest -R memgraph__unit --output-on-failure
|
||||
|
||||
- name: Compute code coverage
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Compute code coverage.
|
||||
cd tools/github
|
||||
./coverage_convert
|
||||
|
||||
# Package code coverage.
|
||||
cd generated
|
||||
tar -czf code_coverage.tar.gz coverage.json html report.json summary.rmu
|
||||
|
||||
- name: Save code coverage
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "Code coverage"
|
||||
path: tools/github/generated/code_coverage.tar.gz
|
||||
|
||||
debug_build:
|
||||
name: "Debug build"
|
||||
runs-on: [self-hosted, Linux, X64, CentOS8]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build debug binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build debug binaries.
|
||||
cd build
|
||||
cmake ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Run leftover CTest tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run leftover CTest tests (all except unit and benchmark tests).
|
||||
cd build
|
||||
ctest -E "(memgraph__unit|memgraph__benchmark)" --output-on-failure
|
||||
|
||||
- name: Run drivers tests
|
||||
run: |
|
||||
./tests/drivers/run.sh
|
||||
|
||||
- name: Run integration tests
|
||||
run: |
|
||||
tests/integration/run.sh
|
||||
|
||||
- name: Run cppcheck and clang-format
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run cppcheck and clang-format.
|
||||
cd tools/github
|
||||
./cppcheck_and_clang_format diff
|
||||
|
||||
- name: Save cppcheck and clang-format errors
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "Code coverage"
|
||||
path: tools/github/cppcheck_and_clang_format.txt
|
||||
|
||||
release_build:
|
||||
name: "Release build"
|
||||
runs-on: [self-hosted, Linux, X64, CentOS8]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
timeout-minutes: 960
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build release binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build release binaries.
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=release ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Create enterprise RPM package
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
cd build
|
||||
|
||||
# create mgconsole
|
||||
# we use the -B to force the build
|
||||
make -j$THREADS -B mgconsole
|
||||
|
||||
# Create enterprise RPM package.
|
||||
mkdir output && cd output
|
||||
cpack -G RPM --config ../CPackConfig.cmake
|
||||
rpmlint memgraph*.rpm
|
||||
|
||||
- name: Save enterprise RPM package
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "Enterprise RPM package"
|
||||
path: build/output/memgraph*.rpm
|
||||
|
||||
- name: Run micro benchmark tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run micro benchmark tests.
|
||||
cd build
|
||||
# The `eval` benchmark needs a large stack limit.
|
||||
ulimit -s 262144
|
||||
ctest -R memgraph__benchmark -V
|
||||
|
||||
- name: Run macro benchmark tests
|
||||
run: |
|
||||
cd tests/macro_benchmark
|
||||
./harness QuerySuite MemgraphRunner \
|
||||
--groups aggregation 1000_create unwind_create dense_expand match \
|
||||
--no-strict
|
||||
|
||||
- name: Run parallel macro benchmark tests
|
||||
run: |
|
||||
cd tests/macro_benchmark
|
||||
./harness QueryParallelSuite MemgraphRunner \
|
||||
--groups aggregation_parallel create_parallel bfs_parallel \
|
||||
--num-database-workers 9 --num-clients-workers 30 \
|
||||
--no-strict
|
||||
|
||||
- name: Run GQL Behave tests
|
||||
run: |
|
||||
cd tests/gql_behave
|
||||
./continuous_integration
|
||||
|
||||
- name: Save quality assurance status
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "GQL Behave Status"
|
||||
path: |
|
||||
tests/gql_behave/gql_behave_status.csv
|
||||
tests/gql_behave/gql_behave_status.html
|
||||
|
||||
- name: Run unit tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run unit tests.
|
||||
cd build
|
||||
ctest -R memgraph__unit --output-on-failure
|
||||
|
||||
- name: Run e2e tests
|
||||
run: |
|
||||
cd tests
|
||||
./setup.sh /opt/toolchain-v4/activate
|
||||
source ve3/bin/activate_e2e
|
||||
cd e2e
|
||||
./run.sh
|
||||
|
||||
- name: Run stress test (plain)
|
||||
run: |
|
||||
cd tests/stress
|
||||
./continuous_integration
|
||||
|
||||
- name: Run stress test (SSL)
|
||||
run: |
|
||||
cd tests/stress
|
||||
./continuous_integration --use-ssl
|
||||
|
||||
- name: Run stress test (large)
|
||||
run: |
|
||||
cd tests/stress
|
||||
./continuous_integration --large-dataset
|
||||
|
||||
- name: Run durability test (plain)
|
||||
run: |
|
||||
cd tests/stress
|
||||
source ve3/bin/activate
|
||||
python3 durability --num-steps 5
|
||||
|
||||
- name: Run durability test (large)
|
||||
run: |
|
||||
cd tests/stress
|
||||
source ve3/bin/activate
|
||||
python3 durability --num-steps 20
|
||||
338
.github/workflows/release_debian10.yaml
vendored
Normal file
338
.github/workflows/release_debian10.yaml
vendored
Normal file
@@ -0,0 +1,338 @@
|
||||
name: Release Debian 10
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
schedule:
|
||||
- cron: "0 22 * * *"
|
||||
|
||||
jobs:
|
||||
community_build:
|
||||
name: "Community build"
|
||||
runs-on: [self-hosted, Linux, X64, Debian10]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
timeout-minutes: 960
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build community binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build community binaries.
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=release -DMG_ENTERPRISE=OFF ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Run unit tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run unit tests.
|
||||
cd build
|
||||
ctest -R memgraph__unit --output-on-failure
|
||||
|
||||
coverage_build:
|
||||
name: "Coverage build"
|
||||
runs-on: [self-hosted, Linux, X64, Debian10]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build coverage binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build coverage binaries.
|
||||
cd build
|
||||
cmake -DTEST_COVERAGE=ON ..
|
||||
make -j$THREADS memgraph__unit
|
||||
|
||||
- name: Run unit tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run unit tests.
|
||||
cd build
|
||||
ctest -R memgraph__unit --output-on-failure
|
||||
|
||||
- name: Compute code coverage
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Compute code coverage.
|
||||
cd tools/github
|
||||
./coverage_convert
|
||||
|
||||
# Package code coverage.
|
||||
cd generated
|
||||
tar -czf code_coverage.tar.gz coverage.json html report.json summary.rmu
|
||||
|
||||
- name: Save code coverage
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "Code coverage"
|
||||
path: tools/github/generated/code_coverage.tar.gz
|
||||
|
||||
debug_build:
|
||||
name: "Debug build"
|
||||
runs-on: [self-hosted, Linux, X64, Debian10]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build debug binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build debug binaries.
|
||||
cd build
|
||||
cmake ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Run leftover CTest tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run leftover CTest tests (all except unit and benchmark tests).
|
||||
cd build
|
||||
ctest -E "(memgraph__unit|memgraph__benchmark)" --output-on-failure
|
||||
|
||||
- name: Run drivers tests
|
||||
run: |
|
||||
./tests/drivers/run.sh
|
||||
|
||||
- name: Run integration tests
|
||||
run: |
|
||||
tests/integration/run.sh
|
||||
|
||||
- name: Run cppcheck and clang-format
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run cppcheck and clang-format.
|
||||
cd tools/github
|
||||
./cppcheck_and_clang_format diff
|
||||
|
||||
- name: Save cppcheck and clang-format errors
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "Code coverage"
|
||||
path: tools/github/cppcheck_and_clang_format.txt
|
||||
|
||||
release_build:
|
||||
name: "Release build"
|
||||
runs-on: [self-hosted, Linux, X64, Debian10]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
timeout-minutes: 960
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build release binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build release binaries.
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=release ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Create enterprise DEB package
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
cd build
|
||||
|
||||
# create mgconsole
|
||||
# we use the -B to force the build
|
||||
make -j$THREADS -B mgconsole
|
||||
|
||||
# Create enterprise DEB package.
|
||||
mkdir output && cd output
|
||||
cpack -G DEB --config ../CPackConfig.cmake
|
||||
|
||||
- name: Save enterprise DEB package
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "Enterprise DEB package"
|
||||
path: build/output/memgraph*.deb
|
||||
|
||||
- name: Run micro benchmark tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run micro benchmark tests.
|
||||
cd build
|
||||
# The `eval` benchmark needs a large stack limit.
|
||||
ulimit -s 262144
|
||||
ctest -R memgraph__benchmark -V
|
||||
|
||||
- name: Run macro benchmark tests
|
||||
run: |
|
||||
cd tests/macro_benchmark
|
||||
./harness QuerySuite MemgraphRunner \
|
||||
--groups aggregation 1000_create unwind_create dense_expand match \
|
||||
--no-strict
|
||||
|
||||
- name: Run parallel macro benchmark tests
|
||||
run: |
|
||||
cd tests/macro_benchmark
|
||||
./harness QueryParallelSuite MemgraphRunner \
|
||||
--groups aggregation_parallel create_parallel bfs_parallel \
|
||||
--num-database-workers 9 --num-clients-workers 30 \
|
||||
--no-strict
|
||||
|
||||
- name: Run GQL Behave tests
|
||||
run: |
|
||||
cd tests/gql_behave
|
||||
./continuous_integration
|
||||
|
||||
- name: Save quality assurance status
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "GQL Behave Status"
|
||||
path: |
|
||||
tests/gql_behave/gql_behave_status.csv
|
||||
tests/gql_behave/gql_behave_status.html
|
||||
|
||||
- name: Run unit tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run unit tests.
|
||||
cd build
|
||||
ctest -R memgraph__unit --output-on-failure
|
||||
|
||||
- name: Run e2e tests
|
||||
run: |
|
||||
cd tests
|
||||
./setup.sh /opt/toolchain-v4/activate
|
||||
source ve3/bin/activate_e2e
|
||||
cd e2e
|
||||
./run.sh
|
||||
|
||||
- name: Run stress test (plain)
|
||||
run: |
|
||||
cd tests/stress
|
||||
./continuous_integration
|
||||
|
||||
- name: Run stress test (SSL)
|
||||
run: |
|
||||
cd tests/stress
|
||||
./continuous_integration --use-ssl
|
||||
|
||||
- name: Run stress test (large)
|
||||
run: |
|
||||
cd tests/stress
|
||||
./continuous_integration --large-dataset
|
||||
|
||||
- name: Run durability test (plain)
|
||||
run: |
|
||||
cd tests/stress
|
||||
source ve3/bin/activate
|
||||
python3 durability --num-steps 5
|
||||
|
||||
- name: Run durability test (large)
|
||||
run: |
|
||||
cd tests/stress
|
||||
source ve3/bin/activate
|
||||
python3 durability --num-steps 20
|
||||
|
||||
release_jepsen_test:
|
||||
name: "Release Jepsen Test"
|
||||
runs-on: [self-hosted, Linux, X64, Debian10, JepsenControl]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
timeout-minutes: 60
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build release binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
# Build only memgraph release binary.
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=release ..
|
||||
make -j$THREADS memgraph
|
||||
|
||||
- name: Run Jepsen tests
|
||||
run: |
|
||||
cd tests/jepsen
|
||||
./run.sh test-all-individually --binary ../../build/memgraph --ignore-run-stdout-logs --ignore-run-stderr-logs
|
||||
|
||||
- name: Save Jepsen report
|
||||
uses: actions/upload-artifact@v3
|
||||
if: ${{ always() }}
|
||||
with:
|
||||
name: "Jepsen Report"
|
||||
path: tests/jepsen/Jepsen.tar.gz
|
||||
69
.github/workflows/release_docker.yaml
vendored
Normal file
69
.github/workflows/release_docker.yaml
vendored
Normal file
@@ -0,0 +1,69 @@
|
||||
name: Publish Docker images
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
version:
|
||||
description: "Memgraph binary version to publish on DockerHub."
|
||||
required: true
|
||||
force_release:
|
||||
type: boolean
|
||||
required: false
|
||||
default: false
|
||||
|
||||
jobs:
|
||||
docker_publish:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
DOCKER_ORGANIZATION_NAME: memgraph
|
||||
DOCKER_REPOSITORY_NAME: memgraph
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v3
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v2
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
uses: docker/setup-buildx-action@v2
|
||||
|
||||
- name: Log in to Docker Hub
|
||||
uses: docker/login-action@v2
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Download memgraph binary
|
||||
run: |
|
||||
cd release/docker
|
||||
curl -L https://download.memgraph.com/memgraph/v${{ github.event.inputs.version }}/debian-11/memgraph_${{ github.event.inputs.version }}-1_amd64.deb > memgraph-amd64.deb
|
||||
curl -L https://download.memgraph.com/memgraph/v${{ github.event.inputs.version }}/debian-11-aarch64/memgraph_${{ github.event.inputs.version }}-1_arm64.deb > memgraph-arm64.deb
|
||||
|
||||
- name: Check if specified version is already pushed
|
||||
run: |
|
||||
EXISTS=$(docker manifest inspect $DOCKER_ORGANIZATION_NAME/$DOCKER_REPOSITORY_NAME:${{ github.event.inputs.version }} > /dev/null; echo $?)
|
||||
echo $EXISTS
|
||||
if [[ ${EXISTS} -eq 0 ]]; then
|
||||
echo 'The specified version has been already released to DockerHub.'
|
||||
if [[ ${{ github.event.inputs.force_release }} = true ]]; then
|
||||
echo 'Forcing the release!'
|
||||
else
|
||||
echo 'Stopping the release!'
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
echo 'All good the specified version has not been release to DockerHub.'
|
||||
fi
|
||||
|
||||
- name: Build & push docker images
|
||||
run: |
|
||||
cd release/docker
|
||||
docker buildx build \
|
||||
--build-arg BINARY_NAME="memgraph-" \
|
||||
--build-arg EXTENSION="deb" \
|
||||
--platform linux/amd64,linux/arm64 \
|
||||
--tag $DOCKER_ORGANIZATION_NAME/$DOCKER_REPOSITORY_NAME:${{ github.event.inputs.version }} \
|
||||
--tag $DOCKER_ORGANIZATION_NAME/$DOCKER_REPOSITORY_NAME:latest \
|
||||
--file memgraph_deb.dockerfile \
|
||||
--push .
|
||||
63
.github/workflows/release_mgbench_client.yaml
vendored
Normal file
63
.github/workflows/release_mgbench_client.yaml
vendored
Normal file
@@ -0,0 +1,63 @@
|
||||
name: "Mgbench Bolt Client Publish Docker Image"
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
version:
|
||||
description: "Mgbench bolt client version to publish on Dockerhub."
|
||||
required: true
|
||||
force_release:
|
||||
type: boolean
|
||||
required: false
|
||||
default: false
|
||||
|
||||
jobs:
|
||||
mgbench_docker_publish:
|
||||
runs-on: ubuntu-latest
|
||||
env:
|
||||
DOCKER_ORGANIZATION_NAME: memgraph
|
||||
DOCKER_REPOSITORY_NAME: mgbench-client
|
||||
|
||||
steps:
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v3
|
||||
|
||||
- name: Set up QEMU
|
||||
uses: docker/setup-qemu-action@v2
|
||||
|
||||
- name: Set up Docker Buildx
|
||||
id: buildx
|
||||
uses: docker/setup-buildx-action@v2
|
||||
|
||||
- name: Log in to Docker Hub
|
||||
uses: docker/login-action@v2
|
||||
with:
|
||||
username: ${{ secrets.DOCKERHUB_USERNAME }}
|
||||
password: ${{ secrets.DOCKERHUB_TOKEN }}
|
||||
|
||||
- name: Check if specified version is already pushed
|
||||
run: |
|
||||
EXISTS=$(docker manifest inspect $DOCKER_ORGANIZATION_NAME/$DOCKER_REPOSITORY_NAME:${{ github.event.inputs.version }} > /dev/null; echo $?)
|
||||
echo $EXISTS
|
||||
if [[ ${EXISTS} -eq 0 ]]; then
|
||||
echo 'The specified version has been already released to DockerHub.'
|
||||
if [[ ${{ github.event.inputs.force_release }} = true ]]; then
|
||||
echo 'Forcing the release!'
|
||||
else
|
||||
echo 'Stopping the release!'
|
||||
exit 1
|
||||
fi
|
||||
else
|
||||
echo 'All good the specified version has not been release to DockerHub.'
|
||||
fi
|
||||
|
||||
- name: Build & push docker images
|
||||
run: |
|
||||
cd tests/mgbench
|
||||
docker buildx build \
|
||||
--build-arg TOOLCHAIN_VERSION=toolchain-v4 \
|
||||
--platform linux/amd64,linux/arm64 \
|
||||
--tag $DOCKER_ORGANIZATION_NAME/$DOCKER_REPOSITORY_NAME:${{ github.event.inputs.version }} \
|
||||
--tag $DOCKER_ORGANIZATION_NAME/$DOCKER_REPOSITORY_NAME:latest \
|
||||
--file Dockerfile.mgbench_client \
|
||||
--push .
|
||||
298
.github/workflows/release_ubuntu2004.yaml
vendored
Normal file
298
.github/workflows/release_ubuntu2004.yaml
vendored
Normal file
@@ -0,0 +1,298 @@
|
||||
name: Release Ubuntu 20.04
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
schedule:
|
||||
- cron: "0 22 * * *"
|
||||
|
||||
jobs:
|
||||
community_build:
|
||||
name: "Community build"
|
||||
runs-on: [self-hosted, Linux, X64, Ubuntu20.04]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
timeout-minutes: 960
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build community binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build community binaries.
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=release -DMG_ENTERPRISE=OFF ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Run unit tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run unit tests.
|
||||
cd build
|
||||
ctest -R memgraph__unit --output-on-failure
|
||||
|
||||
coverage_build:
|
||||
name: "Coverage build"
|
||||
runs-on: [self-hosted, Linux, X64, Ubuntu20.04]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build coverage binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build coverage binaries.
|
||||
cd build
|
||||
cmake -DTEST_COVERAGE=ON ..
|
||||
make -j$THREADS memgraph__unit
|
||||
|
||||
- name: Run unit tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run unit tests.
|
||||
cd build
|
||||
ctest -R memgraph__unit --output-on-failure
|
||||
|
||||
- name: Compute code coverage
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Compute code coverage.
|
||||
cd tools/github
|
||||
./coverage_convert
|
||||
|
||||
# Package code coverage.
|
||||
cd generated
|
||||
tar -czf code_coverage.tar.gz coverage.json html report.json summary.rmu
|
||||
|
||||
- name: Save code coverage
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "Code coverage"
|
||||
path: tools/github/generated/code_coverage.tar.gz
|
||||
|
||||
debug_build:
|
||||
name: "Debug build"
|
||||
runs-on: [self-hosted, Linux, X64, Ubuntu20.04]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build debug binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build debug binaries.
|
||||
cd build
|
||||
cmake ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Run leftover CTest tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run leftover CTest tests (all except unit and benchmark tests).
|
||||
cd build
|
||||
ctest -E "(memgraph__unit|memgraph__benchmark)" --output-on-failure
|
||||
|
||||
- name: Run drivers tests
|
||||
run: |
|
||||
./tests/drivers/run.sh
|
||||
|
||||
- name: Run integration tests
|
||||
run: |
|
||||
tests/integration/run.sh
|
||||
|
||||
- name: Run cppcheck and clang-format
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run cppcheck and clang-format.
|
||||
cd tools/github
|
||||
./cppcheck_and_clang_format diff
|
||||
|
||||
- name: Save cppcheck and clang-format errors
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "Code coverage"
|
||||
path: tools/github/cppcheck_and_clang_format.txt
|
||||
|
||||
release_build:
|
||||
name: "Release build"
|
||||
runs-on: [self-hosted, Linux, X64, Ubuntu20.04]
|
||||
env:
|
||||
THREADS: 24
|
||||
MEMGRAPH_ENTERPRISE_LICENSE: ${{ secrets.MEMGRAPH_ENTERPRISE_LICENSE }}
|
||||
MEMGRAPH_ORGANIZATION_NAME: ${{ secrets.MEMGRAPH_ORGANIZATION_NAME }}
|
||||
timeout-minutes: 960
|
||||
|
||||
steps:
|
||||
- name: Set up repository
|
||||
uses: actions/checkout@v3
|
||||
with:
|
||||
# Number of commits to fetch. `0` indicates all history for all
|
||||
# branches and tags. (default: 1)
|
||||
fetch-depth: 0
|
||||
|
||||
- name: Build release binaries
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Initialize dependencies.
|
||||
./init
|
||||
|
||||
# Build release binaries.
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE=release ..
|
||||
make -j$THREADS
|
||||
|
||||
- name: Create enterprise DEB package
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
cd build
|
||||
|
||||
# create mgconsole
|
||||
# we use the -B to force the build
|
||||
make -j$THREADS -B mgconsole
|
||||
|
||||
# Create enterprise DEB package.
|
||||
mkdir output && cd output
|
||||
cpack -G DEB --config ../CPackConfig.cmake
|
||||
|
||||
- name: Save enterprise DEB package
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "Enterprise DEB package"
|
||||
path: build/output/memgraph*.deb
|
||||
|
||||
- name: Run micro benchmark tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run micro benchmark tests.
|
||||
cd build
|
||||
# The `eval` benchmark needs a large stack limit.
|
||||
ulimit -s 262144
|
||||
ctest -R memgraph__benchmark -V
|
||||
|
||||
- name: Run macro benchmark tests
|
||||
run: |
|
||||
cd tests/macro_benchmark
|
||||
./harness QuerySuite MemgraphRunner \
|
||||
--groups aggregation 1000_create unwind_create dense_expand match \
|
||||
--no-strict
|
||||
|
||||
- name: Run parallel macro benchmark tests
|
||||
run: |
|
||||
cd tests/macro_benchmark
|
||||
./harness QueryParallelSuite MemgraphRunner \
|
||||
--groups aggregation_parallel create_parallel bfs_parallel \
|
||||
--num-database-workers 9 --num-clients-workers 30 \
|
||||
--no-strict
|
||||
|
||||
- name: Run GQL Behave tests
|
||||
run: |
|
||||
cd tests/gql_behave
|
||||
./continuous_integration
|
||||
|
||||
- name: Save quality assurance status
|
||||
uses: actions/upload-artifact@v3
|
||||
with:
|
||||
name: "GQL Behave Status"
|
||||
path: |
|
||||
tests/gql_behave/gql_behave_status.csv
|
||||
tests/gql_behave/gql_behave_status.html
|
||||
|
||||
- name: Run unit tests
|
||||
run: |
|
||||
# Activate toolchain.
|
||||
source /opt/toolchain-v4/activate
|
||||
|
||||
# Run unit tests.
|
||||
cd build
|
||||
ctest -R memgraph__unit --output-on-failure
|
||||
|
||||
- name: Run e2e tests
|
||||
run: |
|
||||
cd tests
|
||||
./setup.sh /opt/toolchain-v4/activate
|
||||
source ve3/bin/activate_e2e
|
||||
cd e2e
|
||||
./run.sh
|
||||
|
||||
- name: Run stress test (plain)
|
||||
run: |
|
||||
cd tests/stress
|
||||
./continuous_integration
|
||||
|
||||
- name: Run stress test (SSL)
|
||||
run: |
|
||||
cd tests/stress
|
||||
./continuous_integration --use-ssl
|
||||
|
||||
- name: Run stress test (large)
|
||||
run: |
|
||||
cd tests/stress
|
||||
./continuous_integration --large-dataset
|
||||
|
||||
- name: Run durability test (plain)
|
||||
run: |
|
||||
cd tests/stress
|
||||
source ve3/bin/activate
|
||||
python3 durability --num-steps 5
|
||||
|
||||
- name: Run durability test (large)
|
||||
run: |
|
||||
cd tests/stress
|
||||
source ve3/bin/activate
|
||||
python3 durability --num-steps 20
|
||||
32
.github/workflows/upload_to_s3.yaml
vendored
Normal file
32
.github/workflows/upload_to_s3.yaml
vendored
Normal file
@@ -0,0 +1,32 @@
|
||||
name: Upload Package All artifacts to S3
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
memgraph_version:
|
||||
description: "Memgraph version to upload as. Format: 'X.Y.Z'"
|
||||
required: true
|
||||
run_number:
|
||||
description: "# of the package_all workflow run to upload artifacts from. Format: '#XYZ'"
|
||||
required: true
|
||||
|
||||
jobs:
|
||||
upload-to-s3:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Download artifacts
|
||||
uses: dawidd6/action-download-artifact@v2
|
||||
with:
|
||||
workflow: package_all.yaml
|
||||
workflow_conclusion: success
|
||||
run_number: "${{ github.event.inputs.run_number }}"
|
||||
path: build/output/release
|
||||
- name: Upload to S3
|
||||
uses: jakejarvis/s3-sync-action@v0.5.1
|
||||
env:
|
||||
AWS_S3_BUCKET: "download.memgraph.com"
|
||||
AWS_ACCESS_KEY_ID: ${{ secrets.S3_AWS_ACCESS_KEY_ID }}
|
||||
AWS_SECRET_ACCESS_KEY: ${{ secrets.S3_AWS_SECRET_ACCESS_KEY }}
|
||||
AWS_REGION: "eu-west-1"
|
||||
SOURCE_DIR: "build/output/release"
|
||||
DEST_DIR: "memgraph/v${{ github.event.inputs.memgraph_version }}/"
|
||||
9
.gitignore
vendored
9
.gitignore
vendored
@@ -34,9 +34,6 @@ TAGS
|
||||
*.fas
|
||||
*.fasl
|
||||
|
||||
# LCP generated C++ files
|
||||
*.lcp.cpp
|
||||
|
||||
src/database/distributed/serialization.hpp
|
||||
src/database/single_node_ha/serialization.hpp
|
||||
src/distributed/bfs_rpc_messages.hpp
|
||||
@@ -50,15 +47,11 @@ src/distributed/pull_produce_rpc_messages.hpp
|
||||
src/distributed/storage_gc_rpc_messages.hpp
|
||||
src/distributed/token_sharing_rpc_messages.hpp
|
||||
src/distributed/updates_rpc_messages.hpp
|
||||
src/query/frontend/ast/ast.hpp
|
||||
src/query/distributed/frontend/ast/ast_serialization.hpp
|
||||
src/durability/distributed/state_delta.hpp
|
||||
src/durability/single_node/state_delta.hpp
|
||||
src/durability/single_node_ha/state_delta.hpp
|
||||
src/query/frontend/semantic/symbol.hpp
|
||||
src/query/distributed/frontend/semantic/symbol_serialization.hpp
|
||||
src/query/distributed/plan/ops.hpp
|
||||
src/query/plan/operator.hpp
|
||||
src/raft/log_entry.hpp
|
||||
src/raft/raft_rpc_messages.hpp
|
||||
src/raft/snapshot_metadata.hpp
|
||||
@@ -66,3 +59,5 @@ src/raft/storage_info_rpc_messages.hpp
|
||||
src/stats/stats_rpc_messages.hpp
|
||||
src/storage/distributed/rpc/concurrent_id_mapper_rpc_messages.hpp
|
||||
src/transactions/distributed/engine_rpc_messages.hpp
|
||||
/tests/manual/js/transaction_timeout/package-lock.json
|
||||
/tests/manual/js/transaction_timeout/node_modules/
|
||||
|
||||
21
.pre-commit-config.yaml
Normal file
21
.pre-commit-config.yaml
Normal file
@@ -0,0 +1,21 @@
|
||||
repos:
|
||||
- repo: https://github.com/pre-commit/pre-commit-hooks
|
||||
rev: v4.4.0
|
||||
hooks:
|
||||
- id: check-yaml
|
||||
args: [--allow-multiple-documents]
|
||||
- id: end-of-file-fixer
|
||||
- id: trailing-whitespace
|
||||
- repo: https://github.com/psf/black
|
||||
rev: 23.1.0
|
||||
hooks:
|
||||
- id: black
|
||||
- repo: https://github.com/pycqa/isort
|
||||
rev: 5.12.0
|
||||
hooks:
|
||||
- id: isort
|
||||
name: isort (python)
|
||||
- repo: https://github.com/pre-commit/mirrors-clang-format
|
||||
rev: v13.0.0
|
||||
hooks:
|
||||
- id: clang-format
|
||||
@@ -1,172 +0,0 @@
|
||||
import os
|
||||
import os.path
|
||||
import fnmatch
|
||||
import logging
|
||||
import ycm_core
|
||||
|
||||
BASE_FLAGS = [
|
||||
'-Wall',
|
||||
'-Wextra',
|
||||
'-Werror',
|
||||
'-Wno-long-long',
|
||||
'-Wno-variadic-macros',
|
||||
'-fexceptions',
|
||||
'-ferror-limit=10000',
|
||||
'-std=c++1z',
|
||||
'-xc++',
|
||||
'-I/usr/lib/',
|
||||
'-I/usr/include/',
|
||||
'-I./src',
|
||||
'-I./include',
|
||||
'-I./libs/fmt',
|
||||
'-I./libs/yaml-cpp',
|
||||
'-I./libs/glog/include',
|
||||
'-I./libs/googletest/googletest/include',
|
||||
'-I./libs/googletest/googlemock/include',
|
||||
'-I./libs/benchmark/include',
|
||||
'-I./libs/cereal/include',
|
||||
# We include cppitertools headers directly from libs directory.
|
||||
'-I./libs',
|
||||
'-I./libs/rapidcheck/include',
|
||||
'-I./libs/antlr4/runtime/Cpp/runtime/src',
|
||||
'-I./libs/gflags/include',
|
||||
'-I./experimental/distributed/src',
|
||||
'-I./libs/postgresql/include',
|
||||
'-I./libs/bzip2',
|
||||
'-I./libs/zlib',
|
||||
'-I./libs/rocksdb/include',
|
||||
'-I./libs/librdkafka/include/librdkafka',
|
||||
'-I./build/include'
|
||||
]
|
||||
|
||||
SOURCE_EXTENSIONS = [
|
||||
'.cpp',
|
||||
'.cxx',
|
||||
'.cc',
|
||||
'.c',
|
||||
'.m',
|
||||
'.mm'
|
||||
]
|
||||
|
||||
HEADER_EXTENSIONS = [
|
||||
'.h',
|
||||
'.hxx',
|
||||
'.hpp',
|
||||
'.hh'
|
||||
]
|
||||
|
||||
# set the working directory of YCMD to be this file
|
||||
os.chdir(os.path.dirname(os.path.realpath(__file__)))
|
||||
|
||||
def IsHeaderFile(filename):
|
||||
extension = os.path.splitext(filename)[1]
|
||||
return extension in HEADER_EXTENSIONS
|
||||
|
||||
def GetCompilationInfoForFile(database, filename):
|
||||
if IsHeaderFile(filename):
|
||||
basename = os.path.splitext(filename)[0]
|
||||
for extension in SOURCE_EXTENSIONS:
|
||||
replacement_file = basename + extension
|
||||
if os.path.exists(replacement_file):
|
||||
compilation_info = database.GetCompilationInfoForFile(replacement_file)
|
||||
if compilation_info.compiler_flags_:
|
||||
return compilation_info
|
||||
return None
|
||||
return database.GetCompilationInfoForFile(filename)
|
||||
|
||||
def FindNearest(path, target):
|
||||
candidate = os.path.join(path, target)
|
||||
if(os.path.isfile(candidate) or os.path.isdir(candidate)):
|
||||
logging.info("Found nearest " + target + " at " + candidate)
|
||||
return candidate;
|
||||
else:
|
||||
parent = os.path.dirname(os.path.abspath(path));
|
||||
if(parent == path):
|
||||
raise RuntimeError("Could not find " + target);
|
||||
return FindNearest(parent, target)
|
||||
|
||||
def MakeRelativePathsInFlagsAbsolute(flags, working_directory):
|
||||
if not working_directory:
|
||||
return list(flags)
|
||||
new_flags = []
|
||||
make_next_absolute = False
|
||||
path_flags = [ '-isystem', '-I', '-iquote', '--sysroot=' ]
|
||||
for flag in flags:
|
||||
new_flag = flag
|
||||
|
||||
if make_next_absolute:
|
||||
make_next_absolute = False
|
||||
if not flag.startswith('/'):
|
||||
new_flag = os.path.join(working_directory, flag)
|
||||
|
||||
for path_flag in path_flags:
|
||||
if flag == path_flag:
|
||||
make_next_absolute = True
|
||||
break
|
||||
|
||||
if flag.startswith(path_flag):
|
||||
path = flag[ len(path_flag): ]
|
||||
new_flag = path_flag + os.path.join(working_directory, path)
|
||||
break
|
||||
|
||||
if new_flag:
|
||||
new_flags.append(new_flag)
|
||||
return new_flags
|
||||
|
||||
|
||||
def FlagsForClangComplete(root):
|
||||
try:
|
||||
clang_complete_path = FindNearest(root, '.clang_complete')
|
||||
clang_complete_flags = open(clang_complete_path, 'r').read().splitlines()
|
||||
return clang_complete_flags
|
||||
except:
|
||||
return None
|
||||
|
||||
def FlagsForInclude(root):
|
||||
try:
|
||||
include_path = FindNearest(root, 'include')
|
||||
flags = []
|
||||
for dirroot, dirnames, filenames in os.walk(include_path):
|
||||
for dir_path in dirnames:
|
||||
real_path = os.path.join(dirroot, dir_path)
|
||||
flags = flags + ["-I" + real_path]
|
||||
return flags
|
||||
except:
|
||||
return None
|
||||
|
||||
def FlagsForCompilationDatabase(root, filename):
|
||||
try:
|
||||
compilation_db_path = FindNearest(root, 'compile_commands.json')
|
||||
compilation_db_dir = os.path.dirname(compilation_db_path)
|
||||
logging.info("Set compilation database directory to " + compilation_db_dir)
|
||||
compilation_db = ycm_core.CompilationDatabase(compilation_db_dir)
|
||||
if not compilation_db:
|
||||
logging.info("Compilation database file found but unable to load")
|
||||
return None
|
||||
compilation_info = GetCompilationInfoForFile(compilation_db, filename)
|
||||
if not compilation_info:
|
||||
logging.info("No compilation info for " + filename + " in compilation database")
|
||||
return None
|
||||
return MakeRelativePathsInFlagsAbsolute(
|
||||
compilation_info.compiler_flags_,
|
||||
compilation_info.compiler_working_dir_)
|
||||
except:
|
||||
return None
|
||||
|
||||
def FlagsForFile(filename):
|
||||
root = os.path.realpath(filename);
|
||||
compilation_db_flags = FlagsForCompilationDatabase(root, filename)
|
||||
if compilation_db_flags:
|
||||
final_flags = compilation_db_flags
|
||||
else:
|
||||
final_flags = BASE_FLAGS
|
||||
clang_flags = FlagsForClangComplete(root)
|
||||
if clang_flags:
|
||||
final_flags = final_flags + clang_flags
|
||||
include_flags = FlagsForInclude(root)
|
||||
if include_flags:
|
||||
final_flags = final_flags + include_flags
|
||||
return {
|
||||
'flags': final_flags,
|
||||
'do_cache': True
|
||||
}
|
||||
301
CHANGELOG.md
301
CHANGELOG.md
@@ -1,298 +1,5 @@
|
||||
# Change Log
|
||||
Change Log for all versions of Memgraph can be found on-line at
|
||||
https://docs.memgraph.com/memgraph/changelog
|
||||
|
||||
## v0.50.0
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
* [Enterprise Ed.] Remove support for Kafka streams.
|
||||
* Snapshot and write-ahead log format changed (not backward compatible).
|
||||
* Removed support for unique constraints.
|
||||
* Label indices aren't created automatically, create them explicitly instead.
|
||||
* Renamed several database flags. Please see the configuration file for a list of current flags.
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* [Enterprise Ed.] Add support for auth module.
|
||||
* [Enterprise Ed.] LDAP support migrated to auth module.
|
||||
* Implemented new graph storage engine.
|
||||
* Add support for disabling properties on edges.
|
||||
* Add support for existence constraints.
|
||||
* Add support for custom openCypher procedures using a C API.
|
||||
* Support loading query modules implementing read-only procedures.
|
||||
* Add `CALL <procedure> YIELD <result>` syntax for invoking loaded procedures.
|
||||
* Add `CREATE INDEX ON :Label` for creating label indices.
|
||||
* Add `DROP INDEX ON :Label` for dropping label indices.
|
||||
* Add `DUMP DATABASE` clause to openCypher.
|
||||
* Add functions for treating character strings as byte strings.
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
* Fix several memory management bugs.
|
||||
* Reduce memory usage in query execution.
|
||||
* Fix bug that crashes the database when `EXPLAIN` is used.
|
||||
|
||||
## v0.15.0
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
* Snapshot and write-ahead log format changed (not backward compatible).
|
||||
* `indexInfo()` function replaced with `SHOW INDEX INFO` syntax.
|
||||
* Removed support for unique index. Use unique constraints instead.
|
||||
* `CREATE UNIQUE INDEX ON :label (property)` replaced with `CREATE CONSTRAINT ON (n:label) ASSERT n.property IS UNIQUE`.
|
||||
* Changed semantics for `COUNTER` openCypher function.
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* [Enterprise Ed.] Add new privilege, `STATS` for accessing storage info.
|
||||
* [Enterprise Ed.] LDAP authentication and authorization support.
|
||||
* [Enterprise Ed.] Add audit logging feature.
|
||||
* Add multiple properties unique constraint which replace unique indices.
|
||||
* Add `SHOW STORAGE INFO` feature.
|
||||
* Add `PROFILE` clause to openCypher.
|
||||
* Add `CREATE CONSTRAINT` clause to openCypher.
|
||||
* Add `DROP CONSTRAINT` clause to openCypher.
|
||||
* Add `SHOW CONSTRAINT INFO` feature.
|
||||
* Add `uniformSample` function to openCypher.
|
||||
* Add regex matching to openCypher.
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
* Fix bug in explicit transaction handling.
|
||||
* Fix bug in edge filtering by edge type and destination.
|
||||
* Fix bug in query comment parsing.
|
||||
* Fix bug in query symbol table.
|
||||
* Fix OpenSSL memory leaks.
|
||||
* Make authentication case insensitive.
|
||||
* Remove `COALESCE` function.
|
||||
* Add movie tutorial.
|
||||
* Add backpacking tutorial.
|
||||
|
||||
## v0.14.0
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
* Write-ahead log format changed (not backward compatible).
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* [Enterprise Ed.] Reduce memory usage in distributed usage.
|
||||
* Add `DROP INDEX` feature.
|
||||
* Improve SSL error messages.
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
* [Enterprise Ed.] Fix issues with reading and writing in a distributed query.
|
||||
* Correctly handle an edge case with unique constraint checks.
|
||||
* Fix a minor issue with `mg_import_csv`.
|
||||
* Fix an issue with `EXPLAIN`.
|
||||
|
||||
## v0.13.0
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
* Write-ahead log format changed (not backward compatible).
|
||||
* Snapshot format changed (not backward compatible).
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* [Enterprise Ed.] Authentication and authorization support.
|
||||
* [Enterprise Ed.] Kafka integration.
|
||||
* [Enterprise Ed.] Support dynamic worker addition in distributed.
|
||||
* Reduce memory usage and improve overall performance.
|
||||
* Add `CREATE UNIQUE INDEX` clause to openCypher.
|
||||
* Add `EXPLAIN` clause to openCypher.
|
||||
* Add `inDegree` and `outDegree` functions to openCypher.
|
||||
* Improve BFS performance when both endpoints are known.
|
||||
* Add new `node-label`, `relationship-type` and `quote` options to
|
||||
`mg_import_csv` tool.
|
||||
* Reduce memory usage of `mg_import_csv`.
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
* [Enterprise Ed.] Fix an edge case in distributed index creation.
|
||||
* [Enterprise Ed.] Fix issues with Cartesian in distributed queries.
|
||||
* Correctly handle large messages in Bolt protocol.
|
||||
* Fix issues when handling explicitly started transactions in queries.
|
||||
* Allow openCypher keywords to be used as variable names.
|
||||
* Revise and make user visible error messages consistent.
|
||||
* Improve aborting time consuming execution.
|
||||
|
||||
## v0.12.0
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
* Snapshot format changed (not backward compatible).
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* Improved Id Cypher function.
|
||||
* Added string functions to openCypher (`lTrim`, `left`, `rTrim`, `replace`,
|
||||
`reverse`, `right`, `split`, `substring`, `toLower`, `toUpper`, `trim`).
|
||||
* Added `timestamp` function to openCypher.
|
||||
* Added support for dynamic property access with `[]` operator.
|
||||
|
||||
## v0.11.0
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* [Enterprise Ed.] Improve Cartesian support in distributed queries.
|
||||
* [Enterprise Ed.] Improve distributed execution of BFS.
|
||||
* [Enterprise Ed.] Dynamic graph partitioner added.
|
||||
* Static nodes/edges id generators exposed through the Id Cypher function.
|
||||
* Properties on disk added.
|
||||
* Telemetry added.
|
||||
* SSL support added.
|
||||
* `toString` function added.
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
* Document issues with Docker on OS X.
|
||||
* Add BFS and Dijkstra's algorithm examples to documentation.
|
||||
|
||||
## v0.10.0
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
* Snapshot format changed (not backward compatible).
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* [Enterprise Ed.] Distributed storage and execution.
|
||||
* `reduce` and `single` functions added to openCypher.
|
||||
* `wShortest` edge expansion added to openCypher.
|
||||
* Support packaging RPM on CentOS 7.
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
* Report an error if updating a deleted element.
|
||||
* Log an error if reading info on available memory fails.
|
||||
* Fix a bug when `MATCH` would stop matching if a result was empty, but later
|
||||
results still contain data to be matched. The simplest case of this was the
|
||||
query: `UNWIND [1,2,3] AS x MATCH (n :Label {prop: x}) RETURN n`. If there
|
||||
was no node `(:Label {prop: 1})`, then the `MATCH` wouldn't even try to find
|
||||
for `x` being 2 or 3.
|
||||
* Report an error if trying to compare a property value with something that
|
||||
cannot be stored in a property.
|
||||
* Fix crashes in some obscure cases.
|
||||
* Commit log automatically garbage collected.
|
||||
* Add minor performance improvements.
|
||||
|
||||
## v0.9.0
|
||||
|
||||
### Breaking Changes
|
||||
|
||||
* Snapshot format changed (not backward compatible).
|
||||
* Snapshot configuration flags changed, general durability flags added.
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* Write-ahead log added.
|
||||
* `nodes` and `relationships` functions added.
|
||||
* `UNION` and `UNION ALL` is implemented.
|
||||
* Concurrent index creation is now enabled.
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
|
||||
## v0.8.0
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* CASE construct (without aggregations).
|
||||
* Named path support added.
|
||||
* Maps can now be stored as node/edge properties.
|
||||
* Map indexing supported.
|
||||
* `rand` function added.
|
||||
* `assert` function added.
|
||||
* `counter` and `counterSet` functions added.
|
||||
* `indexInfo` function added.
|
||||
* `collect` aggregation now supports Map collection.
|
||||
* Changed the BFS syntax.
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
* Use \u to specify 4 digit codepoint and \U for 8 digit
|
||||
* Keywords appearing in header (named expressions) keep original case.
|
||||
* Our Bolt protocol implementation is now completely compatible with the protocol version 1 specification. (https://boltprotocol.org/v1/)
|
||||
* Added a log warning when running out of memory and the `memory_warning_threshold` flag
|
||||
* Edges are no longer additionally filtered after expansion.
|
||||
|
||||
## v0.7.0
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* Variable length path `MATCH`.
|
||||
* Explicitly started transactions (multi-query transactions).
|
||||
* Map literal.
|
||||
* Query parameters (except for parameters in place of property maps).
|
||||
* `all` function in openCypher.
|
||||
* `degree` function in openCypher.
|
||||
* User specified transaction execution timeout.
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
* Concurrent `BUILD INDEX` deadlock now returns an error to the client.
|
||||
* A `MATCH` preceeded by `OPTIONAL MATCH` expansion inconsistencies.
|
||||
* High concurrency Antlr parsing bug.
|
||||
* Indexing improvements.
|
||||
* Query stripping and caching speedups.
|
||||
|
||||
## v0.6.0
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* AST caching.
|
||||
* Label + property index support.
|
||||
* Different logging setup & format.
|
||||
|
||||
## v0.5.0
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* Use label indexes to speed up querying.
|
||||
* Generate multiple query plans and use the cost estimator to select the best.
|
||||
* Snapshots & Recovery.
|
||||
* Abandon old yaml configuration and migrate to gflags.
|
||||
* Query stripping & AST caching support.
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
* Fixed race condition in MVCC. Hints exp+aborted race condition prevented.
|
||||
* Fixed conceptual bug in MVCC GC. Evaluate old records w.r.t. the oldest.
|
||||
transaction's id AND snapshot.
|
||||
* User friendly error messages thrown from the query engine.
|
||||
|
||||
## Build 837
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
* List indexing supported with preceeding IN (for example in query `RETURN 1 IN [[1,2]][0]`).
|
||||
|
||||
## Build 825
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* RETURN *, count(*), OPTIONAL MATCH, UNWIND, DISTINCT (except DISTINCT in aggregate functions), list indexing and slicing, escaped labels, IN LIST operator, range function.
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
* TCP_NODELAY -> import should be faster.
|
||||
* Clear hint bits.
|
||||
|
||||
## Build 783
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* SKIP, LIMIT, ORDER BY.
|
||||
* Math functions.
|
||||
* Initial support for MERGE clause.
|
||||
|
||||
### Bug Fixes and Other Changes
|
||||
|
||||
* Unhandled Lock Timeout Exception.
|
||||
|
||||
## Build 755
|
||||
|
||||
### Major Features and Improvements
|
||||
|
||||
* MATCH, CREATE, WHERE, SET, REMOVE, DELETE.
|
||||
All the updates to the Change Log can be made in the following repository:
|
||||
https://github.com/memgraph/docs
|
||||
|
||||
330
CMakeLists.txt
330
CMakeLists.txt
@@ -1,6 +1,7 @@
|
||||
# MemGraph CMake configuration
|
||||
|
||||
cmake_minimum_required(VERSION 3.8)
|
||||
cmake_minimum_required(VERSION 3.12)
|
||||
cmake_policy(SET CMP0076 NEW)
|
||||
|
||||
# !! IMPORTANT !! run ./project_root/init.sh before cmake command
|
||||
# to download dependencies
|
||||
@@ -18,10 +19,12 @@ set_directory_properties(PROPERTIES CLEAN_NO_CUSTOM TRUE)
|
||||
# during the code coverage process
|
||||
find_program(CCACHE_FOUND ccache)
|
||||
option(USE_CCACHE "ccache:" ON)
|
||||
message(STATUS "CCache: ${USE_CCACHE}")
|
||||
if(CCACHE_FOUND AND USE_CCACHE)
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_COMPILE ccache)
|
||||
set_property(GLOBAL PROPERTY RULE_LAUNCH_LINK ccache)
|
||||
message(STATUS "CCache: Used")
|
||||
else ()
|
||||
message(STATUS "CCache: Not used")
|
||||
endif(CCACHE_FOUND AND USE_CCACHE)
|
||||
|
||||
# choose a compiler
|
||||
@@ -35,21 +38,139 @@ else()
|
||||
message(FATAL_ERROR "Couldn't find clang and/or clang++!")
|
||||
endif()
|
||||
|
||||
# Get current commit hash.
|
||||
execute_process(
|
||||
OUTPUT_VARIABLE COMMIT_HASH
|
||||
COMMAND git rev-parse --short HEAD
|
||||
)
|
||||
string(STRIP ${COMMIT_HASH} COMMIT_HASH)
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
project(memgraph VERSION 0.50.0)
|
||||
project(memgraph LANGUAGES C CXX)
|
||||
|
||||
#TODO: upgrade to cmake 3.24 + CheckIPOSupported
|
||||
#cmake_policy(SET CMP0138 NEW)
|
||||
#include(CheckIPOSupported)
|
||||
#check_ipo_supported()
|
||||
#set(CMAKE_INTERPROCEDURAL_OPTIMIZATION_Release TRUE)
|
||||
#set(CMAKE_INTERPROCEDURAL_OPTIMIZATION_RelWithDebInfo TRUE)
|
||||
|
||||
# Install licenses.
|
||||
install(DIRECTORY ${CMAKE_CURRENT_SOURCE_DIR}/licenses/
|
||||
DESTINATION share/doc/memgraph)
|
||||
|
||||
# For more information about how to release a new version of Memgraph, see
|
||||
# `release/README.md`.
|
||||
|
||||
# Option that is used to specify which version of Memgraph should be built. The
|
||||
# default is `ON` which causes the build system to build Memgraph Enterprise.
|
||||
# Memgraph Community is built if explicitly set to `OFF`.
|
||||
option(MG_ENTERPRISE "Build Memgraph Enterprise Edition" ON)
|
||||
|
||||
# Set the current version here to override the automatic version detection. The
|
||||
# version must be specified as `X.Y.Z`. Primarily used when building new patch
|
||||
# versions.
|
||||
set(MEMGRAPH_OVERRIDE_VERSION "")
|
||||
|
||||
# Custom suffix that this version should have. The suffix can be any arbitrary
|
||||
# string. Primarily used when building a version for a specific customer.
|
||||
set(MEMGRAPH_OVERRIDE_VERSION_SUFFIX "")
|
||||
|
||||
# Variables used to generate the versions.
|
||||
if (MG_ENTERPRISE)
|
||||
set(get_version_offering "")
|
||||
else()
|
||||
set(get_version_offering "--open-source")
|
||||
endif()
|
||||
set(get_version_script "${CMAKE_CURRENT_SOURCE_DIR}/release/get_version.py")
|
||||
|
||||
# Get version that should be used in the binary.
|
||||
execute_process(
|
||||
OUTPUT_VARIABLE MEMGRAPH_VERSION
|
||||
RESULT_VARIABLE MEMGRAPH_VERSION_RESULT
|
||||
COMMAND "${get_version_script}" ${get_version_offering}
|
||||
"${MEMGRAPH_OVERRIDE_VERSION}"
|
||||
"${MEMGRAPH_OVERRIDE_VERSION_SUFFIX}"
|
||||
"--memgraph-root-dir"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}"
|
||||
)
|
||||
if(MEMGRAPH_VERSION_RESULT AND NOT MEMGRAPH_VERSION_RESULT EQUAL 0)
|
||||
message(FATAL_ERROR "Unable to get Memgraph version.")
|
||||
else()
|
||||
MESSAGE(STATUS "Memgraph version: ${MEMGRAPH_VERSION}")
|
||||
endif()
|
||||
|
||||
# Get version that should be used in the DEB package.
|
||||
execute_process(
|
||||
OUTPUT_VARIABLE MEMGRAPH_VERSION_DEB
|
||||
RESULT_VARIABLE MEMGRAPH_VERSION_DEB_RESULT
|
||||
COMMAND "${get_version_script}" ${get_version_offering}
|
||||
--variant deb
|
||||
"${MEMGRAPH_OVERRIDE_VERSION}"
|
||||
"${MEMGRAPH_OVERRIDE_VERSION_SUFFIX}"
|
||||
"--memgraph-root-dir"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}"
|
||||
)
|
||||
if(MEMGRAPH_VERSION_DEB_RESULT AND NOT MEMGRAPH_VERSION_DEB_RESULT EQUAL 0)
|
||||
message(FATAL_ERROR "Unable to get Memgraph DEB version.")
|
||||
else()
|
||||
MESSAGE(STATUS "Memgraph DEB version: ${MEMGRAPH_VERSION_DEB}")
|
||||
endif()
|
||||
|
||||
# Get version that should be used in the RPM package.
|
||||
execute_process(
|
||||
OUTPUT_VARIABLE MEMGRAPH_VERSION_RPM
|
||||
RESULT_VARIABLE MEMGRAPH_VERSION_RPM_RESULT
|
||||
COMMAND "${get_version_script}" ${get_version_offering}
|
||||
--variant rpm
|
||||
"${MEMGRAPH_OVERRIDE_VERSION}"
|
||||
"${MEMGRAPH_OVERRIDE_VERSION_SUFFIX}"
|
||||
"--memgraph-root-dir"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}"
|
||||
)
|
||||
if(MEMGRAPH_VERSION_RPM_RESULT AND NOT MEMGRAPH_VERSION_RPM_RESULT EQUAL 0)
|
||||
message(FATAL_ERROR "Unable to get Memgraph RPM version.")
|
||||
else()
|
||||
MESSAGE(STATUS "Memgraph RPM version: ${MEMGRAPH_VERSION_RPM}")
|
||||
endif()
|
||||
|
||||
# We want the above variables to be updated each time something is committed to
|
||||
# the repository. That is why we include a dependency on the current git HEAD
|
||||
# to trigger a new CMake run when the git repository state changes. This is a
|
||||
# hack, as CMake doesn't have a mechanism to regenerate variables when
|
||||
# something changes (only files can be regenerated).
|
||||
# https://cmake.org/pipermail/cmake/2018-October/068389.html
|
||||
#
|
||||
# The hack in the above link is nearly correct but it has a fatal flaw. The
|
||||
# `CMAKE_CONFIGURE_DEPENDS` isn't a `GLOBAL` property, it is instead a
|
||||
# `DIRECTORY` property and as such must be set in the `DIRECTORY` scope.
|
||||
# https://cmake.org/cmake/help/v3.14/manual/cmake-properties.7.html
|
||||
#
|
||||
# Unlike the above mentioned hack, we don't use the `.git/index` file. That
|
||||
# file changes on every `git add` (even on `git status`) so it triggers
|
||||
# unnecessary recalculations of the release version. The release version only
|
||||
# changes on every `git commit` or `git checkout`. That is why we watch the
|
||||
# following files for changes:
|
||||
# - `.git/HEAD` -> changes each time a `git checkout` is issued
|
||||
# - `.git/refs/heads/...` -> the value in `.git/HEAD` is a branch name (when
|
||||
# you are on a branch) and you have to monitor the file of the specific
|
||||
# branch to detect when a `git commit` was issued
|
||||
# More details about the contents of the `.git` directory and the specific
|
||||
# files used can be seen here:
|
||||
# https://git-scm.com/book/en/v2/Git-Internals-Git-References
|
||||
set(git_directory "${CMAKE_SOURCE_DIR}/.git")
|
||||
# Check for directory because if the repo is cloned as a git submodule, .git is
|
||||
# a file and below code doesn't work.
|
||||
if (IS_DIRECTORY "${git_directory}")
|
||||
set_property(DIRECTORY APPEND PROPERTY
|
||||
CMAKE_CONFIGURE_DEPENDS "${git_directory}/HEAD")
|
||||
file(STRINGS "${git_directory}/HEAD" git_head_data)
|
||||
if (git_head_data MATCHES "^ref: ")
|
||||
string(SUBSTRING "${git_head_data}" 5 -1 git_head_ref)
|
||||
set_property(DIRECTORY APPEND PROPERTY
|
||||
CMAKE_CONFIGURE_DEPENDS "${git_directory}/${git_head_ref}")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
# setup CMake module path, defines path for include() and find_package()
|
||||
# https://cmake.org/cmake/help/latest/variable/CMAKE_MODULE_PATH.html
|
||||
set(CMAKE_MODULE_PATH ${CMAKE_MODULE_PATH} ${PROJECT_SOURCE_DIR}/cmake)
|
||||
list(APPEND CMAKE_MODULE_PATH "${CMAKE_CURRENT_SOURCE_DIR}/cmake")
|
||||
# custom function definitions
|
||||
include(functions)
|
||||
# -----------------------------------------------------------------------------
|
||||
@@ -68,20 +189,24 @@ add_custom_target(clean_all
|
||||
# is easier debugging of compilation and linker flags.
|
||||
set(CMAKE_EXPORT_COMPILE_COMMANDS ON)
|
||||
|
||||
set(CMAKE_CXX_STANDARD 17)
|
||||
set(CMAKE_CXX_STANDARD 20)
|
||||
set(CMAKE_CXX_STANDARD_REQUIRED ON)
|
||||
# c99-designator is disabled because of required mixture of designated and
|
||||
# non-designated initializers in Python Query Module code (`py_module.cpp`).
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -Wall \
|
||||
-Werror=switch -Werror=switch-bool -Werror=return-type \
|
||||
-Werror=return-stack-address")
|
||||
-Werror=return-stack-address \
|
||||
-Wno-c99-designator -Wmissing-field-initializers \
|
||||
-DBOOST_ASIO_USE_TS_EXECUTOR_AS_DEFAULT")
|
||||
|
||||
# Don't omit frame pointer in RelWithDebInfo, for additional callchain debug.
|
||||
set(CMAKE_CXX_FLAGS_RELWITHDEBINFO
|
||||
"${CMAKE_CXX_FLAGS_RELWITHDEBINFO} -fno-omit-frame-pointer")
|
||||
|
||||
# Statically link libgcc and libstdc++, the GCC allows this according to:
|
||||
# https://gcc.gnu.org/onlinedocs/gcc-8.3.0/libstdc++/manual/manual/license.html
|
||||
# https://gcc.gnu.org/onlinedocs/gcc-10.2.0/libstdc++/manual/manual/license.html
|
||||
# https://www.gnu.org/licenses/gcc-exception-faq.html
|
||||
# Last checked for gcc-8.3 which we are using on the build machines.
|
||||
# Last checked for gcc-10.2 which we are using on the build machines.
|
||||
# ** If we change versions, recheck this! **
|
||||
# ** Static linking is allowed only for executables! **
|
||||
set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -static-libgcc -static-libstdc++")
|
||||
@@ -92,6 +217,8 @@ set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -fuse-ld=gold")
|
||||
# release flags
|
||||
set(CMAKE_CXX_FLAGS_RELEASE "-O2 -DNDEBUG")
|
||||
|
||||
SET(CMAKE_CXX_LINK_FLAGS "${CMAKE_CXX_LINK_FLAGS} -pthread")
|
||||
|
||||
#debug flags
|
||||
set(PREFERRED_DEBUGGER "gdb" CACHE STRING
|
||||
"Tunes the debug output for your preferred debugger (gdb or lldb).")
|
||||
@@ -107,13 +234,6 @@ else()
|
||||
set(CMAKE_CXX_FLAGS_DEBUG "-g")
|
||||
endif()
|
||||
|
||||
# ndebug
|
||||
option(NDEBUG "No debug" OFF)
|
||||
message(STATUS "NDEBUG: ${NDEBUG} (be careful CMAKE_BUILD_TYPE can also \
|
||||
append this flag)")
|
||||
if(NDEBUG)
|
||||
add_definitions( -DNDEBUG )
|
||||
endif()
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
# default build type is debug
|
||||
@@ -123,15 +243,22 @@ endif()
|
||||
message(STATUS "CMake build type: ${CMAKE_BUILD_TYPE}")
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
add_definitions( -DCMAKE_BUILD_TYPE_NAME="${CMAKE_BUILD_TYPE}")
|
||||
|
||||
if (NOT MG_ARCH)
|
||||
set(MG_ARCH_DESCR "Host architecture to build Memgraph on. Supported values are x86_64, ARM64.")
|
||||
if (${CMAKE_HOST_SYSTEM_PROCESSOR} MATCHES "aarch64")
|
||||
set(MG_ARCH "ARM64" CACHE STRING ${MG_ARCH_DESCR})
|
||||
else()
|
||||
set(MG_ARCH "x86_64" CACHE STRING ${MG_ARCH_DESCR})
|
||||
endif()
|
||||
endif()
|
||||
message(STATUS "MG_ARCH: ${MG_ARCH}")
|
||||
|
||||
# setup external dependencies -------------------------------------------------
|
||||
|
||||
# threading
|
||||
find_package(Threads REQUIRED)
|
||||
|
||||
# optional Ltalloc
|
||||
option(USE_LTALLOC "Use Ltalloc instead of default allocator (default OFF). \
|
||||
Set this to ON to link with Ltalloc." OFF)
|
||||
|
||||
# optional readline
|
||||
option(USE_READLINE "Use GNU Readline library if available (default ON). \
|
||||
Set this to OFF to prevent linking with Readline even if it is available." ON)
|
||||
@@ -142,68 +269,16 @@ if (USE_READLINE)
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# OpenSSL
|
||||
find_package(OpenSSL REQUIRED)
|
||||
|
||||
set(libs_dir ${CMAKE_SOURCE_DIR}/libs)
|
||||
add_subdirectory(libs EXCLUDE_FROM_ALL)
|
||||
|
||||
include_directories(SYSTEM ${GFLAGS_INCLUDE_DIR})
|
||||
include_directories(SYSTEM ${GLOG_INCLUDE_DIR})
|
||||
include_directories(SYSTEM ${FMT_INCLUDE_DIR})
|
||||
include_directories(SYSTEM ${ANTLR4_INCLUDE_DIR})
|
||||
include_directories(SYSTEM ${BZIP2_INCLUDE_DIR})
|
||||
include_directories(SYSTEM ${ZLIB_INCLUDE_DIR})
|
||||
include_directories(SYSTEM ${ROCKSDB_INCLUDE_DIR})
|
||||
include_directories(SYSTEM ${LIBRDKAFKA_INCLUDE_DIR})
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
# openCypher parser -----------------------------------------------------------
|
||||
set(opencypher_frontend ${CMAKE_SOURCE_DIR}/src/query/frontend/opencypher)
|
||||
set(opencypher_generated ${opencypher_frontend}/generated)
|
||||
set(opencypher_lexer_grammar ${opencypher_frontend}/grammar/MemgraphCypherLexer.g4)
|
||||
set(opencypher_parser_grammar ${opencypher_frontend}/grammar/MemgraphCypher.g4)
|
||||
|
||||
# enumerate all files that are generated from antlr
|
||||
set(antlr_opencypher_generated_src
|
||||
${opencypher_generated}/MemgraphCypherLexer.cpp
|
||||
${opencypher_generated}/MemgraphCypher.cpp
|
||||
${opencypher_generated}/MemgraphCypherBaseVisitor.cpp
|
||||
${opencypher_generated}/MemgraphCypherVisitor.cpp
|
||||
)
|
||||
|
||||
# Provide a command to generate sources if missing. If this were a
|
||||
# custom_target, it would always run and we don't want that.
|
||||
add_custom_command(OUTPUT ${antlr_opencypher_generated_src}
|
||||
COMMAND
|
||||
${CMAKE_COMMAND} -E make_directory ${opencypher_generated}
|
||||
COMMAND
|
||||
java -jar ${CMAKE_SOURCE_DIR}/libs/antlr-4.6-complete.jar -Dlanguage=Cpp -visitor -o ${opencypher_generated} -package antlropencypher ${opencypher_lexer_grammar} ${opencypher_parser_grammar}
|
||||
WORKING_DIRECTORY "${CMAKE_BINARY_DIR}"
|
||||
DEPENDS ${opencypher_lexer_grammar} ${opencypher_parser_grammar}
|
||||
${opencypher_frontend}/grammar/CypherLexer.g4
|
||||
${opencypher_frontend}/grammar/Cypher.g4)
|
||||
|
||||
# add custom target for generation
|
||||
add_custom_target(generate_opencypher_parser
|
||||
DEPENDS ${antlr_opencypher_generated_src})
|
||||
|
||||
add_library(antlr_opencypher_parser_lib STATIC ${antlr_opencypher_generated_src})
|
||||
target_link_libraries(antlr_opencypher_parser_lib antlr4)
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
# Optional subproject configuration -------------------------------------------
|
||||
option(POC "Build proof of concept binaries" OFF)
|
||||
option(EXPERIMENTAL "Build experimental binaries" OFF)
|
||||
option(CUSTOMERS "Build customer binaries" OFF)
|
||||
option(TEST_COVERAGE "Generate coverage reports from running memgraph" OFF)
|
||||
option(TOOLS "Build tools binaries" ON)
|
||||
option(QUERY_MODULES "Build query modules containing custom procedures" ON)
|
||||
option(MG_COMMUNITY "Build Memgraph Community Edition" OFF)
|
||||
option(ASAN "Build with Address Sanitizer. To get a reasonable performance option should be used only in Release or RelWithDebInfo build " OFF)
|
||||
option(TSAN "Build with Thread Sanitizer. To get a reasonable performance option should be used only in Release or RelWithDebInfo build " OFF)
|
||||
option(UBSAN "Build with Undefined Behaviour Sanitizer" OFF)
|
||||
option(THIN_LTO "Build with link time optimization" OFF)
|
||||
|
||||
if (TEST_COVERAGE)
|
||||
string(TOLOWER ${CMAKE_BUILD_TYPE} lower_build_type)
|
||||
@@ -214,12 +289,16 @@ if (TEST_COVERAGE)
|
||||
set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -fprofile-instr-generate -fcoverage-mapping")
|
||||
endif()
|
||||
|
||||
if (MG_COMMUNITY)
|
||||
add_definitions(-DMG_COMMUNITY)
|
||||
if (MG_ENTERPRISE)
|
||||
add_definitions(-DMG_ENTERPRISE)
|
||||
endif()
|
||||
|
||||
set(ENABLE_JEMALLOC ON)
|
||||
|
||||
if (ASAN)
|
||||
# Enable Addres sanitizer and get nicer stack traces in error messages.
|
||||
message(WARNING "Disabling jemalloc as it doesn't work well with ASAN")
|
||||
set(ENABLE_JEMALLOC OFF)
|
||||
# Enable Address sanitizer and get nicer stack traces in error messages.
|
||||
# NOTE: AddressSanitizer uses llvm-symbolizer binary from the Clang
|
||||
# distribution to symbolize the stack traces (note that ideally the
|
||||
# llvm-symbolizer version must match the version of ASan runtime library).
|
||||
@@ -265,34 +344,29 @@ if (UBSAN)
|
||||
# runtime library and c++ standard libraries are present.
|
||||
set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -fsanitize=undefined -fno-omit-frame-pointer -fno-sanitize=vptr")
|
||||
set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -fsanitize=undefined -fno-sanitize=vptr")
|
||||
# Run program with environment variable UBSAN_OPTIONS=print_stacktrace=1
|
||||
# Make sure llvm-symbolizer binary is in path
|
||||
# Run program with environment variable UBSAN_OPTIONS=print_stacktrace=1.
|
||||
# Make sure llvm-symbolizer binary is in path.
|
||||
# To make the program abort on undefined behavior, use UBSAN_OPTIONS=halt_on_error=1.
|
||||
endif()
|
||||
|
||||
if (THIN_LTO)
|
||||
set(CMAKE_CXX_FLAGS"${CMAKE_CXX_FLAGS} -flto=thin")
|
||||
set(CMAKE_EXE_LINKER_FLAGS "${CMAKE_EXE_LINKER_FLAGS} -flto=thin")
|
||||
endif()
|
||||
set(MG_PYTHON_VERSION "" CACHE STRING "Specify the exact Python version used by the query modules")
|
||||
set(MG_PYTHON_PATH "" CACHE STRING "Specify the exact Python path used by the query modules")
|
||||
|
||||
# Add subprojects
|
||||
include_directories(src)
|
||||
add_subdirectory(src)
|
||||
|
||||
if(POC)
|
||||
add_subdirectory(poc)
|
||||
endif()
|
||||
# Release configuration
|
||||
add_subdirectory(release)
|
||||
|
||||
if(EXPERIMENTAL)
|
||||
add_subdirectory(experimental)
|
||||
endif()
|
||||
option(MG_ENABLE_TESTING "Set this to OFF to disable building test binaries" ON)
|
||||
message(STATUS "MG_ENABLE_TESTING: ${MG_ENABLE_TESTING}")
|
||||
|
||||
if(CUSTOMERS)
|
||||
add_subdirectory(customers)
|
||||
if (MG_ENABLE_TESTING)
|
||||
enable_testing()
|
||||
add_subdirectory(tests)
|
||||
endif()
|
||||
|
||||
enable_testing()
|
||||
add_subdirectory(tests)
|
||||
|
||||
if(TOOLS)
|
||||
add_subdirectory(tools)
|
||||
endif()
|
||||
@@ -301,58 +375,6 @@ if(QUERY_MODULES)
|
||||
add_subdirectory(query_modules)
|
||||
endif()
|
||||
|
||||
# -----------------------------------------------------------------------------
|
||||
|
||||
# ---- Setup CPack --------
|
||||
# General setup
|
||||
set(CPACK_PACKAGE_NAME memgraph)
|
||||
set(CPACK_PACKAGE_VENDOR "Memgraph Ltd.")
|
||||
set(CPACK_PACKAGE_DESCRIPTION_SUMMARY
|
||||
"High performance, in-memory, transactional graph database")
|
||||
set(CPACK_PACKAGE_VERSION_MAJOR ${memgraph_VERSION_MAJOR})
|
||||
set(CPACK_PACKAGE_VERSION_MINOR ${memgraph_VERSION_MINOR})
|
||||
set(CPACK_PACKAGE_VERSION_PATCH ${memgraph_VERSION_PATCH})
|
||||
set(CPACK_PACKAGE_VERSION_TWEAK ${memgraph_VERSION_TWEAK})
|
||||
set(CPACK_PACKAGE_FILE_NAME ${CPACK_PACKAGE_NAME}-${memgraph_VERSION}-${COMMIT_HASH}${CPACK_SYSTEM_NAME})
|
||||
|
||||
# DEB specific
|
||||
# Instead of using "name <email>" format, we use "email (name)" to prevent
|
||||
# errors due to full stop, '.' at the end of "Ltd". (See: RFC 822)
|
||||
set(CPACK_DEBIAN_PACKAGE_MAINTAINER "tech@memgraph.com (Memgraph Ltd.)")
|
||||
set(CPACK_DEBIAN_PACKAGE_SECTION non-free/database)
|
||||
set(CPACK_DEBIAN_PACKAGE_HOMEPAGE https://memgraph.com)
|
||||
set(CPACK_DEBIAN_PACKAGE_CONTROL_EXTRA
|
||||
"${CMAKE_SOURCE_DIR}/release/debian/conffiles;"
|
||||
"${CMAKE_SOURCE_DIR}/release/debian/copyright;"
|
||||
"${CMAKE_SOURCE_DIR}/release/debian/prerm;"
|
||||
"${CMAKE_SOURCE_DIR}/release/debian/postrm;"
|
||||
"${CMAKE_SOURCE_DIR}/release/debian/postinst;")
|
||||
set(CPACK_DEBIAN_PACKAGE_SHLIBDEPS ON)
|
||||
# Description formatting is important, summary must be followed with a newline and 1 space.
|
||||
set(CPACK_DEBIAN_PACKAGE_DESCRIPTION "${CPACK_PACKAGE_DESCRIPTION_SUMMARY}
|
||||
Contains Memgraph, the graph database. It aims to deliver developers the
|
||||
speed, simplicity and scale required to build the next generation of
|
||||
applications driver by real-time connected data.")
|
||||
# Add `openssl` package to dependencies list. Used to generate SSL certificates.
|
||||
set(CPACK_DEBIAN_PACKAGE_DEPENDS "openssl (>= 1.1.0)")
|
||||
|
||||
# RPM specific
|
||||
set(CPACK_RPM_PACKAGE_URL https://memgraph.com)
|
||||
set(CPACK_RPM_EXCLUDE_FROM_AUTO_FILELIST_ADDITION
|
||||
/var /var/lib /var/log /etc/logrotate.d
|
||||
/lib /lib/systemd /lib/systemd/system /lib/systemd/system/memgraph.service)
|
||||
set(CPACK_RPM_PACKAGE_REQUIRES_PRE "shadow-utils")
|
||||
# NOTE: user specfile has a bug in cmake 3.7.2, this needs to be patched
|
||||
# manually in: ~/cmake/share/cmake-3.7/Modules/CPackRPM.cmake line 2273
|
||||
# Or newer cmake version used
|
||||
set(CPACK_RPM_USER_BINARY_SPECFILE "${CMAKE_SOURCE_DIR}/release/rpm/memgraph.spec.in")
|
||||
# Description formatting is important, no line must be greater than 80 characters.
|
||||
set(CPACK_RPM_PACKAGE_DESCRIPTION "Contains Memgraph, the graph database.
|
||||
It aims to deliver developers the speed, simplicity and scale required to build
|
||||
the next generation of applications driver by real-time connected data.")
|
||||
# Add `openssl` package to dependencies list. Used to generate SSL certificates.
|
||||
set(CPACK_RPM_PACKAGE_REQUIRES "openssl >= 1.0.0, curl >= 7.29.0")
|
||||
|
||||
# All variables must be set before including.
|
||||
include(CPack)
|
||||
# ---- End Setup CPack ----
|
||||
install(FILES ${CMAKE_BINARY_DIR}/bin/mgconsole
|
||||
PERMISSIONS OWNER_EXECUTE OWNER_READ OWNER_WRITE GROUP_READ GROUP_EXECUTE WORLD_READ WORLD_EXECUTE
|
||||
TYPE BIN)
|
||||
|
||||
127
CODE_OF_CONDUCT.md
Normal file
127
CODE_OF_CONDUCT.md
Normal file
@@ -0,0 +1,127 @@
|
||||
# Contributor Covenant Code of Conduct
|
||||
|
||||
## Our Pledge
|
||||
|
||||
We as members, contributors, and leaders pledge to make participation in our
|
||||
community a harassment-free experience for everyone, regardless of age, body
|
||||
size, visible or invisible disability, ethnicity, sex characteristics, gender
|
||||
identity and expression, level of experience, education, socio-economic status,
|
||||
nationality, personal appearance, race, caste, color, religion, or sexual
|
||||
identity and orientation.
|
||||
|
||||
We pledge to act and interact in ways that contribute to an open, welcoming,
|
||||
diverse, inclusive, and healthy community.
|
||||
|
||||
## Our Standards
|
||||
|
||||
Examples of behavior that contributes to a positive environment for our
|
||||
community include:
|
||||
|
||||
- Demonstrating empathy and kindness toward other people
|
||||
- Being respectful of differing opinions, viewpoints, and experiences
|
||||
- Giving and gracefully accepting constructive feedback
|
||||
- Accepting responsibility and apologizing to those affected by our mistakes,
|
||||
and learning from the experience
|
||||
- Focusing on what is best not just for us as individuals, but for the overall
|
||||
community
|
||||
|
||||
Examples of unacceptable behavior include:
|
||||
|
||||
- The use of sexualized language or imagery, and sexual attention or advances of
|
||||
any kind
|
||||
- Trolling, insulting or derogatory comments, and personal or political attacks
|
||||
- Public or private harassment
|
||||
- Publishing other's private information, such as a physical or email address,
|
||||
without their explicit permission
|
||||
- Other conduct which could reasonably be considered inappropriate in a
|
||||
professional setting
|
||||
|
||||
## Enforcement Responsibilities
|
||||
|
||||
Community leaders are responsible for clarifying and enforcing our standards of
|
||||
acceptable behavior and will take appropriate and fair corrective action in
|
||||
response to any behavior that they deem inappropriate, threatening, offensive,
|
||||
or harmful.
|
||||
|
||||
Community leaders have the right and responsibility to remove, edit, or reject
|
||||
comments, commits, code, wiki edits, issues, and other contributions that are
|
||||
not aligned to this Code of Conduct, and will communicate reasons for moderation
|
||||
decisions when appropriate.
|
||||
|
||||
## Scope
|
||||
|
||||
This Code of Conduct applies within all community spaces, and also applies when
|
||||
an individual is officially representing the community in public spaces.
|
||||
Examples of representing our community include using an official e-mail address,
|
||||
posting via an official social media account, or acting as an appointed
|
||||
representative at an online or offline event.
|
||||
|
||||
## Enforcement
|
||||
|
||||
Instances of abusive, harassing, or otherwise unacceptable behavior may be
|
||||
reported to the community leaders responsible for enforcement at
|
||||
[contact@memgraph.com](contact@memgraph.com). All complaints will be reviewed
|
||||
and investigated promptly and fairly.
|
||||
|
||||
All community leaders are obligated to respect the privacy and security of the
|
||||
reporter of any incident.
|
||||
|
||||
## Enforcement Guidelines
|
||||
|
||||
Community leaders will follow these Community Impact Guidelines in determining
|
||||
the consequences for any action they deem in violation of this Code of Conduct:
|
||||
|
||||
### 1. Correction
|
||||
|
||||
**Community Impact**: Use of inappropriate language or other behavior deemed
|
||||
unprofessional or unwelcome in the community.
|
||||
|
||||
**Consequence**: A private, written warning from community leaders, providing
|
||||
clarity around the nature of the violation and an explanation of why the
|
||||
behavior was inappropriate. A public apology may be requested.
|
||||
|
||||
### 2. Warning
|
||||
|
||||
**Community Impact**: A violation through a single incident or series of
|
||||
actions.
|
||||
|
||||
**Consequence**: A warning with consequences for continued behavior. No
|
||||
interaction with the people involved, including unsolicited interaction with
|
||||
those enforcing the Code of Conduct, for a specified period of time. This
|
||||
includes avoiding interactions in community spaces as well as external channels
|
||||
like social media. Violating these terms may lead to a temporary or permanent
|
||||
ban.
|
||||
|
||||
### 3. Temporary Ban
|
||||
|
||||
**Community Impact**: A serious violation of community standards, including
|
||||
sustained inappropriate behavior.
|
||||
|
||||
**Consequence**: A temporary ban from any sort of interaction or public
|
||||
communication with the community for a specified period of time. No public or
|
||||
private interaction with the people involved, including unsolicited interaction
|
||||
with those enforcing the Code of Conduct, is allowed during this period.
|
||||
Violating these terms may lead to a permanent ban.
|
||||
|
||||
### 4. Permanent Ban
|
||||
|
||||
**Community Impact**: Demonstrating a pattern of violation of community
|
||||
standards, including sustained inappropriate behavior, harassment of an
|
||||
individual, or aggression toward or disparagement of classes of individuals.
|
||||
|
||||
**Consequence**: A permanent ban from any sort of public interaction within the
|
||||
community.
|
||||
|
||||
## Attribution
|
||||
|
||||
This Code of Conduct is adapted from the Contributor Covenant, version 2.1,
|
||||
available at
|
||||
[https://www.contributor-covenant.org/version/2/1/code_of_conduct.html](https://www.contributor-covenant.org/version/2/1/code_of_conduct.html).
|
||||
|
||||
Community Impact Guidelines were inspired by [Mozilla's code of conduct
|
||||
enforcement ladder][mozilla coc].
|
||||
|
||||
For answers to common questions about this code of conduct, see the FAQ at
|
||||
[https://www.contributor-covenant.org/faq](https://www.contributor-covenant.org/faq).
|
||||
Translations are available at
|
||||
[https://www.contributor-covenant.org/translations](https://www.contributor-covenant.org/translations).
|
||||
121
CONTRIBUTING.md
Normal file
121
CONTRIBUTING.md
Normal file
@@ -0,0 +1,121 @@
|
||||
# How to contribute?
|
||||
|
||||
This is a general purpose guide for contributing to Memgraph. We're still
|
||||
working out the kinks to make contributing to this project as easy and
|
||||
transparent as possible, but we're not quite there yet. Hopefully, this document
|
||||
makes the process for contributing clear and answers some questions that you may
|
||||
have.
|
||||
|
||||
- [How to contribute?](#how-to-contribute)
|
||||
- [Open development](#open-development)
|
||||
- [Branch organization](#branch-organization)
|
||||
- [Bugs & changes](#bugs--changes)
|
||||
- [Where to find known issues?](#where-to-find-known-issues)
|
||||
- [Proposing a change](#proposing-a-change)
|
||||
- [Your first pull request](#your-first-pull-request)
|
||||
- [Sending a pull request](#sending-a-pull-request)
|
||||
- [Style guide](#style-guide)
|
||||
- [How to get in touch?](#how-to-get-in-touch)
|
||||
- [Code of Conduct](#code-of-conduct)
|
||||
- [License](#license)
|
||||
- [Attribution](#attribution)
|
||||
|
||||
## Open development
|
||||
|
||||
All work on Memgraph is done via [GitHub](https://github.com/memgraph/memgraph).
|
||||
Both core team members and external contributors send pull requests which go
|
||||
through the same review process.
|
||||
|
||||
## Branch organization
|
||||
|
||||
Most pull requests should target the [`master
|
||||
branch`](https://github.com/memgraph/memgraph/tree/master). We only use separate
|
||||
branches for developing new features and fixing bugs before they are merged with
|
||||
`master`. We do our best to keep `master` in good shape, with all tests passing.
|
||||
|
||||
Code that lands in `master` must be compatible with the latest stable release.
|
||||
It may contain additional features but no breaking changes if it's not
|
||||
absolutely necessary. We should be able to release a new minor version from the
|
||||
tip of `master` at any time.
|
||||
|
||||
## Bugs & changes
|
||||
|
||||
### Where to find known issues?
|
||||
|
||||
We are using [GitHub Issues](https://github.com/memgraph/memgraph/issues) for
|
||||
our public bugs. We keep a close eye on this and try to make it clear when we
|
||||
have an internal fix in progress. Before filing a new task, try to make sure
|
||||
your problem doesn't already exist.
|
||||
|
||||
### Proposing a change
|
||||
|
||||
If you intend to change the public API, or make any non-trivial changes to the
|
||||
implementation, we recommend [filing an
|
||||
issue](https://github.com/memgraph/memgraph/issues/new). This lets us reach an
|
||||
agreement on your proposal before you put significant effort into it.
|
||||
|
||||
If you're only fixing a bug, it's fine to submit a pull request right away but
|
||||
we still recommend to file an issue detailing what you're fixing. This is
|
||||
helpful in case we don't accept that specific fix but want to keep track of the
|
||||
issue.
|
||||
|
||||
### Your first pull request
|
||||
|
||||
Working on your first Pull Request? You can learn how from this free video
|
||||
series:
|
||||
|
||||
**[How to Contribute to an Open Source Project on
|
||||
GitHub](https://app.egghead.io/courses/how-to-contribute-to-an-open-source-project-on-github)**
|
||||
|
||||
If you decide to fix an issue, please be sure to check the comment thread in
|
||||
case somebody is already working on a fix. If nobody is working on it at the
|
||||
moment, please leave a comment stating that you intend to work on it so other
|
||||
people don't accidentally duplicate your effort.
|
||||
|
||||
If somebody claims an issue but doesn't follow up for more than two weeks, it's
|
||||
fine to take it over but you should still leave a comment.
|
||||
|
||||
### Sending a pull request
|
||||
|
||||
The core team is monitoring for pull requests. We will review your pull request
|
||||
and either merge it, request changes to it, or close it with an explanation.
|
||||
**Before submitting a pull request,** please make sure the following is done:
|
||||
|
||||
1. Fork [the repository](https://github.com/memgraph/memgraph) and create your
|
||||
branch from `master`.
|
||||
2. If you've fixed a bug or added code that should be tested, add tests!
|
||||
3. Use the formatter `clang-format` for C/C++ code and `flake8` for Python code.
|
||||
`clang-format` will automatically detect the `.clang-format` file in the root
|
||||
directory while `flake8` can be used with the default configuration.
|
||||
|
||||
### Style guide
|
||||
|
||||
Memgraph uses the [Google Style
|
||||
Guide](https://google.github.io/styleguide/cppguide.html) for C++ in most of its
|
||||
code. You should follow them whenever writing new code.
|
||||
|
||||
## How to get in touch?
|
||||
|
||||
Aside from communicating directly via Pull Requests and Issues, the Memgraph
|
||||
Community [Discord Server](https://discord.gg/memgraph) is the best place for
|
||||
conversing with project maintainers and other community members.
|
||||
|
||||
## [Code of Conduct](https://github.com/memgraph/memgraph/blob/master/CODE_OF_CONDUCT.md)
|
||||
|
||||
Memgraph has adopted the [Contributor
|
||||
Covenant](https://www.contributor-covenant.org/) as its Code of Conduct, and we
|
||||
expect project participants to adhere to it. Please read [the full
|
||||
text](https://github.com/memgraph/memgraph/blob/master/CODE_OF_CONDUCT.md) so
|
||||
that you can understand what actions will and will not be tolerated.
|
||||
|
||||
## License
|
||||
|
||||
By contributing to Memgraph, you agree that your contributions will be licensed
|
||||
under the [Memgraph licensing
|
||||
scheme](https://github.com/memgraph/memgraph/blob/master/LICENSE).
|
||||
|
||||
## Attribution
|
||||
|
||||
This Contributing guide is adapted from the **React.js Contributing guide**
|
||||
available at
|
||||
[https://reactjs.org/docs/how-to-contribute.html](https://reactjs.org/docs/how-to-contribute.html).
|
||||
3
Doxyfile
3
Doxyfile
@@ -51,7 +51,7 @@ PROJECT_BRIEF = "The World's Most Powerful Graph Database"
|
||||
# pixels and the maximum width should not exceed 200 pixels. Doxygen will copy
|
||||
# the logo to the output directory.
|
||||
|
||||
PROJECT_LOGO = Doxylogo.png
|
||||
PROJECT_LOGO = docs/doxygen/memgraph_logo.png
|
||||
|
||||
# The OUTPUT_DIRECTORY tag is used to specify the (relative or absolute) path
|
||||
# into which the generated documentation will be written. If a relative path is
|
||||
@@ -839,7 +839,6 @@ EXCLUDE_PATTERNS += */Testing/*
|
||||
EXCLUDE_PATTERNS += */tests/*
|
||||
EXCLUDE_PATTERNS += */dist/*
|
||||
EXCLUDE_PATTERNS += */tools/*
|
||||
EXCLUDE_PATTERNS += */customers/*
|
||||
|
||||
# The EXCLUDE_SYMBOLS tag can be used to specify one or more symbol names
|
||||
# (namespaces, classes, functions, etc.) that should be excluded from the
|
||||
|
||||
5
LICENSE
Normal file
5
LICENSE
Normal file
@@ -0,0 +1,5 @@
|
||||
Source code in this repository is variously licensed under the Business Source
|
||||
License 1.1 (BSL), the Memgraph Enterprise License (MEL). A copy of each licence
|
||||
can be found in the licences directory. Source code in a given file is licensed
|
||||
under the BSL and the copyright belongs to The Memgraph Authors unless
|
||||
otherwise noted at th beginning of the file.
|
||||
200
README.md
200
README.md
@@ -1,24 +1,186 @@
|
||||
# memgraph
|
||||
<p align="center">
|
||||
<img width="400px" src="https://uploads-ssl.webflow.com/5e7ceb09657a69bdab054b3a/5e7ceb09657a6937ab054bba_Black_Original%20_Logo.png">
|
||||
</p>
|
||||
|
||||
Memgraph is an ACID compliant high performance transactional distributed
|
||||
in-memory graph database featuring runtime native query compiling, lock free
|
||||
data structures, multi-version concurrency control and asynchronous IO.
|
||||
---
|
||||
|
||||
## dependencies
|
||||
<p align="center">
|
||||
<a href="https://github.com/memgraph/memgraph/blob/master/licenses/APL.txt">
|
||||
<img src="https://img.shields.io/badge/license-APL-green" alt="license" title="license"/>
|
||||
</a>
|
||||
<a href="https://github.com/memgraph/memgraph/blob/master/licenses/BSL.txt">
|
||||
<img src="https://img.shields.io/badge/license-BSL-yellowgreen" alt="license" title="license"/>
|
||||
</a>
|
||||
<a href="https://github.com/memgraph/memgraph/blob/master/licenses/MEL.txt" alt="Documentation">
|
||||
<img src="https://img.shields.io/badge/license-MEL-yellow" alt="license" title="license"/>
|
||||
</a>
|
||||
</p>
|
||||
|
||||
Memgraph can be compiled using any modern c++ compiler. It mostly relies on
|
||||
the standard template library, however, some things do require external
|
||||
libraries.
|
||||
<p align="center">
|
||||
<a href="https://github.com/memgraph/memgraph">
|
||||
<img src="https://img.shields.io/github/actions/workflow/status/memgraph/memgraph/release_debian10.yaml?branch=master&label=build%20and%20test&logo=github"/>
|
||||
</a>
|
||||
<a href="https://memgraph.com/docs/" alt="Documentation">
|
||||
<img src="https://img.shields.io/badge/documentation-Memgraph-orange" />
|
||||
</a>
|
||||
</p>
|
||||
|
||||
Some code contains linux-specific libraries and the build is only supported
|
||||
on a 64 bit linux kernel.
|
||||
<p align="center">
|
||||
<a href="https://memgr.ph/join-discord">
|
||||
<img src="https://img.shields.io/badge/Discord-7289DA?style=for-the-badge&logo=discord&logoColor=white" alt="Discord"/>
|
||||
</a>
|
||||
</p>
|
||||
|
||||
* linux
|
||||
* clang 3.8 (good c++11 support, especially lock free atomics)
|
||||
* antlr (compiler frontend)
|
||||
* cppitertools
|
||||
* fmt format
|
||||
* google benchmark
|
||||
* google test
|
||||
* glog
|
||||
* gflags
|
||||
## :clipboard: Description
|
||||
|
||||
Memgraph is an open source graph database built for real-time streaming and
|
||||
compatible with Neo4j. Whether you're a developer or a data scientist with
|
||||
interconnected data, Memgraph will get you the immediate actionable insights
|
||||
fast.
|
||||
|
||||
Memgraph directly connects to your streaming infrastructure. You can ingest data
|
||||
from sources like Kafka, SQL, or plain CSV files. Memgraph provides a standard
|
||||
interface to query your data with Cypher, a widely-used and declarative query
|
||||
language that is easy to write, understand and optimize for performance. This is
|
||||
achieved by using the property graph data model, which stores data in terms of
|
||||
objects, their attributes, and the relationships that connect them. This is a
|
||||
natural and effective way to model many real-world problems without relying on
|
||||
complex SQL schemas.
|
||||
|
||||
Memgraph is implemented in C/C++ and leverages an in-memory first architecture
|
||||
to ensure that you’re getting the [best possible
|
||||
performance](http://memgraph.com/benchgraph) consistently and without surprises.
|
||||
It’s also ACID-compliant and highly available.
|
||||
|
||||
## :zap: Features
|
||||
|
||||
- Run Python, Rust, and C/C++ code natively, check out the
|
||||
[MAGE](https://github.com/memgraph/mage) graph algorithm library
|
||||
- Native support for machine learning
|
||||
- Streaming support
|
||||
- Replication
|
||||
- Authentication and authorization
|
||||
- ACID compliance
|
||||
|
||||
|
||||
## :video_game: Memgraph Playground
|
||||
|
||||
You don't need to install anything to try out Memgraph. Check out
|
||||
our **[Memgraph Playground](https://playground.memgraph.com/)** sandboxes in
|
||||
your browser.
|
||||
|
||||
<p align="left">
|
||||
<a href="https://playground.memgraph.com/">
|
||||
<img width="450px" alt="Memgraph Playground" src="https://download.memgraph.com/asset/github/memgraph/memgraph-playground.png">
|
||||
</a>
|
||||
</p>
|
||||
|
||||
## :floppy_disk: Download & Install
|
||||
|
||||
### Windows
|
||||
|
||||
[](https://memgraph.com/docs/memgraph/install-memgraph-on-windows-docker)
|
||||
[](https://memgraph.com/docs/memgraph/install-memgraph-on-windows-wsl)
|
||||
|
||||
### macOS
|
||||
|
||||
[](https://memgraph.com/docs/memgraph/install-memgraph-on-macos-docker)
|
||||
[](https://memgraph.com/docs/memgraph/install-memgraph-on-ubuntu)
|
||||
|
||||
### Linux
|
||||
|
||||
[](https://memgraph.com/docs/memgraph/install-memgraph-on-linux-docker)
|
||||
[](https://memgraph.com/docs/memgraph/install-memgraph-on-debian)
|
||||
[](https://memgraph.com/docs/memgraph/install-memgraph-on-ubuntu)
|
||||
[](https://memgraph.com/docs/memgraph/install-memgraph-from-rpm)
|
||||
[](https://memgraph.com/docs/memgraph/install-memgraph-from-rpm)
|
||||
[](https://memgraph.com/docs/memgraph/install-memgraph-from-rpm)
|
||||
|
||||
You can find the binaries and Docker images on the [Download
|
||||
Hub](https://memgraph.com/download) and the installation instructions in the
|
||||
[official documentation](https://memgraph.com/docs/memgraph/installation).
|
||||
|
||||
|
||||
## :cloud: Memgraph Cloud
|
||||
|
||||
Check out [Memgraph Cloud](https://memgraph.com/docs/memgraph-cloud) - a cloud service fully managed on AWS and available in 6 geographic regions around the world. Memgraph Cloud allows you to create projects with Enterprise instances of MemgraphDB from your browser.
|
||||
|
||||
<p align="left">
|
||||
<a href="https://memgraph.com/docs/memgraph-cloud">
|
||||
<img width="450px" alt="Memgraph Cloud" src="https://public-assets.memgraph.com/memgraph-gifs%2Fcloud.gif">
|
||||
</a>
|
||||
</p>
|
||||
|
||||
## :link: Connect to Memgraph
|
||||
|
||||
[Connect to the database](https://memgraph.com/docs/memgraph/connect-to-memgraph) using Memgraph Lab, mgconsole, various drivers (Python, C/C++ and others) and WebSocket.
|
||||
|
||||
### :microscope: Memgraph Lab
|
||||
|
||||
Visualize graphs and play with queries to understand your data. [Memgraph Lab](https://memgraph.com/docs/memgraph-lab) is a user interface that helps you explore and manipulate the data stored in Memgraph. Visualize graphs, execute ad hoc queries, and optimize their performance.
|
||||
|
||||
<p align="left">
|
||||
<a href="https://memgraph.com/docs/memgraph-lab">
|
||||
<img width="450px" alt="Memgraph Cloud" src="https://public-assets.memgraph.com/memgraph-gifs%2Flab.gif">
|
||||
</a>
|
||||
</p>
|
||||
|
||||
## :file_folder: Import data
|
||||
|
||||
[Import data](https://memgraph.com/docs/memgraph/import-data) into Memgraph using Kafka, RedPanda or Pulsar streams, CSV and JSON files, or Cypher commands.
|
||||
|
||||
## :bookmark_tabs: Documentation
|
||||
|
||||
The Memgraph documentation is available at
|
||||
[memgraph.com/docs](https://memgraph.com/docs).
|
||||
|
||||
## :question: Configuration
|
||||
|
||||
Command line options that Memgraph accepts are available in the [reference
|
||||
guide](https://memgraph.com/docs/memgraph/reference-guide/configuration).
|
||||
|
||||
## :trophy: Contributing
|
||||
|
||||
The main purpose of this repository is to continue evolving Memgraph, making it
|
||||
faster and easier to use. Development of Memgraph happens in the open on GitHub,
|
||||
and we are grateful to the community for contributing bug fixes and
|
||||
improvements. Read below to learn how you can take part in improving Memgraph.
|
||||
|
||||
### Code of Conduct
|
||||
|
||||
Memgraph has adopted a Code of Conduct that we expect project participants to
|
||||
adhere to. Please read [the full text](CODE_OF_CONDUCT.md) so that you can
|
||||
understand what actions will and will not be tolerated.
|
||||
|
||||
### Contributing Guide
|
||||
|
||||
Read our [contributing guide](CONTRIBUTING.md) to learn about our development
|
||||
process and how to propose bug fixes and improvements.
|
||||
|
||||
### Internals
|
||||
|
||||
Read our
|
||||
[internal](https://memgraph.notion.site/Memgraph-Internals-12b69132d67a417898972927d6870bd2)
|
||||
docs to learn more about Memgraph's architecture, how to build the project from
|
||||
source and how to start contributing. All information related to the database,
|
||||
can be found in the aforementioned docs.
|
||||
|
||||
### :scroll: License
|
||||
|
||||
Memgraph Community is available under the [BSL
|
||||
license](./licenses/BSL.txt).</br> Memgraph Enterprise is available under the
|
||||
[MEL license](./licenses/MEL.txt).
|
||||
|
||||
## :busts_in_silhouette: Community
|
||||
|
||||
- :purple_heart: [**Discord**](https://discord.gg/memgraph)
|
||||
- :ocean: [**Stack Overflow**](https://stackoverflow.com/questions/tagged/memgraphdb)
|
||||
- :bird: [**Twitter**](https://twitter.com/memgraphdb)
|
||||
- :movie_camera:
|
||||
[**YouTube**](https://www.youtube.com/channel/UCZ3HOJvHGxtQ_JHxOselBYg)
|
||||
|
||||
<p align="center">
|
||||
<a href="#">
|
||||
<img src="https://img.shields.io/badge/⬆️ back_to_top_⬆️-white" alt="Back to top" title="Back to top"/>
|
||||
</a>
|
||||
</p>
|
||||
|
||||
@@ -1,39 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
|
||||
# paths
|
||||
SCRIPT_DIR = os.path.dirname(os.path.realpath(__file__))
|
||||
BUILD_OUTPUT_DIR = os.path.normpath(os.path.join(SCRIPT_DIR, "build_release", "output"))
|
||||
|
||||
# helpers
|
||||
def run_cmd(cmd, cwd):
|
||||
return subprocess.run(cmd, cwd=cwd, check=True,
|
||||
stdout=subprocess.PIPE).stdout.decode("utf-8")
|
||||
|
||||
# check project
|
||||
if re.search(r"release", os.environ.get("PROJECT", "")) is None:
|
||||
print(json.dumps([]))
|
||||
sys.exit(0)
|
||||
|
||||
# generate archive
|
||||
deb_name = run_cmd(["find", ".", "-maxdepth", "1", "-type", "f",
|
||||
"-name", "memgraph*.deb"], BUILD_OUTPUT_DIR).split("\n")[0][2:]
|
||||
arch = run_cmd(["dpkg", "--print-architecture"], BUILD_OUTPUT_DIR).split("\n")[0]
|
||||
version = deb_name.split("-")[1]
|
||||
# generate Debian package file name as expected by Debian Policy
|
||||
standard_deb_name = "memgraph_{}-1_{}.deb".format(version, arch)
|
||||
|
||||
archive_path = os.path.relpath(os.path.join(BUILD_OUTPUT_DIR,
|
||||
deb_name), SCRIPT_DIR)
|
||||
|
||||
archives = [{
|
||||
"name": "Release (deb package)",
|
||||
"archive": archive_path,
|
||||
"filename": standard_deb_name,
|
||||
}]
|
||||
|
||||
print(json.dumps(archives, indent=4, sort_keys=True))
|
||||
@@ -1,17 +0,0 @@
|
||||
- name: Binaries
|
||||
archive:
|
||||
- build_debug/memgraph
|
||||
- build_debug/memgraph_ha
|
||||
- build_release/memgraph
|
||||
- build_release/memgraph_ha
|
||||
- build_release/tools/src/mg_client
|
||||
- build_release/tools/src/mg_import_csv
|
||||
- config
|
||||
filename: binaries.tar.gz
|
||||
|
||||
- name: Doxygen documentation
|
||||
cd: docs/doxygen/html
|
||||
archive:
|
||||
- .
|
||||
filename: documentation.tar.gz
|
||||
host: true
|
||||
@@ -1,94 +0,0 @@
|
||||
- name: Diff build
|
||||
project: ^mg-master-diff$
|
||||
commands: |
|
||||
# Activate toolchain
|
||||
export PATH=/opt/toolchain-v1/bin:$PATH
|
||||
export LD_LIBRARY_PATH=/opt/toolchain-v1/lib:/opt/toolchain-v1/lib64
|
||||
|
||||
# Copy untouched repository to parent folder.
|
||||
cd ..
|
||||
cp -r memgraph parent
|
||||
cd memgraph
|
||||
|
||||
# Initialize and create documentation.
|
||||
TIMEOUT=1200 ./init
|
||||
doxygen Doxyfile
|
||||
|
||||
# Remove default build directory.
|
||||
rm -r build
|
||||
|
||||
# Build debug binaries.
|
||||
mkdir build_debug
|
||||
cd build_debug
|
||||
cmake ..
|
||||
TIMEOUT=1200 make -j$THREADS
|
||||
|
||||
# Build coverage binaries.
|
||||
cd ..
|
||||
# TODO: uncomment this build once single node and ha are split
|
||||
# mkdir build_coverage
|
||||
# cd build_coverage
|
||||
# cmake -DTEST_COVERAGE=ON ..
|
||||
# TIMEOUT=1200 make -j$THREADS memgraph__unit
|
||||
ln -s build_debug build_coverage
|
||||
|
||||
# Build release binaries.
|
||||
# cd ..
|
||||
mkdir build_release
|
||||
cd build_release
|
||||
cmake -DCMAKE_BUILD_TYPE=release ..
|
||||
TIMEOUT=1200 make -j$THREADS
|
||||
cd ..
|
||||
|
||||
# Checkout to parent commit and initialize.
|
||||
cd ../parent
|
||||
git checkout HEAD~1
|
||||
TIMEOUT=1200 ./init
|
||||
|
||||
# Build parent release binaries.
|
||||
mkdir build_release
|
||||
cd build_release
|
||||
cmake -DCMAKE_BUILD_TYPE=release ..
|
||||
TIMEOUT=1200 make -j$THREADS memgraph memgraph__macro_benchmark
|
||||
|
||||
|
||||
# release build is the default one
|
||||
- name: Release build
|
||||
commands: |
|
||||
# Activate toolchain
|
||||
export PATH=/opt/toolchain-v1/bin:$PATH
|
||||
export LD_LIBRARY_PATH=/opt/toolchain-v1/lib:/opt/toolchain-v1/lib64
|
||||
|
||||
# Initialize and create documentation.
|
||||
TIMEOUT=1200 ./init
|
||||
doxygen Doxyfile
|
||||
|
||||
# Remove default build directory.
|
||||
rm -r build
|
||||
|
||||
# Build debug binaries.
|
||||
mkdir build_debug
|
||||
cd build_debug
|
||||
cmake ..
|
||||
TIMEOUT=1200 make -j$THREADS
|
||||
|
||||
# Build coverage binaries.
|
||||
cd ..
|
||||
# TODO: uncomment this build once single node and ha are split
|
||||
# mkdir build_coverage
|
||||
# cd build_coverage
|
||||
# cmake -DTEST_COVERAGE=ON ..
|
||||
# TIMEOUT=1200 make -j$THREADS memgraph__unit
|
||||
ln -s build_debug build_coverage
|
||||
|
||||
# Build release binaries.
|
||||
# cd ..
|
||||
mkdir build_release
|
||||
cd build_release
|
||||
cmake -DCMAKE_BUILD_TYPE=Release -DUSE_READLINE=OFF ..
|
||||
TIMEOUT=1200 make -j$THREADS
|
||||
|
||||
# Create Debian package.
|
||||
mkdir output
|
||||
cd output
|
||||
cpack -G DEB --config ../CPackConfig.cmake
|
||||
55
cmake/FindJemalloc.cmake
Normal file
55
cmake/FindJemalloc.cmake
Normal file
@@ -0,0 +1,55 @@
|
||||
# Try to find jemalloc library
|
||||
#
|
||||
# Use this module as:
|
||||
# find_package(Jemalloc)
|
||||
#
|
||||
# or:
|
||||
# find_package(Jemalloc REQUIRED)
|
||||
#
|
||||
# This will define the following variables:
|
||||
#
|
||||
# Jemalloc_FOUND True if the system has the jemalloc library.
|
||||
# Jemalloc_INCLUDE_DIRS Include directories needed to use jemalloc.
|
||||
# Jemalloc_LIBRARIES Libraries needed to link to jemalloc.
|
||||
#
|
||||
# The following cache variables may also be set:
|
||||
#
|
||||
# Jemalloc_INCLUDE_DIR The directory containing jemalloc/jemalloc.h.
|
||||
# Jemalloc_LIBRARY The path to the jemalloc static library.
|
||||
|
||||
find_path(Jemalloc_INCLUDE_DIR NAMES jemalloc/jemalloc.h PATH_SUFFIXES include)
|
||||
|
||||
find_library(Jemalloc_LIBRARY NAMES libjemalloc.a PATH_SUFFIXES lib)
|
||||
|
||||
include(FindPackageHandleStandardArgs)
|
||||
find_package_handle_standard_args(Jemalloc
|
||||
FOUND_VAR Jemalloc_FOUND
|
||||
REQUIRED_VARS
|
||||
Jemalloc_LIBRARY
|
||||
Jemalloc_INCLUDE_DIR
|
||||
)
|
||||
|
||||
if(Jemalloc_FOUND)
|
||||
set(Jemalloc_LIBRARIES ${Jemalloc_LIBRARY})
|
||||
set(Jemalloc_INCLUDE_DIRS ${Jemalloc_INCLUDE_DIR})
|
||||
else()
|
||||
if(Jemalloc_FIND_REQUIRED)
|
||||
message(FATAL_ERROR "Cannot find jemalloc!")
|
||||
else()
|
||||
message(WARNING "jemalloc is not found!")
|
||||
endif()
|
||||
endif()
|
||||
|
||||
if(Jemalloc_FOUND AND NOT TARGET Jemalloc::Jemalloc)
|
||||
add_library(Jemalloc::Jemalloc UNKNOWN IMPORTED)
|
||||
set_target_properties(Jemalloc::Jemalloc
|
||||
PROPERTIES
|
||||
IMPORTED_LOCATION "${Jemalloc_LIBRARY}"
|
||||
INTERFACE_INCLUDE_DIRECTORIES "${Jemalloc_INCLUDE_DIR}"
|
||||
)
|
||||
endif()
|
||||
|
||||
mark_as_advanced(
|
||||
Jemalloc_INCLUDE_DIR
|
||||
Jemalloc_LIBRARY
|
||||
)
|
||||
@@ -39,18 +39,22 @@ modifications:
|
||||
value: "/var/log/memgraph/memgraph.log"
|
||||
override: true
|
||||
|
||||
- name: "bolt_cert_file"
|
||||
value: "/etc/memgraph/ssl/cert.pem"
|
||||
override: true
|
||||
|
||||
- name: "bolt_key_file"
|
||||
value: "/etc/memgraph/ssl/key.pem"
|
||||
- name: "log_level"
|
||||
value: "WARNING"
|
||||
override: true
|
||||
|
||||
- name: "bolt_num_workers"
|
||||
value: ""
|
||||
override: false
|
||||
|
||||
- name: "bolt_cert_file"
|
||||
value: "/etc/memgraph/ssl/cert.pem"
|
||||
override: false
|
||||
|
||||
- name: "bolt_key_file"
|
||||
value: "/etc/memgraph/ssl/key.pem"
|
||||
override: false
|
||||
|
||||
- name: "storage_properties_on_edges"
|
||||
value: "true"
|
||||
override: true
|
||||
@@ -87,15 +91,27 @@ modifications:
|
||||
value: "/usr/lib/memgraph/auth_module/example.py"
|
||||
override: false
|
||||
|
||||
- name: "memory_limit"
|
||||
value: "0"
|
||||
override: true
|
||||
|
||||
- name: "isolation_level"
|
||||
value: "SNAPSHOT_ISOLATION"
|
||||
override: true
|
||||
|
||||
- name: "allow_load_csv"
|
||||
value: "true"
|
||||
override: false
|
||||
|
||||
- name: "storage_parallel_index_recovery"
|
||||
value: "false"
|
||||
override: true
|
||||
|
||||
undocumented:
|
||||
- "flag_file"
|
||||
- "log_file_mode"
|
||||
- "log_link_basename"
|
||||
- "log_prefix"
|
||||
- "max_log_size"
|
||||
- "min_log_level"
|
||||
- "also_log_to_stderr"
|
||||
- "help"
|
||||
- "help_xml"
|
||||
- "stderr_threshold"
|
||||
- "stop_logging_if_full_disk"
|
||||
- "version"
|
||||
- "organization_name"
|
||||
- "license_key"
|
||||
|
||||
@@ -5,12 +5,10 @@ import os
|
||||
import subprocess
|
||||
import sys
|
||||
import textwrap
|
||||
|
||||
import xml.etree.ElementTree as ET
|
||||
|
||||
import yaml
|
||||
|
||||
|
||||
SCRIPT_DIR = os.path.dirname(os.path.realpath(__file__))
|
||||
CONFIG_FILE = os.path.join(SCRIPT_DIR, "flags.yaml")
|
||||
WIDTH = 80
|
||||
@@ -18,14 +16,21 @@ WIDTH = 80
|
||||
|
||||
def wrap_text(s, initial_indent="# "):
|
||||
return "\n#\n".join(
|
||||
map(lambda x: textwrap.fill(x, WIDTH, initial_indent=initial_indent,
|
||||
subsequent_indent="# "), s.split("\n")))
|
||||
map(lambda x: textwrap.fill(x, WIDTH, initial_indent=initial_indent, subsequent_indent="# "), s.split("\n"))
|
||||
)
|
||||
|
||||
|
||||
def extract_flags(binary_path):
|
||||
ret = {}
|
||||
data = subprocess.run([binary_path, "--help-xml"],
|
||||
stdout=subprocess.PIPE).stdout.decode("utf-8")
|
||||
data = subprocess.run([binary_path, "--help-xml"], stdout=subprocess.PIPE).stdout.decode("utf-8")
|
||||
# If something is printed out before the help output, it will break the the
|
||||
# XML parsing -> filter out if something is not XML line because something
|
||||
# can be logged before gflags output (e.g. during the global objects init).
|
||||
# This gets called during memgraph build phase to generate default config
|
||||
# file later installed under /etc/memgraph/memgraph.conf
|
||||
# NOTE: Don't use \n in the gflags description strings.
|
||||
# NOTE: Check here if gflags version changes because of the XML format.
|
||||
data = "\n".join([line for line in data.split("\n") if line.startswith("<")])
|
||||
root = ET.fromstring(data)
|
||||
for child in root:
|
||||
if child.tag == "usage" and child.text.lower().count("warning"):
|
||||
@@ -46,8 +51,7 @@ def apply_config_to_flags(config, flags):
|
||||
for modification in config["modifications"]:
|
||||
name = modification["name"]
|
||||
if name not in flags:
|
||||
print("WARNING: Flag '" + name + "' missing from binary!",
|
||||
file=sys.stderr)
|
||||
print("WARNING: Flag '" + name + "' missing from binary!", file=sys.stderr)
|
||||
continue
|
||||
flags[name]["default"] = modification["value"]
|
||||
flags[name]["override"] = modification["override"]
|
||||
@@ -75,8 +79,9 @@ def extract_sections(flags):
|
||||
else:
|
||||
sections.append((current_section, current_flags))
|
||||
sections.append(("other", other))
|
||||
assert set(sum(map(lambda x: x[1], sections), [])) == set(flags.keys()), \
|
||||
"The section extraction algorithm lost some flags!"
|
||||
assert set(sum(map(lambda x: x[1], sections), [])) == set(
|
||||
flags.keys()
|
||||
), "The section extraction algorithm lost some flags!"
|
||||
return sections
|
||||
|
||||
|
||||
@@ -89,8 +94,7 @@ def generate_config_file(sections, flags):
|
||||
helpstr = flag["meaning"] + " [" + flag["type"] + "]"
|
||||
ret += wrap_text(helpstr) + "\n"
|
||||
prefix = "# " if not flag["override"] else ""
|
||||
ret += prefix + "--" + flag["name"].replace("_", "-") + \
|
||||
"=" + flag["default"] + "\n\n"
|
||||
ret += prefix + "--" + flag["name"].replace("_", "-") + "=" + flag["default"] + "\n\n"
|
||||
ret += "\n"
|
||||
ret += wrap_text(config["footer"])
|
||||
return ret.strip() + "\n"
|
||||
@@ -98,13 +102,9 @@ def generate_config_file(sections, flags):
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("memgraph_binary",
|
||||
help="path to Memgraph binary")
|
||||
parser.add_argument("output_file",
|
||||
help="path where to store the generated Memgraph "
|
||||
"configuration file")
|
||||
parser.add_argument("--config-file", default=CONFIG_FILE,
|
||||
help="path to generator configuration file")
|
||||
parser.add_argument("memgraph_binary", help="path to Memgraph binary")
|
||||
parser.add_argument("output_file", help="path where to store the generated Memgraph " "configuration file")
|
||||
parser.add_argument("--config-file", default=CONFIG_FILE, help="path to generator configuration file")
|
||||
|
||||
args = parser.parse_args()
|
||||
flags = extract_flags(args.memgraph_binary)
|
||||
|
||||
@@ -1,2 +0,0 @@
|
||||
project(mg_customers)
|
||||
add_subdirectory(otto)
|
||||
@@ -1,70 +0,0 @@
|
||||
DISCLAIMER: this is just an initial test, graph might not resemble
|
||||
the graph in the use case at all and the data might be completely
|
||||
irrelevant.
|
||||
|
||||
We tried generating a few sample graphs from the vague description
|
||||
given in the use case doc. Then we tried writing queries that would
|
||||
solve the problem of updating nodes when a leaf value changes,
|
||||
assuming all the internal nodes compute only the sum function.
|
||||
|
||||
We start by creating an index on `id` property to improve initial lookup
|
||||
performance:
|
||||
|
||||
CREATE INDEX ON :Leaf(id)
|
||||
|
||||
Set values of all leafs to 1:
|
||||
|
||||
MATCH (u:Leaf) SET u.value = 1
|
||||
|
||||
Now we initialize the values of all other nodes in the graph:
|
||||
|
||||
MATCH (u) WHERE NOT u:Leaf SET u.value = 0
|
||||
|
||||
MATCH (u) WITH u
|
||||
ORDER BY u.topological_index DESC
|
||||
MATCH (u)-->(v) SET u.value = u.value + v.value
|
||||
|
||||
Change the value of a leaf:
|
||||
|
||||
MATCH (u:Leaf {id: "18"}) SET u.value = 10
|
||||
|
||||
We have to reset all the updated nodes to a neutral element:
|
||||
|
||||
MATCH (u:Leaf {id: "18"})<-[* bfs]-(v)
|
||||
WHERE NOT v:Leaf SET v.value = 0
|
||||
|
||||
Finally, we recalculate their values in topological order:
|
||||
|
||||
MATCH (u:Leaf {id: "18"})<-[* bfs]-(v)
|
||||
WITH v ORDER BY v.topological_index DESC
|
||||
MATCH (v)-->(w) SET v.value = v.value + w.value
|
||||
|
||||
There are a few assumptions made worth pointing out.
|
||||
|
||||
* We are able to efficiently maintain topological order
|
||||
of vertices in the graph.
|
||||
|
||||
* It is possible to accumulate the value of the function. Formally:
|
||||
$$f(x_1, x_2, ..., x_n) = g(...(g(g(x_1, x_2), x_3), ...), x_n)$$
|
||||
|
||||
* There is a neutral element for the operation. However, this
|
||||
assumption can be dropped by introducing an artificial neutral element.
|
||||
|
||||
Number of operations required is proportional to the sum of degrees of affected
|
||||
nodes.
|
||||
|
||||
We generated graph with $10^5$ nodes ($20\ 000$ nodes in each layer), varied the
|
||||
degree distribution in node layers and measured time for the query to execute:
|
||||
|
||||
| # | Root-Category-Group degree | Group-CustomGroup-Leaf degree | Time |
|
||||
|:-:|:---------------------------:|:-----------------------------:|:---------:|
|
||||
| 1 | [1, 10] | [20, 40] | ~1.1s |
|
||||
| 2 | [1, 10] | [50, 100] | ~2.5s |
|
||||
| 3 | [10, 50] | [50, 100] | ~3.3s |
|
||||
|
||||
Due to the structure of the graph, update of a leaf required update of almost
|
||||
all the nodes in the graph so we don't show times required for initial graph
|
||||
update and update after leaf change separately.
|
||||
|
||||
However, there is not enough info on the use case to make the test more
|
||||
sophisticated.
|
||||
@@ -1,71 +0,0 @@
|
||||
---
|
||||
title: "Elliott Management"
|
||||
subtitle: "Proof of Concept Report"
|
||||
header-title: "Elliott Management POC"
|
||||
date: 2017-10-28
|
||||
copyright: "©2017 Memgraph Ltd. All rights reserved."
|
||||
titlepage: true
|
||||
titlepage-color: FFFFFF
|
||||
titlepage-text-color: 101010
|
||||
titlepage-rule-color: 101010
|
||||
titlepage-rule-height: 1
|
||||
...
|
||||
|
||||
# Introduction
|
||||
|
||||
We tried generating a few sample graphs from the description given at
|
||||
the in-person meetings. Then, we tried writing queries that would solve
|
||||
the problem of updating nodes when a leaf value changes, assuming all the
|
||||
internal nodes compute only the sum function.
|
||||
|
||||
# Technical details
|
||||
|
||||
We started by creating an index on `id` property to improve initial lookup
|
||||
performance:
|
||||
|
||||
CREATE INDEX ON :Leaf(id)
|
||||
|
||||
Afther that, we set values of all leafs to 1:
|
||||
|
||||
MATCH (u:Leaf) SET u.value = 1
|
||||
|
||||
We then initialized the values of all other nodes in the graph:
|
||||
|
||||
MATCH (u) WHERE NOT u:Leaf SET u.value = 0
|
||||
|
||||
MATCH (u) WITH u
|
||||
ORDER BY u.topological_index DESC
|
||||
MATCH (u)-->(v) SET u.value = u.value + v.value
|
||||
|
||||
Leaf value change and update of affected values in the graph can
|
||||
be done using three queries. To change the value of a leaf:
|
||||
|
||||
MATCH (u:Leaf {id: "18"}) SET u.value = 10
|
||||
|
||||
Then we had to reset all the affected nodes to the neutral element:
|
||||
|
||||
MATCH (u:Leaf {id: "18"})<-[* bfs]-(v)
|
||||
WHERE NOT v:Leaf SET v.value = 0
|
||||
|
||||
Finally, we recalculated their values in topological order:
|
||||
|
||||
MATCH (u:Leaf {id: "18"})<-[* bfs]-(v)
|
||||
WITH v ORDER BY v.topological_index DESC
|
||||
MATCH (v)-->(w) SET v.value = v.value + w.value
|
||||
|
||||
There are a few assumptions necessary for the approach above to work.
|
||||
|
||||
* We are able to maintain topological order of vertices during graph
|
||||
structure changes.
|
||||
|
||||
* It is possible to accumulate the value of the function. Formally:
|
||||
$$f(x_1, x_2, ..., x_n) = g(...(g(g(x_1, x_2), x_3), ...), x_n)$$
|
||||
|
||||
* There is a neutral element for the operation. However, this
|
||||
assumption can be dropped by introducing an artificial neutral element.
|
||||
|
||||
Above assumptions could be changed, relaxed or dropped, depending on the
|
||||
specifics of the use case.
|
||||
|
||||
Number of operations required is proportional to the sum of degrees of affected
|
||||
nodes.
|
||||
@@ -1,12 +0,0 @@
|
||||
CREATE INDEX ON :Leaf(id);
|
||||
MATCH (u:Leaf) SET u.value = 1;
|
||||
MATCH (u) WHERE NOT u:Leaf SET u.value = 0;
|
||||
MATCH (u) WITH u
|
||||
ORDER BY u.topological_index DESC
|
||||
MATCH (u)-->(v) SET u.value = u.value + v.value;
|
||||
MATCH (u:Leaf {id: "85000"}) SET u.value = 10;
|
||||
MATCH (u:Leaf {id: "85000"})<-[* bfs]-(v)
|
||||
WHERE NOT v:Leaf SET v.value = 0;
|
||||
MATCH (u:Leaf {id: "85000"})<-[* bfs]-(v)
|
||||
WITH v ORDER BY v.topological_index DESC
|
||||
MATCH (v)-->(w) SET v.value = v.value + w.value;
|
||||
@@ -1,125 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
"""Generates a DAG from JSON spec in [config] and outputs nodes to
|
||||
[filename]_nodes, and edges to [filename]_edges in format convertible
|
||||
to Memgraph snapshot.
|
||||
|
||||
Here's an example JSON spec:
|
||||
|
||||
{
|
||||
"layers": [
|
||||
{
|
||||
"name": "A",
|
||||
"sublayers": 1,
|
||||
"degree_lo": 1,
|
||||
"degree_hi": 3,
|
||||
"nodes": 4
|
||||
},
|
||||
{
|
||||
"name": "B",
|
||||
"sublayers": 3,
|
||||
"degree_lo": 2,
|
||||
"degree_hi": 3,
|
||||
"nodes": 10
|
||||
},
|
||||
{
|
||||
"name": "C",
|
||||
"sublayers": 1,
|
||||
"degree_lo": 1,
|
||||
"degree_hi": 1,
|
||||
"nodes": 5
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
Nodes from each layer will be randomly divided into sublayers. A node can
|
||||
only have edges pointing to nodes in lower sublayers of the same layer, or
|
||||
to nodes from the layer directly below it. Out-degree is chosen uniformly
|
||||
random from [degree_lo, degree_hi] interval."""
|
||||
|
||||
import argparse
|
||||
from itertools import accumulate
|
||||
import json
|
||||
import random
|
||||
|
||||
|
||||
def _split_into_sum(n, k):
|
||||
assert 1 <= n, "n should be at least 1"
|
||||
assert k <= n, "k shouldn't be greater than n"
|
||||
xs = [0] + sorted(random.sample(range(1, n), k-1)) + [n]
|
||||
return [b - a for a, b in zip(xs, xs[1:])]
|
||||
|
||||
|
||||
def generate_dag(graph_config, seed=None):
|
||||
random.seed(seed)
|
||||
|
||||
nodes = []
|
||||
edges = []
|
||||
|
||||
layer_lo = 1
|
||||
for layer in graph_config:
|
||||
sublayers = _split_into_sum(layer['nodes'], layer['sublayers'])
|
||||
sub_range = accumulate([layer_lo] + sublayers)
|
||||
layer['sublayer_range'] = list(sub_range)
|
||||
nodes.extend([
|
||||
(u, layer['name'])
|
||||
for u in range(layer_lo, layer_lo + layer['nodes'])
|
||||
])
|
||||
layer_lo += layer['nodes']
|
||||
|
||||
edges = []
|
||||
|
||||
for layer, next_layer in zip(graph_config, graph_config[1:]):
|
||||
degree_lo = layer['degree_lo']
|
||||
degree_hi = layer['degree_hi']
|
||||
|
||||
sub_range = layer['sublayer_range']
|
||||
sub_range_next = next_layer['sublayer_range']
|
||||
|
||||
layer_lo = sub_range[0]
|
||||
next_layer_hi = sub_range_next[-1]
|
||||
|
||||
for sub_lo, sub_hi in zip(sub_range, sub_range[1:]):
|
||||
for u in range(sub_lo, sub_hi):
|
||||
num_edges = random.randint(degree_lo, degree_hi)
|
||||
for _ in range(num_edges):
|
||||
v = random.randint(sub_hi, next_layer_hi - 1)
|
||||
edges.append((u, v))
|
||||
|
||||
for sub_lo, sub_hi in zip(sub_range_next, sub_range_next[1:]):
|
||||
for u in range(sub_lo, sub_hi):
|
||||
v = random.randint(layer_lo, sub_lo - 1)
|
||||
edges.append((v, u))
|
||||
|
||||
return nodes, edges
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
parser = argparse.ArgumentParser(
|
||||
formatter_class=argparse.RawDescriptionHelpFormatter,
|
||||
description=__doc__)
|
||||
parser.add_argument('config', type=str, help='graph config JSON file')
|
||||
parser.add_argument('filename', type=str,
|
||||
help='nodes will be stored to filename_nodes, '
|
||||
'edges to filename_edges')
|
||||
parser.add_argument('--seed', type=int,
|
||||
help='seed for the random generator (default = '
|
||||
'current system time)')
|
||||
args = parser.parse_args()
|
||||
|
||||
with open(args.config, 'r') as f:
|
||||
graph_config = json.loads(f.read())['layers']
|
||||
|
||||
nodes, edges = generate_dag(graph_config, seed=args.seed)
|
||||
|
||||
# print nodes into CSV file
|
||||
with open('{}_nodes'.format(args.filename), 'w') as out:
|
||||
out.write('nodeId:ID(Node),name,topological_index:Int,:LABEL\n')
|
||||
for node_id, layer in nodes:
|
||||
out.write('{0},{1}{0},{0},{1}\n'.format(node_id, layer))
|
||||
|
||||
# print edges into CSV file
|
||||
with open('{}_edges'.format(args.filename), 'w') as out:
|
||||
out.write(':START_ID(Node),:END_ID(Node),:TYPE\n')
|
||||
for u, v in edges:
|
||||
out.write('{},{},child\n'.format(u, v))
|
||||
@@ -1,39 +0,0 @@
|
||||
{
|
||||
"layers": [
|
||||
{
|
||||
"name": "Root",
|
||||
"sublayers": 1,
|
||||
"degree_lo": 1,
|
||||
"degree_hi": 10,
|
||||
"nodes": 20000
|
||||
},
|
||||
{
|
||||
"name": "Category",
|
||||
"sublayers": 5,
|
||||
"degree_lo": 1,
|
||||
"degree_hi": 10,
|
||||
"nodes": 20000
|
||||
},
|
||||
{
|
||||
"name": "Group",
|
||||
"sublayers": 1,
|
||||
"degree_lo": 20,
|
||||
"degree_hi": 40,
|
||||
"nodes": 20000
|
||||
},
|
||||
{
|
||||
"name": "CustomGroup",
|
||||
"sublayers": 15,
|
||||
"degree_lo": 20,
|
||||
"degree_hi": 40,
|
||||
"nodes": 20000
|
||||
},
|
||||
{
|
||||
"name": "Leaf",
|
||||
"sublayers": 1,
|
||||
"degree_lo": 1,
|
||||
"degree_hi": 1,
|
||||
"nodes": 20000
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -1,39 +0,0 @@
|
||||
{
|
||||
"layers": [
|
||||
{
|
||||
"name": "Root",
|
||||
"sublayers": 1,
|
||||
"degree_lo": 1,
|
||||
"degree_hi": 10,
|
||||
"nodes": 20000
|
||||
},
|
||||
{
|
||||
"name": "Category",
|
||||
"sublayers": 5,
|
||||
"degree_lo": 1,
|
||||
"degree_hi": 10,
|
||||
"nodes": 20000
|
||||
},
|
||||
{
|
||||
"name": "Group",
|
||||
"sublayers": 1,
|
||||
"degree_lo": 50,
|
||||
"degree_hi": 100,
|
||||
"nodes": 20000
|
||||
},
|
||||
{
|
||||
"name": "CustomGroup",
|
||||
"sublayers": 15,
|
||||
"degree_lo": 50,
|
||||
"degree_hi": 100,
|
||||
"nodes": 20000
|
||||
},
|
||||
{
|
||||
"name": "Leaf",
|
||||
"sublayers": 1,
|
||||
"degree_lo": 50,
|
||||
"degree_hi": 100,
|
||||
"nodes": 20000
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -1,39 +0,0 @@
|
||||
{
|
||||
"layers": [
|
||||
{
|
||||
"name": "Root",
|
||||
"sublayers": 1,
|
||||
"degree_lo": 10,
|
||||
"degree_hi": 50,
|
||||
"nodes": 20000
|
||||
},
|
||||
{
|
||||
"name": "Category",
|
||||
"sublayers": 5,
|
||||
"degree_lo": 10,
|
||||
"degree_hi": 50,
|
||||
"nodes": 20000
|
||||
},
|
||||
{
|
||||
"name": "Group",
|
||||
"sublayers": 1,
|
||||
"degree_lo": 50,
|
||||
"degree_hi": 100,
|
||||
"nodes": 20000
|
||||
},
|
||||
{
|
||||
"name": "CustomGroup",
|
||||
"sublayers": 15,
|
||||
"degree_lo": 50,
|
||||
"degree_hi": 100,
|
||||
"nodes": 20000
|
||||
},
|
||||
{
|
||||
"name": "Leaf",
|
||||
"sublayers": 1,
|
||||
"degree_lo": 50,
|
||||
"degree_hi": 100,
|
||||
"nodes": 20000
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -1,3 +0,0 @@
|
||||
set(exec_name customers_otto_parallel_connected_components)
|
||||
add_executable(${exec_name} parallel_connected_components.cpp)
|
||||
target_link_libraries(${exec_name} memgraph_lib)
|
||||
@@ -1,118 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
# -*- coding: utf-8 -*-
|
||||
|
||||
"""
|
||||
This script attempts to evaluate the feasibility of using Memgraph for
|
||||
Otto group's usecase. The usecase is finding connected componentes in
|
||||
a large, very sparse graph (cca 220M nodes, 250M edges), based on a dynamic
|
||||
inclusion / exclusion of edges (w.r.t variable parameters and the source node
|
||||
type).
|
||||
|
||||
This implementation defines a random graph with the given number of nodes
|
||||
and edges and looks for connected components using breadth-first expansion.
|
||||
Edges are included / excluded based on a simple expression, only demonstrating
|
||||
possible usage.
|
||||
"""
|
||||
|
||||
from argparse import ArgumentParser
|
||||
import logging
|
||||
from time import time
|
||||
from collections import defaultdict
|
||||
from math import log2
|
||||
from random import randint
|
||||
|
||||
from neo4j.v1 import GraphDatabase
|
||||
|
||||
log = logging.getLogger(__name__)
|
||||
|
||||
|
||||
def generate_graph(sess, node_count, edge_count):
|
||||
# An index that will speed-up edge creation.
|
||||
sess.run("CREATE INDEX ON :Node(id)").consume()
|
||||
|
||||
# Create the given number of nodes with a randomly selected type from:
|
||||
# [0.5, 1.5, 2.5].
|
||||
sess.run(("UNWIND range(0, {} - 1) AS id CREATE "
|
||||
"(:Node {{id: id, type: 0.5 + tointeger(rand() * 3)}})").format(
|
||||
node_count)).consume()
|
||||
|
||||
# Create the given number of edges, each with a 'value' property of
|
||||
# a random [0, 3.0) float. Each edge connects two random nodes, so the
|
||||
# expected node degree is (edge_count * 2 / node_count). Generate edges
|
||||
# so the connectivity is non-uniform (to produce connected components of
|
||||
# various sizes).
|
||||
sess.run(("UNWIND range(0, {0} - 1) AS id WITH id "
|
||||
"MATCH (from:Node {{id: tointeger(rand() * {1})}}), "
|
||||
"(to:Node {{id: tointeger(rand() * {1} * id / {0})}}) "
|
||||
"CREATE (from)-[:Edge {{value: 3 * rand()}}]->(to)").format(
|
||||
edge_count, node_count)).consume()
|
||||
|
||||
|
||||
def get_connected_ids(sess, node_id):
|
||||
# Matches a node with the given ID and returns the IDs of all the nodes
|
||||
# it is connected to. Note that within the BFS lambda expression there
|
||||
# is an expression used to filter out edges expanded over.
|
||||
return sess.run((
|
||||
"MATCH (from:Node {{id: {}}})-"
|
||||
"[*bfs (e, n | abs(from.type - e.value) < 0.80)]-(d) "
|
||||
"RETURN count(*) AS c").format(node_id)).data()[0]['c']
|
||||
|
||||
|
||||
def parse_args():
|
||||
parser = ArgumentParser(description=__doc__)
|
||||
parser.add_argument('--endpoint', type=str, default='localhost:7687',
|
||||
help='Memgraph instance endpoint. ')
|
||||
parser.add_argument('--node-count', type=int, default=1000,
|
||||
help='The number of nodes in the graph')
|
||||
parser.add_argument('--edge-count', type=int, default=1000,
|
||||
help='The number of edges in the graph')
|
||||
parser.add_argument('--sample-count', type=int, default=None,
|
||||
help='The number of samples to take')
|
||||
return parser.parse_args()
|
||||
|
||||
|
||||
def main():
|
||||
args = parse_args()
|
||||
logging.basicConfig(level=logging.INFO)
|
||||
log.info("Memgraph - Otto test database generator")
|
||||
logging.getLogger("neo4j").setLevel(logging.WARNING)
|
||||
|
||||
driver = GraphDatabase.driver(
|
||||
'bolt://' + args.endpoint,
|
||||
auth=("ignored", "ignored"),
|
||||
encrypted=False)
|
||||
sess = driver.session()
|
||||
|
||||
sess.run("MATCH (n) DETACH DELETE n").consume()
|
||||
|
||||
log.info("Generating graph with %s nodes and %s edges...",
|
||||
args.node_count, args.edge_count)
|
||||
generate_graph(sess, args.node_count, args.edge_count)
|
||||
|
||||
# Track which vertices have been found as part of a component.
|
||||
start_time = time()
|
||||
max_query_time = 0
|
||||
log.info("Looking for connected components...")
|
||||
# Histogram of log2 sizes of connected components found.
|
||||
histogram = defaultdict(int)
|
||||
sample_count = args.sample_count if args.sample_count else args.node_count
|
||||
for i in range(sample_count):
|
||||
node_id = randint(0, args.node_count - 1)
|
||||
query_start_time = time()
|
||||
log2_size = int(log2(1 + get_connected_ids(sess, node_id)))
|
||||
max_query_time = max(max_query_time, time() - query_start_time)
|
||||
histogram[log2_size] += 1
|
||||
elapsed = time() - start_time
|
||||
log.info("Connected components found in %.2f sec (avg %.2fms, max %.2fms)",
|
||||
elapsed, elapsed / sample_count * 1000, max_query_time * 1000)
|
||||
log.info("Component size histogram (count | range)")
|
||||
for log2_size, count in sorted(histogram.items()):
|
||||
log.info("\t%5d | %d - %d", count, 2 ** log2_size,
|
||||
2 ** (log2_size + 1) - 1)
|
||||
|
||||
sess.close()
|
||||
driver.close()
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
@@ -1,207 +0,0 @@
|
||||
#include <algorithm>
|
||||
#include <limits>
|
||||
#include <mutex>
|
||||
#include <random>
|
||||
#include <set>
|
||||
#include <stack>
|
||||
#include <thread>
|
||||
|
||||
#include "gflags/gflags.h"
|
||||
#include "glog/logging.h"
|
||||
|
||||
#include "data_structures/union_find.hpp"
|
||||
#include "database/graph_db.hpp"
|
||||
#include "database/graph_db_accessor.hpp"
|
||||
#include "storage/property_value.hpp"
|
||||
#include "threading/sync/spinlock.hpp"
|
||||
#include "utils/bound.hpp"
|
||||
#include "utils/timer.hpp"
|
||||
|
||||
DEFINE_int32(thread_count, 1, "Number of threads");
|
||||
DEFINE_int32(vertex_count, 1000, "Number of vertices");
|
||||
DEFINE_int32(edge_count, 1000, "Number of edges");
|
||||
DECLARE_int32(gc_cycle_sec);
|
||||
|
||||
static const std::string kLabel{"kLabel"};
|
||||
static const std::string kProperty{"kProperty"};
|
||||
|
||||
void GenerateGraph(database::GraphDb &db) {
|
||||
{
|
||||
database::GraphDbAccessor dba{db};
|
||||
dba.BuildIndex(dba.Label(kLabel), dba.Property(kProperty));
|
||||
dba.Commit();
|
||||
}
|
||||
|
||||
// Randomize the sequence of IDs of created vertices and edges to simulate
|
||||
// real-world lack of locality.
|
||||
auto make_id_vector = [](size_t size) {
|
||||
gid::Generator generator{0};
|
||||
std::vector<gid::Gid> ids(size);
|
||||
for (size_t i = 0; i < size; ++i)
|
||||
ids[i] = generator.Next(std::experimental::nullopt);
|
||||
std::random_shuffle(ids.begin(), ids.end());
|
||||
return ids;
|
||||
};
|
||||
|
||||
std::vector<VertexAccessor> vertices;
|
||||
vertices.reserve(FLAGS_vertex_count);
|
||||
{
|
||||
CHECK(FLAGS_vertex_count % FLAGS_thread_count == 0)
|
||||
<< "Thread count must be a factor of vertex count";
|
||||
LOG(INFO) << "Generating " << FLAGS_vertex_count << " vertices...";
|
||||
utils::Timer timer;
|
||||
auto vertex_ids = make_id_vector(FLAGS_vertex_count);
|
||||
|
||||
std::vector<std::thread> threads;
|
||||
SpinLock vertices_lock;
|
||||
for (int i = 0; i < FLAGS_thread_count; ++i) {
|
||||
threads.emplace_back([&db, &vertex_ids, &vertices, &vertices_lock, i]() {
|
||||
database::GraphDbAccessor dba{db};
|
||||
auto label = dba.Label(kLabel);
|
||||
auto property = dba.Property(kProperty);
|
||||
auto batch_size = FLAGS_vertex_count / FLAGS_thread_count;
|
||||
for (int j = i * batch_size; j < (i + 1) * batch_size; ++j) {
|
||||
auto vertex = dba.InsertVertex(vertex_ids[j]);
|
||||
vertex.add_label(label);
|
||||
vertex.PropsSet(property, static_cast<int64_t>(vertex_ids[j]));
|
||||
vertices_lock.lock();
|
||||
vertices.emplace_back(vertex);
|
||||
vertices_lock.unlock();
|
||||
}
|
||||
dba.Commit();
|
||||
});
|
||||
}
|
||||
for (auto &t : threads) t.join();
|
||||
LOG(INFO) << "Generated " << FLAGS_vertex_count << " vertices in "
|
||||
<< timer.Elapsed().count() << " seconds.";
|
||||
}
|
||||
{
|
||||
database::GraphDbAccessor dba{db};
|
||||
for (int i = 0; i < FLAGS_vertex_count; ++i)
|
||||
vertices[i] = *dba.Transfer(vertices[i]);
|
||||
|
||||
LOG(INFO) << "Generating " << FLAGS_edge_count << " edges...";
|
||||
auto edge_ids = make_id_vector(FLAGS_edge_count);
|
||||
std::mt19937 pseudo_rand_gen{std::random_device{}()};
|
||||
std::uniform_int_distribution<> rand_dist{0, FLAGS_vertex_count - 1};
|
||||
auto edge_type = dba.EdgeType("edge");
|
||||
utils::Timer timer;
|
||||
for (int i = 0; i < FLAGS_edge_count; ++i)
|
||||
dba.InsertEdge(vertices[rand_dist(pseudo_rand_gen)],
|
||||
vertices[rand_dist(pseudo_rand_gen)], edge_type,
|
||||
edge_ids[i]);
|
||||
dba.Commit();
|
||||
LOG(INFO) << "Generated " << FLAGS_edge_count << " edges in "
|
||||
<< timer.Elapsed().count() << " seconds.";
|
||||
}
|
||||
}
|
||||
|
||||
auto EdgeIteration(database::GraphDb &db) {
|
||||
database::GraphDbAccessor dba{db};
|
||||
int64_t sum{0};
|
||||
for (auto edge : dba.Edges(false)) sum += edge.from().gid() + edge.to().gid();
|
||||
return sum;
|
||||
}
|
||||
|
||||
auto VertexIteration(database::GraphDb &db) {
|
||||
database::GraphDbAccessor dba{db};
|
||||
int64_t sum{0};
|
||||
for (auto v : dba.Vertices(false))
|
||||
for (auto e : v.out()) sum += e.gid() + e.to().gid();
|
||||
return sum;
|
||||
}
|
||||
|
||||
auto ConnectedComponentsEdges(database::GraphDb &db) {
|
||||
UnionFind<int64_t> connectivity{FLAGS_vertex_count};
|
||||
database::GraphDbAccessor dba{db};
|
||||
for (auto edge : dba.Edges(false))
|
||||
connectivity.Connect(edge.from().gid(), edge.to().gid());
|
||||
return connectivity.Size();
|
||||
}
|
||||
|
||||
auto ConnectedComponentsVertices(database::GraphDb &db) {
|
||||
UnionFind<int64_t> connectivity{FLAGS_vertex_count};
|
||||
database::GraphDbAccessor dba{db};
|
||||
for (auto from : dba.Vertices(false)) {
|
||||
for (auto out_edge : from.out())
|
||||
connectivity.Connect(from.gid(), out_edge.to().gid());
|
||||
}
|
||||
return connectivity.Size();
|
||||
}
|
||||
|
||||
auto ConnectedComponentsVerticesParallel(database::GraphDb &db) {
|
||||
UnionFind<int64_t> connectivity{FLAGS_vertex_count};
|
||||
SpinLock connectivity_lock;
|
||||
|
||||
// Define bounds of vertex IDs for each thread to use.
|
||||
std::vector<PropertyValue> bounds;
|
||||
for (int64_t i = 0; i < FLAGS_thread_count; ++i)
|
||||
bounds.emplace_back(i * FLAGS_vertex_count / FLAGS_thread_count);
|
||||
bounds.emplace_back(std::numeric_limits<int64_t>::max());
|
||||
|
||||
std::vector<std::thread> threads;
|
||||
for (int i = 0; i < FLAGS_thread_count; ++i) {
|
||||
threads.emplace_back(
|
||||
[&connectivity, &connectivity_lock, &bounds, &db, i]() {
|
||||
database::GraphDbAccessor dba{db};
|
||||
for (auto from :
|
||||
dba.Vertices(dba.Label(kLabel), dba.Property(kProperty),
|
||||
utils::MakeBoundInclusive(bounds[i]),
|
||||
utils::MakeBoundExclusive(bounds[i + 1]), false)) {
|
||||
for (auto out_edge : from.out()) {
|
||||
std::lock_guard<SpinLock> lock{connectivity_lock};
|
||||
connectivity.Connect(from.gid(), out_edge.to().gid());
|
||||
}
|
||||
}
|
||||
});
|
||||
}
|
||||
for (auto &t : threads) t.join();
|
||||
return connectivity.Size();
|
||||
}
|
||||
|
||||
auto Expansion(database::GraphDb &db) {
|
||||
std::vector<int> component_ids(FLAGS_vertex_count, -1);
|
||||
int next_component_id{0};
|
||||
std::stack<VertexAccessor> expansion_stack;
|
||||
database::GraphDbAccessor dba{db};
|
||||
for (auto v : dba.Vertices(false)) {
|
||||
if (component_ids[v.gid()] != -1) continue;
|
||||
auto component_id = next_component_id++;
|
||||
expansion_stack.push(v);
|
||||
while (!expansion_stack.empty()) {
|
||||
auto next_v = expansion_stack.top();
|
||||
expansion_stack.pop();
|
||||
if (component_ids[next_v.gid()] != -1) continue;
|
||||
component_ids[next_v.gid()] = component_id;
|
||||
for (auto e : next_v.out()) expansion_stack.push(e.to());
|
||||
for (auto e : next_v.in()) expansion_stack.push(e.from());
|
||||
}
|
||||
}
|
||||
|
||||
return next_component_id;
|
||||
}
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
gflags::ParseCommandLineFlags(&argc, &argv, true);
|
||||
google::InitGoogleLogging(argv[0]);
|
||||
FLAGS_gc_cycle_sec = -1;
|
||||
|
||||
database::SingleNode db;
|
||||
GenerateGraph(db);
|
||||
auto timed_call = [&db](auto callable, const std::string &descr) {
|
||||
LOG(INFO) << "Running " << descr << "...";
|
||||
utils::Timer timer;
|
||||
auto result = callable(db);
|
||||
LOG(INFO) << "\tDone in " << timer.Elapsed().count()
|
||||
<< " seconds, result: " << result;
|
||||
};
|
||||
timed_call(EdgeIteration, "Edge iteration");
|
||||
timed_call(VertexIteration, "Vertex iteration");
|
||||
timed_call(ConnectedComponentsEdges, "Connected components - Edges");
|
||||
timed_call(ConnectedComponentsVertices, "Connected components - Vertices");
|
||||
timed_call(ConnectedComponentsVerticesParallel,
|
||||
"Parallel connected components - Vertices");
|
||||
timed_call(Expansion, "Expansion");
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -1,5 +0,0 @@
|
||||
WITH tointeger(rand() * 40000000) AS from_id
|
||||
MATCH (from:Node {id : from_id}) WITH from
|
||||
MATCH path = (from)-[*bfs..50 (e, n | degree(n) < 50)]->(to) WITH path LIMIT 10000 WHERE to.fraudulent
|
||||
RETURN path, size(path)
|
||||
|
||||
@@ -1,31 +0,0 @@
|
||||
{
|
||||
"indexes":[
|
||||
"Node.id"
|
||||
],
|
||||
"nodes":[
|
||||
{
|
||||
"count":40000000,
|
||||
"labels":[
|
||||
"Node"
|
||||
],
|
||||
"properties":{
|
||||
"id":{
|
||||
"type":"counter",
|
||||
"param":"Node.id"
|
||||
},
|
||||
"fraudulent":{
|
||||
"type":"bernoulli",
|
||||
"param":0.0005
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"edges":[
|
||||
{
|
||||
"count":80000000,
|
||||
"from":"Node",
|
||||
"to":"Node",
|
||||
"type":"Edge"
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -1,20 +0,0 @@
|
||||
{
|
||||
"indexes" : ["Card.id", "Pos.id", "Transaction.fraud_reported"],
|
||||
"nodes" : [
|
||||
{
|
||||
"count_per_worker" : 1250000,
|
||||
"label" : "Card"
|
||||
},
|
||||
{
|
||||
"count_per_worker" : 1250000,
|
||||
"label" : "Pos"
|
||||
},
|
||||
{
|
||||
"count_per_worker" : 2500000,
|
||||
"label" : "Transaction"
|
||||
}
|
||||
],
|
||||
"compromised_pos_probability" : 0.2,
|
||||
"fraud_reported_probability" : 0.1,
|
||||
"hop_probability" : 0.1
|
||||
}
|
||||
230
docs/csv-import-tool/README.md
Normal file
230
docs/csv-import-tool/README.md
Normal file
@@ -0,0 +1,230 @@
|
||||
# CSV Import Tool Documentation
|
||||
|
||||
CSV is a universal and very versatile data format used to store large quantities
|
||||
of data. Each Memgraph database instance has a CSV import tool installed called
|
||||
`mg_import_csv`. The CSV import tool should be used for initial bulk ingestion
|
||||
of data into the database. Upon ingestion, the CSV importer creates a snapshot
|
||||
that will be used by the database to recover its state on its next startup.
|
||||
|
||||
If you are already familiar with the Neo4j bulk import tool, then using the
|
||||
`mg_import_csv` tool should be easy. The CSV import tool is fully compatible
|
||||
with the [Neo4j CSV
|
||||
format](https://neo4j.com/docs/operations-manual/current/tools/import/). If you
|
||||
already have a pipeline set-up for Neo4j, you should only replace `neo4j-admin
|
||||
import` with `mg_import_csv`.
|
||||
|
||||
## CSV File Format
|
||||
|
||||
Each row of a CSV file represents a single entry that should be imported into
|
||||
the database. Both nodes and relationships can be imported into the database
|
||||
using CSV files.
|
||||
|
||||
Each set of CSV files must have a header that describes the data that is stored
|
||||
in the CSV files. Each field in the CSV header is in the format
|
||||
`<name>[:<type>]` which identifies the name that should be used for that column
|
||||
and the type that should be used for that column. The type is optional and
|
||||
defaults to `string` (see the following chapter).
|
||||
|
||||
Each CSV field must be divided using the delimiter and each CSV field can either
|
||||
be quoted or unquoted. When the field is quoted, the first and last character in
|
||||
the field *must* be the quote character. If the field isn't quoted, and a quote
|
||||
character appears in it, it is treated as a regular character. If a quote
|
||||
character appears inside a quoted string then the quote character must be
|
||||
doubled in order to escape it. Line feeds and carriage returns are ignored in
|
||||
the CSV file, also, the file can't contain a NULL character.
|
||||
|
||||
## Properties
|
||||
|
||||
Both nodes and relationships can have properties added to them. When importing
|
||||
properties, the CSV importer uses the name specified in the header of the
|
||||
corresponding CSV column for the name of the property. A property is designated
|
||||
by specifying one of the following types in the header:
|
||||
- `integer`, `int`, `long`, `byte`, `short`: creates an integer property
|
||||
- `float`, `double`: creates a float property
|
||||
- `boolean`, `bool`: creates a boolean property
|
||||
- `string`, `char`: creates a string property
|
||||
|
||||
When importing a boolean value, the CSV field should contain exactly the text
|
||||
`true` to import a `True` boolean value. All other text values are treated as a
|
||||
boolean value `False`.
|
||||
|
||||
If you want to import an array of values, you can do so by appending `[]` to any
|
||||
of the above types. The values of the array are then determined by splitting
|
||||
the raw CSV value using the array delimiter character.
|
||||
|
||||
Assuming that the array delimiter is `;`, the following example:
|
||||
```plaintext
|
||||
first_name,last_name:string,number:integer,aliases:string[]
|
||||
John,Doe,1,Johnny;Jo;J-man
|
||||
Melissa,Doe,2,Mel
|
||||
```
|
||||
|
||||
Will yield these results:
|
||||
```plaintext
|
||||
CREATE ({first_name: "John", last_name: "Doe", number: 1, aliases: ["Johnny", "Jo", "J-man"]});
|
||||
CREATE ({first_name: "Melissa", last_name: "Doe", number: 2, aliases: ["Mel"]});
|
||||
```
|
||||
### Nodes
|
||||
|
||||
When importing nodes, several more types can be specified in the header of the
|
||||
CSV file (along with all property types):
|
||||
- `ID`: id of the node that should be used as the node ID when importing
|
||||
relationships
|
||||
- `LABEL`: designates that the field contains additional labels for the node
|
||||
- `IGNORE`: designates that the field should be ignored
|
||||
|
||||
The `ID` field type sets the internal ID that will be used for the node when
|
||||
creating relationships. It is optional and nodes that don't have an ID value
|
||||
specified will be imported, but can't be connected to any relationships. If you
|
||||
want to save the ID value as a property in the database, just specify a name for
|
||||
the ID (`user_id:ID`). If you just want to use the ID during the import, leave
|
||||
out the name of the field (`:ID`). The `ID` field also supports creating
|
||||
separate ID spaces. The ID space is specified with the ID space name appended
|
||||
to the `ID` type in parentheses (`ID(user)`). That allows you to have the same
|
||||
IDs (by value) for multiple different node files (for example, numbers from 1 to
|
||||
N). The IDs in each ID space will be treated as an independent set of IDs that
|
||||
don't interfere with IDs in another ID space.
|
||||
|
||||
The `LABEL` field type adds additional labels to the node. The value is treated
|
||||
as an array type so that multiple additional labels can be specified for each
|
||||
node. The value is split using the array delimiter (`--array-delimiter` flag).
|
||||
|
||||
### Relationships
|
||||
|
||||
In order to be able to import relationships, you must import the nodes in the
|
||||
same invocation of `mg_import_csv` that is used to import the relationships.
|
||||
|
||||
When importing relationships, several more types can be specified in the header
|
||||
of the CSV file (along with all property types):
|
||||
- `START_ID`: id of the start node that should be connected with the
|
||||
relationship
|
||||
- `END_ID`: id of the end node that should be connected with the relationship
|
||||
- `TYPE`: designates the type of the relationship
|
||||
- `IGNORE`: designates that the field should be ignored
|
||||
|
||||
The `START_ID` field type sets the start node that should be connected with the
|
||||
relationship to the end node. The field *must* be specified and the node ID
|
||||
must be one of the node IDs that were specified in the node CSV files. The name
|
||||
of this field is ignored. If the node ID is in an ID space, you can specify the
|
||||
ID space for the in the same way as for the node ID (`START_ID(user)`).
|
||||
|
||||
The `END_ID` field type sets the end node that should be connected with the
|
||||
relationship to the start node. The field *must* be specified and the node ID
|
||||
must be one of the node IDs that were specified in the node CSV files. The name
|
||||
of this field is ignored. If the node ID is in an ID space, you can specify the
|
||||
ID space for the in the same way as for the node ID (`END_ID(user)`).
|
||||
|
||||
The `TYPE` field type sets the type of the relationship. Each relationship
|
||||
*must* have a relationship type, but it doesn't necessarily need to be specified
|
||||
in the CSV file, it can also be set externally for the whole CSV file. The name
|
||||
of this field is ignored.
|
||||
|
||||
## CSV Importer Flags
|
||||
|
||||
The importer has many command line options that allow you to customize the way
|
||||
the importer loads your data.
|
||||
|
||||
The two main flags that are used to specify the input CSV files are `--nodes`
|
||||
and `--relationships`. Basic description of these flags is provided in the table
|
||||
and more detailed explainion can be found further down bellow.
|
||||
|
||||
|
||||
| Flag | Description |
|
||||
|-----------------------| -------------- |
|
||||
|`--nodes` | Used to specify CSV files that contain the nodes to the importer. |
|
||||
|`--relationships` | Used to specify CSV files that contain the relationships to the importer.|
|
||||
|`--delimiter` | Sets the delimiter that should be used when splitting the CSV fields (default `,`)|
|
||||
|`--quote` | Sets the quote character that should be used to quote a CSV field (default `"`)|
|
||||
|`--array-delimiter` | Sets the delimiter that should be used when splitting array values (default `;`)|
|
||||
|`--id-type` | Specifies which data type should be used to store the supplied <br /> node IDs when storing them as properties (if the field name is supplied). <br /> The supported values are either `STRING` or `INTEGER`. (default `STRING`)|
|
||||
|`--ignore-empty-strings` | Instructs the importer to treat all empty strings as `Null` values <br /> instead of an empty string value (default `false`)|
|
||||
|`--ignore-extra-columns` | Instructs the importer to ignore all columns (instead of raising an error) <br /> that aren't specified after the last specified column in the CSV header. (default `false`) |
|
||||
| `--skip-bad-relationships`| Instructs the importer to ignore all relationships (instead of raising an error) <br /> that refer to nodes that don't exist in the node files. (default `false`) |
|
||||
|`--skip-duplicate-nodes` | Instructs the importer to ignore all duplicate nodes (instead of raising an error). <br /> Duplicate nodes are nodes that have an ID that is the same as another node that was already imported. (default `false`) |
|
||||
| `--trim-strings`| Instructs the importer to trim all of the loaded CSV field values before processing them further. <br /> Trimming the fields removes all leading and trailing whitespace from them. (default `false`) |
|
||||
|
||||
The `--nodes` and `--relationships` flags are used to specify CSV files that
|
||||
contain the nodes and relationships to the importer. Multiple files can be
|
||||
specified in each supplied `--nodes` or `--relationships` flag. Files that are
|
||||
supplied in one `--nodes` or `--relationships` flag are treated by the CSV
|
||||
parser as one big CSV file. Only the first line of the first file is parsed for
|
||||
the CSV header, all other files (and rows) are treated as data. This is useful
|
||||
when you have a very large CSV file and don't want to edit its first line just
|
||||
to add a CSV header. Instead, you can specify the header in a separate file
|
||||
(e.g. `users_header.csv` or `friendships_header.csv`) and have the data intact
|
||||
in the large file (e.g. `users.csv` or `friendships.csv`). Also, you can supply
|
||||
additional labels for each set of node files.
|
||||
|
||||
The format of `--nodes` flag is:
|
||||
`[<label>[:<label>]...=]<file>[,<file>][,<file>]...`. Take note that only the
|
||||
first `<file>` part is mandatory, all other parts of the flag value are
|
||||
optional. Multiple `--nodes` flags can be supplied to describe multiple sets of
|
||||
different node files. For the importer to work, at least one `--nodes` flag
|
||||
*must* be supplied.
|
||||
|
||||
The format of `--relationships` flag is: `[<type>=]<file>[,<file>][,<file>]...`.
|
||||
Take note that only the first `<file>` part is mandatory, all other parts of the
|
||||
flag value are optional. Multiple `--relationships` flags can be supplied to
|
||||
describe multiple sets of different relationship files. The `--relationships`
|
||||
flag isn't mandatory.
|
||||
|
||||
## CSV Parser Logic
|
||||
|
||||
The CSV parser uses the same logic as the standard Python CSV parser. The data
|
||||
is parsed in the same way as the following snippet:
|
||||
|
||||
```python
|
||||
import csv
|
||||
for row in csv.reader(stream, strict=True):
|
||||
# process 'row'
|
||||
```
|
||||
|
||||
Python uses 'excel' as the default dialect when parsing CSV files and the
|
||||
default settings for the CSV parser are:
|
||||
- delimiter: `','`
|
||||
- doublequote: `True`
|
||||
- escapechar: `None`
|
||||
- lineterminator: `'\r\n'`
|
||||
- quotechar: `'"'`
|
||||
- skipinitialspace: `False`
|
||||
|
||||
The above snippet can be expanded to:
|
||||
|
||||
```python
|
||||
import csv
|
||||
for row in csv.reader(stream, delimiter=',', doublequote=True,
|
||||
escapechar=None, lineterminator='\r\n',
|
||||
quotechar='"', skipinitialspace=False,
|
||||
strict=True):
|
||||
# process 'row'
|
||||
```
|
||||
|
||||
For more information about the meaning of the above values, see:
|
||||
https://docs.python.org/3/library/csv.html#csv.Dialect
|
||||
|
||||
## Errors
|
||||
|
||||
1. [Skipping duplicate node with ID '{}'. For more details, visit:
|
||||
memgr.ph/csv-import-tool.](#error-1)
|
||||
2. [Skipping bad relationship with START_ID '{}'. For more details, visit:
|
||||
memgr.ph/csv-import-tool.](#error-2)
|
||||
3. [Skipping bad relationship with END_ID '{}'. For more details, visit:
|
||||
memgr.ph/csv-import-tool.](#error-3)
|
||||
|
||||
## Skipping duplicate node with ID {} {#error-1}
|
||||
|
||||
Duplicate nodes are nodes that have an ID that is the same as another node that
|
||||
was already imported. You can instruct the importer to ignore all duplicate
|
||||
nodes (instead of raising an error) by using the `--skip-duplicate-nodes` flag.
|
||||
|
||||
## Skipping bad relationship with START_ID {} {#error-2}
|
||||
|
||||
A node with the id `START_ID` doesn't exist. You can instruct the importer to
|
||||
ignore all bad relationships (instead of raising an error) that refer to nodes
|
||||
that don't exist in the node files by using the `--skip-bad-relationships` flag.
|
||||
|
||||
## Skipping bad relationship with END_ID {} {#error-3}
|
||||
|
||||
A node with the id `END_ID` doesn't exist. You can instruct the importer to
|
||||
ignore all bad relationships (instead of raising an error) that refer to nodes
|
||||
that don't exist in the node files by using the `--skip-bad-relationships` flag.
|
||||
@@ -1,269 +0,0 @@
|
||||
# Code Review Guidelines
|
||||
|
||||
This chapter describes some of the things you should be on the lookout when
|
||||
reviewing someone else's code.
|
||||
|
||||
## Exceptions
|
||||
|
||||
Although the Google C++ Style Guide forbids exceptions, we do allow them in
|
||||
our codebase. As a reviewer you should watch out for the following.
|
||||
|
||||
The documentation of throwing functions needs to be in-sync with the
|
||||
implementation. This must be enforced recursively. I.e. if a function A now
|
||||
throws a new exception, and the function B uses A, then B needs to handle that
|
||||
exception or have its documentation updated and so on. Naturally, the same
|
||||
applies when an exception is removed.
|
||||
|
||||
Transitive callers of the function which throws a new exception must be OK
|
||||
with that. This ties into the previous point. You need to check that all users
|
||||
of the new exception either handle it correctly or propagate it.
|
||||
|
||||
Exceptions should not escape out of class destructors, because that will
|
||||
terminate the program. The code should be changed so that such cases are not
|
||||
possible.
|
||||
|
||||
Exceptions being thrown in class constructors. Although this is well defined
|
||||
in C++, it usually implies that a constructor is doing too much work and the
|
||||
class construction & initialization should be redesigned. Usual approaches are
|
||||
using the (Static) Factory Method pattern or having some sort of an
|
||||
initialization method that needs to be called after the construction is done.
|
||||
Prefer the Factory Method.
|
||||
|
||||
Don't forget that STL functions may also throw!
|
||||
|
||||
## Pointers & References
|
||||
|
||||
In cases when some code passes a pointer or reference, or if a code stores a
|
||||
pointer or reference you should take a careful look at the following.
|
||||
|
||||
* Lifetime of the pointed to value (this includes both ownership and
|
||||
multithreaded access).
|
||||
* In case of a class, check validity of destructor and move/copy
|
||||
constructors.
|
||||
* Is the pointed to value mutated, if not it should be `const` (`const Type
|
||||
*` or `const Type &`).
|
||||
|
||||
## Allocators & Memory Resources
|
||||
|
||||
With the introduction of polymorphic allocators (C++17 `<memory_resource>` and
|
||||
our `utils/memory.hpp`) we get a more convenient type signatures for
|
||||
containers so as to keep the outward facing API nice. This convenience comes
|
||||
at a cost of less static checks on the type level due to type erasure.
|
||||
|
||||
For example:
|
||||
|
||||
std::pmr::vector<int> first_vec(std::pmr::null_memory_resource());
|
||||
std::pmr::vector<int> second_vec(std::pmr::new_delete_resource());
|
||||
|
||||
second_vec = first_vec // What happens here?
|
||||
|
||||
// Or with our implementation
|
||||
utils::MonotonicBufferResource monotonic_memory(1024);
|
||||
std::vector<int, utils::Allocator<int>> first_vec(&monotonic_memory);
|
||||
std::vector<int, utils::Allocator<int>> second_vec(utils::NewDeleteResource());
|
||||
|
||||
second_vec = first_vec // What happens here?
|
||||
|
||||
In the above, both `first_vec` and `second_vec` have the same type, but have
|
||||
*different* allocators! This can lead to ambiguity when moving or copying
|
||||
elements between them.
|
||||
|
||||
You need to watch out for the following.
|
||||
|
||||
* Swapping can lead to undefined behaviour if the allocators are not equal.
|
||||
* Is the move construction done with the right allocator.
|
||||
* Is the move assignment done correctly, also it may throw an exception.
|
||||
* Is the copy construction done with the right allocator.
|
||||
* Is the copy assignment done correctly.
|
||||
* Using `auto` makes allocator propagation rules rather ambiguous.
|
||||
|
||||
## Classes & Object Oriented Programming
|
||||
|
||||
A common mistake is to use classes, inheritance and "OOP" when it's not
|
||||
needed. This sections shows examples of encountered cases.
|
||||
|
||||
### Classes without (Meaningful) Members
|
||||
|
||||
class MyCoolClass {
|
||||
public:
|
||||
int BeCool(int a, int b) { return a + b; }
|
||||
|
||||
void SaySomethingCool() { std::cout << "Hello!"; }
|
||||
};
|
||||
|
||||
The above class has no members (i.e. state) which affect the behaviour of
|
||||
methods. This class should need not exist, it can be easily replaced with a
|
||||
more modular (and shorter) design -- top level functions.
|
||||
|
||||
int BeCool(int a, int b) { return a + b; }
|
||||
|
||||
void SaySomethingCool() { std::cout << "Hello!"; }
|
||||
|
||||
### Classes with a Single Public Method
|
||||
|
||||
clas MyAwesomeClass {
|
||||
public:
|
||||
MyAwesomeClass(int state) : state_(state) {}
|
||||
|
||||
int GetAwesome() { return GetAwesomeImpl() + 1; }
|
||||
|
||||
private:
|
||||
int state_;
|
||||
|
||||
int GetAwesomeImpl() { return state_; }
|
||||
};
|
||||
|
||||
The above class has a `state_` and even a private method, but there's only one
|
||||
public method -- `GetAwesome`.
|
||||
|
||||
You should check "Does the stored state have any meaningful influence on the
|
||||
public method?", similarly to the previous point.
|
||||
|
||||
In the above case it doesn't, and the class should be replaced with a public
|
||||
function in `.hpp` while the private method should become a private function
|
||||
in `.cpp` (static or in anonymous namespace).
|
||||
|
||||
// hpp
|
||||
int GetAwesome(int state);
|
||||
|
||||
// cpp
|
||||
namespace {
|
||||
int GetAwesomeImpl(int state) { return state; }
|
||||
}
|
||||
int GetAwesome(int state) { return GetAwesomeImpl(state) + 1; }
|
||||
|
||||
A counterexample is when the state is meaningful.
|
||||
|
||||
class Counter {
|
||||
public:
|
||||
Counter(int state) : state_(state) {}
|
||||
|
||||
int Get() { return state_++; }
|
||||
|
||||
private:
|
||||
int state_;
|
||||
};
|
||||
|
||||
But even that could be replaced with a closure.
|
||||
|
||||
auto MakeCounter(int state) {
|
||||
return [state]() mutable { return state++; };
|
||||
}
|
||||
|
||||
### Private Methods
|
||||
|
||||
Instead of private methods, top level functions should be preferred. The
|
||||
reasoning is completely explained in "Effective C++" Item 23 by Scott Meyers.
|
||||
In our codebase, even improvements to compilation times can be noticed if
|
||||
private methods in interface (`.hpp`) files are replaced with top level
|
||||
functions in implementation (`.cpp`) files.
|
||||
|
||||
### Inheritance
|
||||
|
||||
The rule is simple -- if there are no virtual methods (but maybe destructor),
|
||||
then the class should be marked as `final` and never inherited.
|
||||
|
||||
If there are virtual methods (i.e. class is meant to be inherited), make sure
|
||||
that either a public virtual destructor or a protected non-virtual destructor
|
||||
exist. See "Effective C++" Item 7 by Scott Meyers. Also take a look at
|
||||
"Effective C++" Items 32---39 by Scott Meyers.
|
||||
|
||||
An example of how inheritance with no virtual methods is replaced with
|
||||
composition.
|
||||
|
||||
class MyBase {
|
||||
public:
|
||||
virtual ~MyBase() {}
|
||||
|
||||
void DoSomethingBase() { ... }
|
||||
};
|
||||
|
||||
class MyDerived final : public MyBase {
|
||||
public:
|
||||
void DoSomethingNew() { ... DoSomethingBase(); ... }
|
||||
};
|
||||
|
||||
With composition, the above becomes.
|
||||
|
||||
class MyBase final {
|
||||
public:
|
||||
void DoSomethingBase() { ... }
|
||||
};
|
||||
|
||||
class MyDerived final {
|
||||
MyBase base_;
|
||||
|
||||
public:
|
||||
void DoSomethingNew() { ... base_.DoSomethingBase(); ... }
|
||||
};
|
||||
|
||||
The composition approach is preferred as it encapsulates the fact that
|
||||
`MyBase` is used for the implementation and users only interact with the
|
||||
public interface of `MyDerived`. Additionally, you can easily replace `MyBase
|
||||
base_;` with a C++ PIMPL idiom (`std::unique_ptr<MyBase> base_;`) to make the
|
||||
code more modular with regards to compilation.
|
||||
|
||||
More advanced C++ users will recognize that the encapsulation feature of the
|
||||
non-PIMPL composition can be replaced with private inheritance.
|
||||
|
||||
class MyDerived final : private MyBase {
|
||||
public:
|
||||
void DoSomethingNew() { ... MyBase::DoSomethingBase(); ... }
|
||||
};
|
||||
|
||||
One of the common "counterexample" is the ability to store objects of
|
||||
different type in a container or pass them to a function. Unfortunately, this
|
||||
is not that good of a design. For example.
|
||||
|
||||
class MyBase {
|
||||
... // No virtual methods (but the destructor)
|
||||
};
|
||||
|
||||
class MyFirstClass final : public MyBase { ... };
|
||||
|
||||
class MySecondClass final : public MyBase { ... };
|
||||
|
||||
std::vector<std::unique_ptr<MyBase>> first_and_second_classes;
|
||||
first_and_second_classes.push_back(std::make_unique<MyFirstClass>());
|
||||
first_and_second_classes.push_back(std::make_unique<MySecondClass>());
|
||||
|
||||
void FunctionOnFirstOrSecond(const MyBase &first_or_second, ...) { ... }
|
||||
|
||||
With C++17, the containers for different types should be implemented with
|
||||
`std::variant`, and as before the functions can be templated.
|
||||
|
||||
class MyFirstClass final { ... };
|
||||
|
||||
class MySecondClass final { ... };
|
||||
|
||||
std::vector<std::variant<MyFirstClass, MySecondClass>> first_and_second_classes;
|
||||
// Notice no heap allocation, since we don't store a pointer
|
||||
first_and_second_classes.emplace_back(MyFirstClass());
|
||||
first_and_second_classes.emplace_back(MySecondClass());
|
||||
|
||||
// You can also use `std::variant` here instead of template
|
||||
template <class TFirstOrSecond>
|
||||
void FunctionOnFirstOrSecond(const TFirstOrSecond &first_or_second, ...) { ... }
|
||||
|
||||
Naturally, if the base class has meaningful virtual methods (i.e. other than
|
||||
destructor) it maybe is OK to use inheritance but also consider alternatives.
|
||||
See "Effective C++" Items 32---39 by Scott Meyers.
|
||||
|
||||
### Multiple Inheritance
|
||||
|
||||
Multiple inheritance should not be used unless all base classes are pure
|
||||
interface classes. This decision is inherited from [Google C++ Style
|
||||
Guide](https://google.github.io/styleguide/cppguide.html#Inheritance). For
|
||||
example on how to design with and around multiple inheritance refer to
|
||||
"Effective C++" Item 40 by Scott Meyers.
|
||||
|
||||
Naturally, if there *really* is no better design, then multiple inheritance is
|
||||
allowed. An example of this can be found in our codebase when inheriting
|
||||
Visitor classes (though even that could be replaced with `std::variant` for
|
||||
example).
|
||||
|
||||
## Code Format & Style
|
||||
|
||||
If something doesn't conform to our code formatting and style, just refer the
|
||||
author to either [C++ Style](cpp-code-conventions.md) or [Other Code
|
||||
Conventions](other-code-conventions.md).
|
||||
@@ -1,263 +0,0 @@
|
||||
# C++ Code Conventions
|
||||
|
||||
This chapter describes code conventions which should be followed when writing
|
||||
C++ code.
|
||||
|
||||
## Code Style
|
||||
|
||||
Memgraph uses the
|
||||
[Google Style Guide for C++](https://google.github.io/styleguide/cppguide.html)
|
||||
in most of its code. You should follow them whenever writing new code.
|
||||
Besides following the style guide, take a look at
|
||||
[Code Review Guidelines](code-review.md) for common design issues and pitfalls
|
||||
with C++ as well as [Required Reading](required-reading.md).
|
||||
|
||||
### Often Overlooked Style Conventions
|
||||
|
||||
#### Pointers & References
|
||||
|
||||
References provide a shorter syntax for accessing members and better declare
|
||||
the intent that a pointer *should* not be `nullptr`. They do not prevent
|
||||
accessing a `nullptr` and obfuscate the client/calling code because the
|
||||
reference argument is passed just like a value. Errors with such code have
|
||||
been very difficult to debug. Therefore, pointers are always used. They will
|
||||
not prevent bugs but will make some of them more obvious when reading code.
|
||||
|
||||
The only time a reference can be used is if it is `const`. Note that this
|
||||
kind of reference is not allowed if it is stored somewhere, i.e. in a class.
|
||||
You should use a pointer to `const` then. The primary reason being is that
|
||||
references obscure the semantics of moving an object, thus making bugs with
|
||||
references pointing to invalid memory harder to track down.
|
||||
|
||||
[Style guide reference](https://google.github.io/styleguide/cppguide.html#Reference_Arguments)
|
||||
|
||||
#### Constructors & RAII
|
||||
|
||||
RAII (Resource Acquisition is Initialization) is a nice mechanism for managing
|
||||
resources. It is especially useful when exceptions are used, such as in our
|
||||
code. Unfortunately, they do have 2 major downsides.
|
||||
|
||||
* Only exceptions can be used for to signal failure.
|
||||
* Calls to virtual methods are not resolved as expected.
|
||||
|
||||
For those reasons the style guide recommends minimal work that cannot fail.
|
||||
Using virtual methods or doing a lot more should be delegated to some form of
|
||||
`Init` method, possibly coupled with static factory methods. Similar rules
|
||||
apply to destructors, which are not allowed to even throw exceptions.
|
||||
|
||||
[Style guide reference](https://google.github.io/styleguide/cppguide.html#Doing_Work_in_Constructors)
|
||||
|
||||
### Additional Style Conventions
|
||||
|
||||
Old code may have broken Google C++ Style accidentally, but the new code
|
||||
should adhere to it as close as possible. We do have some exceptions
|
||||
to Google style as well as additions for unspecified conventions.
|
||||
|
||||
#### Using C++ Exceptions
|
||||
|
||||
Unlike Google, we do not forbid using exceptions.
|
||||
|
||||
But, you should be very careful when using them and introducing new ones. They
|
||||
are indeed handy, but cause problems with understanding the control flow since
|
||||
exceptions are another form of `goto`. It also becomes very hard to determine
|
||||
that the program is in correct state after the stack is unwound and the thrown
|
||||
exception handled. Other than those issues, throwing exceptions in destructors
|
||||
will terminate the program. The same will happen if a thread doesn't handle an
|
||||
exception even though it is not the main thread.
|
||||
|
||||
[Style guide reference](https://google.github.io/styleguide/cppguide.html#Exceptions)
|
||||
|
||||
In general, when introducing a new exception, either via `throw` statement or
|
||||
calling a function which throws, you must examine all transitive callers and
|
||||
update their implementation and/or documentation.
|
||||
|
||||
#### Assertions
|
||||
|
||||
We use `CHECK` and `DCHECK` macros from glog library. You are encouraged to
|
||||
use them as often as possible to both document and validate various pre and
|
||||
post conditions of a function.
|
||||
|
||||
`CHECK` remains even in release build and should be preferred over it's cousin
|
||||
`DCHECK` which only exists in debug builds. The primary reason is that you
|
||||
want to trigger assertions in release builds in case the tests didn't
|
||||
completely validate all code paths. It is better to fail fast and crash the
|
||||
program, than to leave it in undefined state and potentially corrupt end
|
||||
user's work. In cases when profiling shows that `CHECK` is causing visible
|
||||
slowdown you should switch to `DCHECK`.
|
||||
|
||||
#### Template Parameter Naming
|
||||
|
||||
Template parameter names should start with capital letter 'T' followed by a
|
||||
short descriptive name. For example:
|
||||
|
||||
```cpp
|
||||
template <typename TKey, typename TValue>
|
||||
class KeyValueStore
|
||||
```
|
||||
|
||||
## Code Formatting
|
||||
|
||||
You should install `clang-format` and run it on code you change or add. The
|
||||
root of Memgraph's project contains the `.clang-format` file, which specifies
|
||||
how formatting should behave. Running `clang-format -style=file` in the
|
||||
project's root will read the file and behave as expected. For ease of use, you
|
||||
should integrate formatting with your favourite editor.
|
||||
|
||||
The code formatting isn't enforced, because sometimes manual formatting may
|
||||
produce better results. Though, running `clang-format` is strongly encouraged.
|
||||
|
||||
## Documentation
|
||||
|
||||
Besides following the comment guidelines from [Google Style
|
||||
Guide](https://google.github.io/styleguide/cppguide.html#Comments), your
|
||||
documentation of the public API should be
|
||||
[Doxygen](https://github.com/doxygen/doxygen) compatible. For private parts of
|
||||
the code or for comments accompanying the implementation, you are free to
|
||||
break doxygen compatibility. In both cases, you should write your
|
||||
documentation as full sentences, correctly written in English.
|
||||
|
||||
## Doxygen
|
||||
|
||||
To start a Doxygen compatible documentation string, you should open your
|
||||
comment with either a JavaDoc style block comment (`/**`) or a line comment
|
||||
containing 3 slashes (`///`). Take a look at the 2 examples below.
|
||||
|
||||
### Block Comment
|
||||
|
||||
```cpp
|
||||
/**
|
||||
* One sentence, brief description.
|
||||
*
|
||||
* Long form description.
|
||||
*/
|
||||
```
|
||||
|
||||
### Line Comment
|
||||
|
||||
```cpp
|
||||
/// One sentence, brief description.
|
||||
///
|
||||
/// Long form description.
|
||||
```
|
||||
|
||||
If you only have a brief description, you may collapse the documentation into
|
||||
a single line.
|
||||
|
||||
### Block Comment
|
||||
|
||||
```cpp
|
||||
/** Brief description. */
|
||||
```
|
||||
|
||||
### Line Comment
|
||||
|
||||
```cpp
|
||||
/// Brief description.
|
||||
```
|
||||
|
||||
Whichever style you choose, keep it consistent across the whole file.
|
||||
|
||||
Doxygen supports various commands in comments, such as `@file` and `@param`.
|
||||
These help Doxygen to render specified things differently or to track them for
|
||||
cross referencing. If you want to learn more, take a look at these two links:
|
||||
|
||||
* http://www.stack.nl/~dimitri/doxygen/manual/docblocks.html
|
||||
* http://www.stack.nl/~dimitri/doxygen/manual/commands.html
|
||||
|
||||
## Examples
|
||||
|
||||
Below are a few examples of documentation from the codebase.
|
||||
|
||||
### Function
|
||||
|
||||
```cpp
|
||||
/**
|
||||
* Removes whitespace characters from the start and from the end of a string.
|
||||
*
|
||||
* @param s String that is going to be trimmed.
|
||||
*
|
||||
* @return Trimmed string.
|
||||
*/
|
||||
inline std::string Trim(const std::string &s);
|
||||
```
|
||||
|
||||
### Class
|
||||
|
||||
```cpp
|
||||
/** Base class for logical operators.
|
||||
*
|
||||
* Each operator describes an operation, which is to be performed on the
|
||||
* database. Operators are iterated over using a @c Cursor. Various operators
|
||||
* can serve as inputs to others and thus a sequence of operations is formed.
|
||||
*/
|
||||
class LogicalOperator
|
||||
: public ::utils::Visitable<HierarchicalLogicalOperatorVisitor> {
|
||||
public:
|
||||
/** Constructs a @c Cursor which is used to run this operator.
|
||||
*
|
||||
* @param GraphDbAccessor Used to perform operations on the database.
|
||||
*/
|
||||
virtual std::unique_ptr<Cursor> MakeCursor(GraphDbAccessor &db) const = 0;
|
||||
|
||||
/** Return @c Symbol vector where the results will be stored.
|
||||
*
|
||||
* Currently, outputs symbols are only generated in @c Produce operator.
|
||||
* @c Skip, @c Limit and @c OrderBy propagate the symbols from @c Produce (if
|
||||
* it exists as input operator). In the future, we may want this method to
|
||||
* return the symbols that will be set in this operator.
|
||||
*
|
||||
* @param SymbolTable used to find symbols for expressions.
|
||||
* @return std::vector<Symbol> used for results.
|
||||
*/
|
||||
virtual std::vector<Symbol> OutputSymbols(const SymbolTable &) const {
|
||||
return std::vector<Symbol>();
|
||||
}
|
||||
|
||||
virtual ~LogicalOperator() {}
|
||||
};
|
||||
```
|
||||
|
||||
### File Header
|
||||
|
||||
```cpp
|
||||
/// @file visitor.hpp
|
||||
///
|
||||
/// This file contains the generic implementation of visitor pattern.
|
||||
///
|
||||
/// There are 2 approaches to the pattern:
|
||||
///
|
||||
/// * classic visitor pattern using @c Accept and @c Visit methods, and
|
||||
/// * hierarchical visitor which also uses @c PreVisit and @c PostVisit
|
||||
/// methods.
|
||||
///
|
||||
/// Classic Visitor
|
||||
/// ===============
|
||||
///
|
||||
/// Explanation on the classic visitor pattern can be found from many
|
||||
/// sources, but here is the link to hopefully most easily accessible
|
||||
/// information: https://en.wikipedia.org/wiki/Visitor_pattern
|
||||
///
|
||||
/// The idea behind the generic implementation of classic visitor pattern is to
|
||||
/// allow returning any type via @c Accept and @c Visit methods. Traversing the
|
||||
/// class hierarchy is relegated to the visitor classes. Therefore, visitor
|
||||
/// should call @c Accept on children when visiting their parents. To implement
|
||||
/// such a visitor refer to @c Visitor and @c Visitable classes.
|
||||
///
|
||||
/// Hierarchical Visitor
|
||||
/// ====================
|
||||
///
|
||||
/// Unlike the classic visitor, the intent of this design is to allow the
|
||||
/// visited structure itself to control the traversal. This way the internal
|
||||
/// children structure of classes can remain private. On the other hand,
|
||||
/// visitors may want to differentiate visiting composite types from leaf types.
|
||||
/// Composite types are those which contain visitable children, unlike the leaf
|
||||
/// nodes. Differentiation is accomplished by providing @c PreVisit and @c
|
||||
/// PostVisit methods, which should be called inside @c Accept of composite
|
||||
/// types. Regular @c Visit is only called inside @c Accept of leaf types.
|
||||
/// To implement such a visitor refer to @c CompositeVisitor, @c LeafVisitor and
|
||||
/// @c Visitable classes.
|
||||
///
|
||||
/// Implementation of hierarchical visiting is modelled after:
|
||||
/// http://wiki.c2.com/?HierarchicalVisitorPattern
|
||||
```
|
||||
|
||||
@@ -1,288 +0,0 @@
|
||||
// dot -Tpng dependencies.dot -o /path/to/output.png
|
||||
|
||||
// TODO (buda): Put PropertyValueStore to storage namespace
|
||||
|
||||
digraph {
|
||||
// At the beginning of each block there is a default style for that block
|
||||
label="Memgraph Dependencies Diagram"; fontname="Roboto Bold"; fontcolor=black;
|
||||
fontsize=26; labelloc=top; labeljust=right;
|
||||
compound=true; // If true, allow edges between clusters
|
||||
rankdir=TB; // Alternatives: LR
|
||||
node [shape=record fontname="Roboto", fontsize=12, fontcolor=white];
|
||||
edge [color="#B5AFB7"];
|
||||
|
||||
// -- Legend --
|
||||
// dir=both arrowtail=diamond arrowhead=vee -> group ownership
|
||||
// dir=both arrowtail=none, arrowhead=vee -> ownership; stack or uptr
|
||||
|
||||
subgraph cluster_tcp_end_client_communication {
|
||||
label="TCP End Client Communication"; fontsize=14;
|
||||
node [style=filled, color="#DD2222" fillcolor="#DD2222"];
|
||||
|
||||
// Owned elements
|
||||
"communication::Server";
|
||||
"io::network::Socket";
|
||||
|
||||
// Intracluster connections
|
||||
"communication::Server" -> "io::network::Socket"
|
||||
[label="socket_" dir=both arrowtail=none arrowhead=vee];
|
||||
}
|
||||
|
||||
subgraph cluster_bolt_server {
|
||||
label="Bolt Server"; fontsize=14;
|
||||
node [style=filled, color="#62A2CA" fillcolor="#62A2CA"];
|
||||
|
||||
// Owned elements
|
||||
"communication::bolt::SessionData";
|
||||
"communication::bolt::Session";
|
||||
"communication::bolt::Encoder";
|
||||
"communication::bolt::Decoder";
|
||||
|
||||
// Intracluster connections
|
||||
"communication::bolt::Session" -> "communication::bolt::Encoder"
|
||||
[label="encoder_", dir=both arrowtail=none, arrowhead=vee];
|
||||
"communication::bolt::Session" -> "communication::bolt::Decoder"
|
||||
[label="decoder_", dir=both arrowtail=none, arrowhead=vee];
|
||||
}
|
||||
|
||||
subgraph cluster_opencypher_engine {
|
||||
label="openCypher Engine"; fontsize=14;
|
||||
node [style=filled, color="#68BDF6" fillcolor="#68BDF6"];
|
||||
|
||||
// Owned Elements
|
||||
"query::Interpreter";
|
||||
"query::AstTreeStorage";
|
||||
"query::TypedValue"
|
||||
"query::Path";
|
||||
"query::Simbol";
|
||||
"query::Context";
|
||||
"query::ExpressionEvaluator";
|
||||
"query::Frame";
|
||||
"query::SymbolTable";
|
||||
"query::plan::LogicalOperator";
|
||||
"query::plan::Cursor";
|
||||
"query::plan::CostEstimator";
|
||||
|
||||
// Intracluster connections
|
||||
"query::Interpreter" -> "query::AstTreeStorage"
|
||||
[label="ast_cache" dir=both arrowtail=diamond arrowhead=vee];
|
||||
"query::TypedValue" -> "query::Path";
|
||||
"query::plan::Cursor" -> "query::Frame";
|
||||
"query::plan::Cursor" -> "query::Context";
|
||||
"query::plan::LogicalOperator" -> "query::Symbol";
|
||||
"query::plan::LogicalOperator" -> "query::SymbolTable";
|
||||
"query::plan::LogicalOperator" -> "query::plan::Cursor";
|
||||
}
|
||||
|
||||
|
||||
subgraph cluster_storage {
|
||||
label="Storage" fontsize=14;
|
||||
node [style=filled, color="#FB6E00" fillcolor="#FB6E00"];
|
||||
|
||||
// Owned Elements
|
||||
"database::GraphDb";
|
||||
"database::GraphDbAccessor";
|
||||
"storage::Record";
|
||||
"storage::Vertex";
|
||||
"storage::Edge";
|
||||
"storage::RecordAccessor";
|
||||
"storage::VertexAccessor";
|
||||
"storage::EdgeAccessor";
|
||||
"storage::Common";
|
||||
"storage::Label";
|
||||
"storage::EdgeType";
|
||||
"storage::Property";
|
||||
"storage::compression";
|
||||
"storage::SingleNodeConcurrentIdMapper";
|
||||
"storage::Location";
|
||||
"storage::StorageTypesLocation";
|
||||
"PropertyValueStore";
|
||||
"storage::RecordLock";
|
||||
"mvcc::Version";
|
||||
"mvcc::Record";
|
||||
"mvcc::VersionList";
|
||||
|
||||
// Intracluster connections
|
||||
"storage::VertexAccessor" -> "storage::RecordAccessor"
|
||||
[arrowhead=onormal];
|
||||
"storage::EdgeAccessor" -> "storage::RecordAccessor"
|
||||
[arrowhead=onormal];
|
||||
"storage::RecordAccessor" -> "database::GraphDbAccessor"
|
||||
[style=dashed arrowhead=vee];
|
||||
"storage::Vertex" -> "mvcc::Record"
|
||||
[arrowhead=onormal];
|
||||
"storage::Edge" -> "mvcc::Record"
|
||||
[arrowhead=onormal];
|
||||
"storage::Edge" -> "PropertyValueStore"
|
||||
[arrowhead=vee];
|
||||
"storage::Vertex" -> "PropertyValueStore"
|
||||
[arrowhead=vee];
|
||||
"storage::Edge" -> "mvcc::VersionList"
|
||||
[label="from,to" arrowhead=vee style=dashed];
|
||||
"storage::VertexAccessor" -> "storage::Vertex"
|
||||
[arrowhead=vee];
|
||||
"storage::EdgeAccessor" -> "storage::Edge"
|
||||
[arrowhead=vee];
|
||||
"storage::SingleNodeConcurrentIdMapper" -> "storage::StorageTypesLocation"
|
||||
[arrowhead=vee];
|
||||
"storage::StorageTypesLocation" -> "storage::Location"
|
||||
[arrowhead=vee];
|
||||
"storage::Storage" -> "storage::StorageTypesLocation"
|
||||
[arrowhead=vee];
|
||||
"storage::Property" -> "storage::Common"
|
||||
[arrowhead=onormal];
|
||||
"storage::Label" -> "storage::Common"
|
||||
[arrowhead=onormal];
|
||||
"storage::EdgeType" -> "storage::Common"
|
||||
[arrowhead=onormal];
|
||||
"storage::Property" -> "storage::Location"
|
||||
[arrowhead=vee];
|
||||
"PropertyValueStore" -> "storage::Property"
|
||||
[arrowhead=vee];
|
||||
"PropertyValueStore" -> "storage::Location"
|
||||
[arrowhead=vee];
|
||||
"database::GraphDbAccessor" -> "database::GraphDb"
|
||||
[arrowhead=vee];
|
||||
"database::GraphDbAccessor" -> "tx::TransactionId"
|
||||
[arrowhead=vee];
|
||||
"mvcc::VersionList" -> "storge::RecordLock"
|
||||
[label="lock" arrowhead=vee];
|
||||
"mvcc::VersionList" -> "mvcc::Record"
|
||||
[label="head" arrowhead=vee];
|
||||
"mvcc::Record" -> "mvcc::Version"
|
||||
[arrowhead=onormal];
|
||||
|
||||
// Explicit positioning
|
||||
{rank=same;
|
||||
"database::GraphDbAccessor";
|
||||
"storage::VertexAccessor";
|
||||
"storage::EdgeAccessor";}
|
||||
{rank=same;
|
||||
"storage::Common";
|
||||
"storage::compression";}
|
||||
}
|
||||
|
||||
subgraph cluster_properties_on_disk {
|
||||
label="Properties on Disk" fontsize=14;
|
||||
node [style=filled, color="#102647" fillcolor="#102647"];
|
||||
|
||||
// Owned Elements
|
||||
"storage::KVStore";
|
||||
"rocksdb";
|
||||
|
||||
// Intracluster connections
|
||||
"storage::KVStore" -> "rocksdb";
|
||||
}
|
||||
|
||||
subgraph cluster_distributed {
|
||||
label="Distributed" fontsize=14;
|
||||
node [style=filled, color="#FFC500" fillcolor="#FFC500"];
|
||||
|
||||
// Owned Elements
|
||||
"distributed::DataManager";
|
||||
"distributed::DataRpcClients";
|
||||
|
||||
// Intracluster connections
|
||||
"distributed::DataManager" -> "distributed::DataRpcClients"
|
||||
[arrowhead=vee];
|
||||
"storage::RecordAccessor" -> "distributed::DataManager"
|
||||
[style=dashed arrowhead=vee];
|
||||
}
|
||||
|
||||
subgraph cluster_dynamic_partitioning {
|
||||
label="Dynamic Partitioning" fontsize=14;
|
||||
node [style=filled, color="#720096" fillcolor="#720096"];
|
||||
|
||||
// Owned Elements
|
||||
"DynamicPartitioner";
|
||||
}
|
||||
|
||||
subgraph cluster_security {
|
||||
label="Security" fontsize=14;
|
||||
node [style=filled, color="#857F87" fillcolor="#857F87"];
|
||||
|
||||
// Owned Elements
|
||||
"Communication Encryption";
|
||||
"Data Encryption";
|
||||
"Access Control";
|
||||
"Audit Logging";
|
||||
}
|
||||
|
||||
subgraph cluster_web_dashboard {
|
||||
label="Dashaboard" fontsize=14;
|
||||
node [style=filled, color="#FF0092" fillcolor="#FF0092"];
|
||||
|
||||
// Owned Elements
|
||||
"Memgraph Ops / Memgraph Cockpit";
|
||||
}
|
||||
|
||||
subgraph cluster_rpc {
|
||||
label="RPC" fontsize=14;
|
||||
node [style=filled, color="#857F87" fillcolor="#857F87"];
|
||||
|
||||
// Owned Elements
|
||||
"communication::rpc::Server";
|
||||
"communication::rpc::Client";
|
||||
}
|
||||
|
||||
subgraph cluster_ingestion {
|
||||
label="Ingestion" fontsize=14;
|
||||
node [style=filled, color="#0B6D88" fillcolor="#0B6D88"];
|
||||
|
||||
// Owned Elements
|
||||
"Extract";
|
||||
"Transform";
|
||||
"Load";
|
||||
"Amazon S3";
|
||||
"Kafka";
|
||||
|
||||
// Intracluster connections
|
||||
"Extract" -> "Amazon S3";
|
||||
"Extract" -> "Kafka";
|
||||
|
||||
// Explicit positioning
|
||||
{rank=same;"Extract";"Transform";"Load";}
|
||||
}
|
||||
|
||||
// -- Intercluster connections --
|
||||
// cluster_tcp_end_client_communication -- cluster_bolt_server
|
||||
"communication::Server" -> "communication::bolt::SessionData" [color=black];
|
||||
"communication::Server" -> "communication::bolt::Session" [color=black];
|
||||
// cluster_bolt_server -> cluster_storage
|
||||
"communication::bolt::SessionData" -> "database::GraphDb" [color=red];
|
||||
"communication::bolt::Session" -> "database::GraphDbAccessor" [color=red];
|
||||
// cluster_bolt_server -> cluster_opencypher_engine
|
||||
"communication::bolt::SessionData" -> "query::Interpreter" [color=red];
|
||||
// cluster_opencypher_engine -- cluster_storage
|
||||
"query::Interpreter" -> "database::GraphDbAccessor" [color=black];
|
||||
"query::Interpreter" -> "storage::VertexAccessor" [color=black];
|
||||
"query::Interpreter" -> "storage::EdgeAccessor" [color=black];
|
||||
"query::TypedValue" -> "storage::VertexAccessor" [color=black];
|
||||
"query::TypedValue" -> "storage::EdgeAccessor" [color=black];
|
||||
"query::Path" -> "storage::VertexAccessor"
|
||||
[label="vertices" dir=both arrowtail=diamond arrowhead=vee color=black];
|
||||
"query::Path" -> "storage::EdgeAccessor"
|
||||
[label="edges" dir=both arrowtail=diamond arrowhead=vee color=black];
|
||||
"query::plan::LogicalOperator" -> "database::GraphDbAccessor"
|
||||
[color=black arrowhead=vee];
|
||||
// cluster_distributed -- cluster_storage
|
||||
"distributed::DataManager" -> "database::GraphDb"
|
||||
[arrowhead=vee style=dashed color=red];
|
||||
"distributed::DataManager" -> "tx::TransactionId"
|
||||
[label="ves_caches_key" dir=both arrowhead=none arrowtail=diamond
|
||||
color=red];
|
||||
"distributed::DataManager" -> "storage::Vertex"
|
||||
[label="vertices_caches" dir=both arrowhead=none arrowtail=diamond
|
||||
color=red];
|
||||
"distributed::DataManager" -> "storage::Edge"
|
||||
[label="edges_caches" dir=both arrowhead=none arrowtail=diamond
|
||||
color=red];
|
||||
// cluster_storage -- cluster_properties_on_disk
|
||||
"PropertyValueStore" -> "storage::KVStore"
|
||||
[label="static" arrowhead=vee color=black];
|
||||
// cluster_dynamic_partitioning -- cluster_storage
|
||||
"database::GraphDb" -> "DynamicPartitioner"
|
||||
[arrowhead=vee color=red];
|
||||
"DynamicPartitioner" -> "database::GraphDbAccessor"
|
||||
[arrowhead=vee color=black];
|
||||
}
|
||||
@@ -1,22 +0,0 @@
|
||||
digraph {
|
||||
// label="Dynamig Graph Partitioning";
|
||||
fontname="Roboto Bold"; fontcolor=black;
|
||||
fontsize=26; labelloc=top; labeljust=center;
|
||||
compound=true; // If true, allow edges between clusters
|
||||
rankdir=TB; // Alternatives: LR
|
||||
node [shape=record fontname="Roboto", fontsize=12, fontcolor=white
|
||||
style=filled, color="#FB6E00" fillcolor="#FB6E00"];
|
||||
edge [color="#B5AFB7"];
|
||||
|
||||
"distributed::DistributedGraphDb" -> "distributed::TokenSharingRpcServer";
|
||||
|
||||
"distributed::TokenSharingRpcServer" -> "communication::rpc::Server";
|
||||
"distributed::TokenSharingRpcServer" -> "distributed::Coordination";
|
||||
"distributed::TokenSharingRpcServer" -> "distributed::TokenSharingRpcClients";
|
||||
"distributed::TokenSharingRpcServer" -> "distributed::dgp::Partitioner";
|
||||
|
||||
"distributed::dgp::Partitioner" -> "distributed::DistributedGraphDb" [style=dashed];
|
||||
|
||||
"distributed::dgp::Partitioner" -> "distributed::dgp::VertexMigrator";
|
||||
"distributed::dgp::VertexMigrator" -> "database::GraphDbAccessor" [style=dashed];
|
||||
}
|
||||
@@ -1,43 +0,0 @@
|
||||
# Distributed addressing
|
||||
|
||||
In distributed Memgraph a single graph element must be owned by exactly
|
||||
one worker. It is possible that multiple workers have cached copies of
|
||||
a single graph element (which is inevitable), but there is only one
|
||||
owner.
|
||||
|
||||
The owner of a graph element can change. This is not yet implemented,
|
||||
but is intended. Graph partitioning is intended to be dynamic.
|
||||
|
||||
Graph elements refer to other graph elements that are possibly on some
|
||||
other worker. Even though each graph element is identified with a unique
|
||||
ID, that ID does not contain the information about where that element
|
||||
currently resides (which worker is the owner).
|
||||
|
||||
Thus we introduce the concept of a global address. It indicates both
|
||||
which graph element is referred to (it's global ID), and where it
|
||||
resides. Semantically it's a pair of two elements, but for efficiency
|
||||
it's stored in 64 bits.
|
||||
|
||||
The global address is efficient for usage in a cluster: it indicates
|
||||
where something can be found. However, finding a graph element based on
|
||||
it's ID is still not a free operation (in the current implementation
|
||||
it's a skiplist lookup). So, whenever possible, it's better to use local
|
||||
addresses (pointers).
|
||||
|
||||
Succinctly, the requirements for addressing are:
|
||||
- global addressing containing location info
|
||||
- fast local addressing
|
||||
- storage of both types in the same location efficiently
|
||||
- translation between the two
|
||||
|
||||
The `storage::Address` class handles the enumerated storage
|
||||
requirements. It stores either a local or global address in the size of
|
||||
a local pointer (typically 8 bytes).
|
||||
|
||||
Conversion between the two is done in multiple places. The general
|
||||
approach is to use local addresses (when possible) only for local
|
||||
in-memory handling. All the communication and persistence uses global
|
||||
addresses. Also, when receiving address from another worker, attempt to
|
||||
localize addresses as soon as possible, so that least code has to worry
|
||||
about potential inefficiency of using a global address for a local graph
|
||||
element.
|
||||
@@ -1,50 +0,0 @@
|
||||
# Distributed durability
|
||||
|
||||
Durability in distributed is slightly different then in single-node as
|
||||
the state itself is shared between multiple workers and none of those
|
||||
states are independent.
|
||||
|
||||
Note that recovering from persistent storage must result in a stable
|
||||
database state. This means that across the cluster the state
|
||||
modification of every transaction that was running is either recovered
|
||||
fully or not at all. Also, if transaction A committed before transaction B,
|
||||
then if B is recovered so must A.
|
||||
|
||||
## Snapshots
|
||||
|
||||
It is possibly avoidable but highly desirable that the database can be
|
||||
recovered from snapshot only, without relying on WAL files. For this to
|
||||
be possible in distributed, it must be ensured that the same
|
||||
transactions are recovered on all the workers (including master) in the
|
||||
cluster. Since the snapshot does not contain information about which
|
||||
state change happened in which transaction, the only way to achieve this
|
||||
is to have synchronized snapshots. This means that the process of
|
||||
creating a snapshot, which is in itself transactional (it happens within
|
||||
a transaction and thus observes some consistent database state), must
|
||||
happen in the same transaction. This is achieved by the master starting
|
||||
a snapshot generating transaction and triggering the process on all
|
||||
workers in the cluster.
|
||||
|
||||
## WAL
|
||||
|
||||
Unlike the snapshot, write-ahead logs contain the information on which
|
||||
transaction made which state change. This makes it possible to include
|
||||
or exclude transactions during the recovery process. What is necessary
|
||||
however is a global consensus on which of the transactions should be
|
||||
recovered and which not, to ensure recovery into a consistent state.
|
||||
|
||||
It would be possible to achieve this with some kind of synchronized
|
||||
recovery process, but it would impose constraints on cluster startup and
|
||||
would not be trivial.
|
||||
|
||||
A simpler alternative is that the consensus is achieved beforehand,
|
||||
while the database (to be recovered) is still operational. What is
|
||||
necessary is to keep track of which transactions are guaranteed to
|
||||
have been flushed to the WAL files on all the workers in the cluster. It
|
||||
makes sense to keep this record on the master, so a mechanism is
|
||||
introduced which periodically pings all the workers, telling them to
|
||||
flush their WALs, and writes some sort of a log indicating that this has
|
||||
been confirmed. The downside of this is a periodic broadcast must be
|
||||
done, and that potentially slightly less data can be recovered in the
|
||||
case of a crash then if using a post-crash consensus. It is however much
|
||||
simpler to implement.
|
||||
@@ -1,51 +0,0 @@
|
||||
## Dynamic Graph Partitioning
|
||||
|
||||
Memgraph supports dynamic graph partitioning similar to the Spinner algorithm,
|
||||
mentioned in this paper: [https://arxiv.org/pdf/1404.3861.pdf].
|
||||
|
||||
Dgp is useful because it tries to group `local` date on the same worker, i.e.
|
||||
it tries to keep closely connected data on one worker. It tries to avoid jumps
|
||||
across workers when querying/traversing the distributed graph.
|
||||
|
||||
### Our implementation
|
||||
|
||||
It works independently on each worker but it is always running the migration
|
||||
on only one worker at the same time. It achieves that by sharing a token
|
||||
between workers, and the token ownership is transferred to the next worker
|
||||
when the current worker finishes its migration step.
|
||||
|
||||
The reason that we want workers to work in disjoint time slots is it avoid
|
||||
serialization errors caused by creating/removing edges of vertices during
|
||||
migrations, which might cause an update of some vertex from two or more
|
||||
different transactions.
|
||||
|
||||
### Migrations
|
||||
|
||||
For each vertex and workerid (label in the context of Dgp algorithm) we define
|
||||
a score function. Score function takes into account labels of surrounding
|
||||
endpoints of vertex edges (in/out) and the capacity of the worker with said
|
||||
label. Score function loosely looks like this
|
||||
```
|
||||
locality(v, l) =
|
||||
count endpoints of edges of vertex `v` with label `l` / degree of `v`
|
||||
|
||||
capacity(l) =
|
||||
number of vertices on worker `l` divided by the worker capacity
|
||||
(usually equal to the average number of vertices per worker)
|
||||
|
||||
score(v, l) = locality(v, l) - capacity(l)
|
||||
```
|
||||
We also define two flags alongside ```dynamic_graph_partitioner_enabled```,
|
||||
```dgp_improvement_threshold``` and ```dgp_max_batch_size```.
|
||||
|
||||
These two flags are used during the migration phase.
|
||||
When deciding if we need to migrate some vertex `v` from worker `l1` to worker
|
||||
`l2` we examine the difference in scores, i.e.
|
||||
if score(v, l1) - dgp_improvement_threshold / 100 < score(v, l2) then we
|
||||
migrate the vertex.
|
||||
|
||||
Max batch size flag limits the number of vertices we can transfer in one batch
|
||||
(one migration step).
|
||||
Setting this value to a too large value will probably cause
|
||||
a lot of interference with client queries, and having it a small value
|
||||
will slow down convergence of the algorithm.
|
||||
@@ -1,54 +0,0 @@
|
||||
# Memgraph distributed
|
||||
|
||||
This chapter describes some of the concepts used in distributed
|
||||
Memgraph. By "distributed" here we mean the sharding of a single graph
|
||||
onto multiple processing units (servers).
|
||||
|
||||
## Conceptual organization
|
||||
|
||||
There is a single master and multiple workers. The master contains all
|
||||
the global sources of truth (transaction engine,
|
||||
[label|edge-type|property] to name mappings). Also, in the current
|
||||
organization it is the only one that contains a Bolt server (for
|
||||
communication with the end client) and an interpretation engine. Workers
|
||||
contain the data and means of subquery interpretation (query plans
|
||||
recieved from the master) and means of communication with the master and
|
||||
other workers.
|
||||
|
||||
In many query plans the load on the master is much larger then the load
|
||||
on the workers. For that reason it might be beneficial to make the
|
||||
master contain less data (or none at all), and/or having multiple
|
||||
interpretation masters.
|
||||
|
||||
## Logic organization
|
||||
|
||||
Both the distributed and the single node Memgraph use the same codebase.
|
||||
In cases where the behavior in single-node differs from that in
|
||||
distributed, some kind of dynamic behavior change is implemented (either
|
||||
through inheritance or conditional logic).
|
||||
|
||||
### GraphDb
|
||||
|
||||
The `database::GraphDb` is an "umbrella" object for parts of the
|
||||
database such as storage, garbage collection, transaction engine etc.
|
||||
There is a class heirarchy of `GraphDb` implementations, as well as a
|
||||
base interface object. There are subclasses for single-node, master and
|
||||
worker deplotyments. Which implementation is used depends on the
|
||||
configuration processed in the `main` entry point of memgraph.
|
||||
|
||||
The `GraphDb` interface exposes getters to base classes of
|
||||
other similar heirarchies (for example to `tx::Engine`). In that way
|
||||
much of the code that uses those objects (for example query plan
|
||||
interpretation) is agnostic to the type of deployment.
|
||||
|
||||
### RecordAccessors
|
||||
|
||||
The functionality of `RecordAccessors` and it's subclasses is already
|
||||
documented. It's important to note that the same implementation of
|
||||
accessors is used in all deployments, with internal changes of behavior
|
||||
depending on the locality of the graph element (vertex or edge) the
|
||||
accessor represents. For example, if the graph element is local, an
|
||||
update operation on an accessor will make the necessary MVCC ops, update
|
||||
local data, indexes, the write-ahead log etc. However, if the accessor
|
||||
represents a remote graph element, an update will trigger an RPC message
|
||||
to the owner about the update and a change in the local cache.
|
||||
@@ -1,103 +0,0 @@
|
||||
# Distributed updates
|
||||
|
||||
Operations that modify the graph state are somewhat more complex in the
|
||||
distributed system, as opposed to a single-node Memgraph deployment. The
|
||||
complexity arises from two factors.
|
||||
|
||||
First, the data being modified is not necessarily owned by the worker
|
||||
performing the modification. This situation is completely valid workers
|
||||
execute parts of the query plan and parts must be executed by the
|
||||
master.
|
||||
|
||||
Second, there are less guarantees regarding multi-threaded access. In
|
||||
single-node Memgraph it was guaranteed that only one transaction will be
|
||||
performing database work in a single transaction. This implied that
|
||||
per-version storage could be thread-unsafe. In distributed Memgraph it
|
||||
is possible that multiple threads could be performing work in the same
|
||||
transaction as a consequence of the query being executed at the same
|
||||
time on multiple workers and those executions interacting with the
|
||||
globally partitioned database state.
|
||||
|
||||
## Deferred state modification
|
||||
|
||||
Making the per-version data storage thread-safe would most likely have a
|
||||
performance impact very undesirable in a transactional database intended
|
||||
for high throughput.
|
||||
|
||||
An alternative is that state modification over unsafe structures is not
|
||||
performed immediately when requested, but postponed until it is safe to
|
||||
do (there is is a guarantee of no concurrent access).
|
||||
|
||||
Since local query plan execution is done the same way on local data as
|
||||
it is in single-node Memgraph, it is not possible to deffer that part of
|
||||
the modification story. What can be deferred are modifications requested
|
||||
by other workers. Since local query plan execution still is
|
||||
single-threaded, this approach is safe.
|
||||
|
||||
At the same time those workers requesting the remote update can update
|
||||
local copies (caches) of the not-owned data since that cache is only
|
||||
being used by the single, local-execution thread.
|
||||
|
||||
### Visibility
|
||||
|
||||
Since updates are deferred the question arises: when do the updates
|
||||
become visible? The above described process offers the following
|
||||
visibility guarantees:
|
||||
- updates done on the local state are visible to the owner
|
||||
- updates done on the local state are NOT visible to anyone else during
|
||||
the same (transaction + command)
|
||||
- updates done on remote state are deferred on the owner and not
|
||||
visible to the owner until applied
|
||||
- updates done on the remote state are applied immediately to the local
|
||||
caches and thus visible locally
|
||||
|
||||
This implies an inconsistent view of the database state. In a concurrent
|
||||
execution of a single query this can hardly be avoided and is accepted
|
||||
as such. It does not change the Cypher query execution semantic in any
|
||||
of the well-defined scenarios. It possibly changes some of the behaviors
|
||||
in which the semantic is not well defined even in single-node execution.
|
||||
|
||||
### Synchronization, update application
|
||||
|
||||
In many queries it is mandatory to observe the latest global graph state
|
||||
(typically when returning it to the client). That means that before that
|
||||
happens all the deferred updates need to be applied, and all the caches
|
||||
to remote data invalidated. Exactly this happens when executing queries
|
||||
that modify the graph state. At some point a global synchronization
|
||||
point is reached. First it is waited that all workers finish the
|
||||
execution of query plan parts performing state modifications. After that
|
||||
all the workers are told to apply the deferred updates they received to
|
||||
their graph state. Since there is no concurrent query plan execution,
|
||||
this is safe. Once that is done all the local caches are cleared and the
|
||||
requested data can be returned to the client.
|
||||
|
||||
### Command advancement
|
||||
|
||||
In complex queries where a read part follows a state modification part
|
||||
the synchronization process after the state modification part is
|
||||
followed by command advancement, like in single-node execution.
|
||||
|
||||
## Creation
|
||||
|
||||
Graph element creation is not deferred. This is practical because the
|
||||
response to a creation is the global ID of the newly created element. At
|
||||
the same time it is safe because no other worker (including the owner)
|
||||
will be using the newly added graph element.
|
||||
|
||||
## Updating
|
||||
|
||||
Updating is deferred, as described. Note that this also means that
|
||||
record locking conflicts are deferred and serialization errors
|
||||
(including lock timeouts) are postponed until the deferred update
|
||||
application phase. In certain scenarios it might be beneficial to force
|
||||
these errors to happen earlier, when the deferred update request is
|
||||
processed.
|
||||
|
||||
## Deletion
|
||||
|
||||
Deletion is also deferred. Deleting an edge implies a modification of
|
||||
it's endpoint vertices, which must be deferred as those data structures
|
||||
are not thread-safe. Deleting a vertex is either with detaching, in
|
||||
which case an arbitrary number of updates are implied in the vertex's
|
||||
neighborhood, or without detaching which relies on checking the current
|
||||
state of the graph which is generally impossible in distributed.
|
||||
@@ -1,22 +0,0 @@
|
||||
# Snapshots
|
||||
|
||||
A "snapshot" is a record of the current database state stored in permanent
|
||||
storage. Note that the term "snapshot" is used also in the context of
|
||||
the transaction engine to denote a set of running transactions.
|
||||
|
||||
A snapshot is written to the file by Memgraph periodically if so
|
||||
configured. The snapshot creation process is done within a transaction created
|
||||
specifically for that purpose. The transaction is needed to ensure that
|
||||
the stored state is internally consistent.
|
||||
|
||||
The database state can be recovered from the snapshot during startup, if
|
||||
so configured. This recovery works in conjunction with write-ahead log
|
||||
recovery.
|
||||
|
||||
A single snapshot contains all the data needed to recover a database. In
|
||||
that sense snapshots are independent of each other and old snapshots can
|
||||
be deleted once the new ones are safely stored, if it is not necessary
|
||||
to revert the database to some older state.
|
||||
|
||||
The exact format of the snapshot file is defined inline in the snapshot
|
||||
creation code.
|
||||
@@ -1,55 +0,0 @@
|
||||
# Write-ahead logging
|
||||
|
||||
Typically WAL denotes the process of writing a "log" of database
|
||||
operations (state changes) to persistent storage before committing the
|
||||
transaction, thus ensuring that the state can be recovered (in the case
|
||||
of a crash) for all the transactions which the database committed.
|
||||
|
||||
The WAL is a fine-grained durability format. It's purpose is to store
|
||||
database changes fast. It's primary purpose is not to provide
|
||||
space-efficient storage, nor to support fast recovery. For that reason
|
||||
it's often used in combination with a different persistence mechanism
|
||||
(in Memgraph's case the "snapshot") that has complementary
|
||||
characteristics.
|
||||
|
||||
### Guarantees
|
||||
|
||||
Ensuring that the log is written before the transaction is committed can
|
||||
slow down the database. For that reason this guarantee is most often
|
||||
configurable in databases.
|
||||
|
||||
Memgraph offers two options for the WAL. The default option, where the WAL is
|
||||
flushed to the disk periodically and transactions do not wait for this to
|
||||
complete, introduces the risk of database inconsistency because an operating
|
||||
system or hardware crash might lead to missing transactions in the WAL. Memgraph
|
||||
will handle this as if those transactions never happened. The second option,
|
||||
called synchronous commit, will instruct Memgraph to wait for the WAL to be
|
||||
flushed to the disk when a transactions completes and the transaction will wait
|
||||
for this to complete. This option can be turned on with the
|
||||
`--synchronous-commit` command line flag.
|
||||
|
||||
### Format
|
||||
|
||||
The WAL file contains a series of DB state changes called `StateDelta`s.
|
||||
Each of them describes what the state change is and in which transaction
|
||||
it happened. Also some kinds of meta-information needed to ensure proper
|
||||
state recovery are recorded (transaction beginnings and commits/abort).
|
||||
|
||||
The following is guaranteed w.r.t. `StateDelta` ordering in
|
||||
a single WAL file:
|
||||
- For two ops in the same transaction, if op A happened before B in the
|
||||
database, that ordering is preserved in the log.
|
||||
- Transaction begin/commit/abort messages also appear in exactly the
|
||||
same order as they were executed in the transactional engine.
|
||||
|
||||
### Recovery
|
||||
|
||||
The database can recover from the WAL on startup. This works in
|
||||
conjunction with snapshot recovery. The database attempts to recover from
|
||||
the latest snapshot and then apply as much as possible from the WAL
|
||||
files. Only those transactions that were not recovered from the snapshot
|
||||
are recovered from the WAL, for speed efficiency. It is possible (but
|
||||
inefficient) to recover the database from WAL only, provided all the WAL
|
||||
files created from DB start are available. It is not possible to recover
|
||||
partial database state (i.e. from some suffix of WAL files, without the
|
||||
preceding snapshot).
|
||||
1349
docs/dev/lcp.md
1349
docs/dev/lcp.md
File diff suppressed because it is too large
Load Diff
@@ -1,15 +0,0 @@
|
||||
# Other Code Conventions
|
||||
|
||||
While we are mainly programming in C++, we do use other programming languages
|
||||
when appropriate. This chapter describes conventions for such code.
|
||||
|
||||
## Python
|
||||
|
||||
Code written in Python should adhere to
|
||||
[PEP 8](https://www.python.org/dev/peps/pep-0008/). You should run `flake8` on
|
||||
your code to automatically check compliance.
|
||||
|
||||
## Common Lisp
|
||||
|
||||
Code written in Common Lisp should adhere to
|
||||
[Google Common Lisp Style](https://google.github.io/styleguide/lispguide.xml).
|
||||
1
docs/dev/query/.gitignore
vendored
1
docs/dev/query/.gitignore
vendored
@@ -1 +0,0 @@
|
||||
html/
|
||||
@@ -1,16 +0,0 @@
|
||||
#!/bin/bash
|
||||
|
||||
script_dir="$( cd "$( dirname "${BASH_SOURCE[0]}" )" && pwd )"
|
||||
|
||||
mkdir -p $script_dir/html
|
||||
|
||||
for markdown_file in $(find $script_dir -name '*.md'); do
|
||||
name=$(basename -s .md $markdown_file)
|
||||
sed -e 's/.md/.html/' $markdown_file | \
|
||||
pandoc -s -f markdown -t html -o $script_dir/html/$name.html
|
||||
done
|
||||
|
||||
for dot_file in $(find $script_dir -name '*.dot'); do
|
||||
name=$(basename -s .dot $dot_file)
|
||||
dot -Tpng $dot_file -o $script_dir/html/$name.png
|
||||
done
|
||||
@@ -1,34 +0,0 @@
|
||||
# Query Parsing, Planning and Execution
|
||||
|
||||
This part of the documentation deals with query execution.
|
||||
|
||||
Memgraph currently supports only query interpretation. Each new query is
|
||||
parsed, analysed and translated into a sequence of operations which are then
|
||||
executed on the main database storage. Query execution is organized into the
|
||||
following phases:
|
||||
|
||||
1. [Lexical Analysis (Tokenization)](parsing.md)
|
||||
2. [Syntactic Analysis (Parsing)](parsing.md)
|
||||
3. [Semantic Analysis and Symbol Generation](semantic.md)
|
||||
4. [Logical Planning](planning.md)
|
||||
5. [Logical Plan Execution](execution.md)
|
||||
|
||||
The main entry point is `Interpreter::operator()`, which takes a query text
|
||||
string and produces a `Results` object. To instantiate the object,
|
||||
`Interpreter` needs to perform the above steps from 1 to 4. If any of the
|
||||
steps fail, a `QueryException` is thrown. The complete `LogicalPlan` is
|
||||
wrapped into a `CachedPlan` and stored for reuse. This way we can skip the
|
||||
whole process of analysing a query if it appears to be the same as before.
|
||||
|
||||
When we have valid plan, the client code can invoke `Results::PullAll` with a
|
||||
stream object. The `Results` instance will then execute the plan and fill the
|
||||
stream with the obtained results.
|
||||
|
||||
Since we want to optionally run Memgraph as a distributed database, we have
|
||||
hooks for creating a different plan of logical operators.
|
||||
`DistributedInterpreter` inherits from `Interpreter` and overrides
|
||||
`MakeLogicalPlan` method. This method needs to return a concrete instance of
|
||||
`LogicalPlan`, and in case of distributed database that will be
|
||||
`DistributedLogicalPlan`.
|
||||
|
||||

|
||||
@@ -1,373 +0,0 @@
|
||||
# Logical Plan Execution
|
||||
|
||||
We implement classical iterator style operators. Logical operators define
|
||||
operations on database. They encapsulate the following info: what the input is
|
||||
(another `LogicalOperator`), what to do with the data, and how to do it.
|
||||
|
||||
Currently logical operators can have zero or more input operations, and thus a
|
||||
`LogicalOperator` tree is formed. Most `LogicalOperator` types have only one
|
||||
input, so we are mostly working with chains instead of full fledged trees.
|
||||
You can find information on each operator in `src/query/plan/operator.lcp`.
|
||||
|
||||
## Cursor
|
||||
|
||||
Logical operators do not perform database work themselves. Instead they create
|
||||
`Cursor` objects that do the actual work, based on the info in the operator.
|
||||
Cursors expose a `Pull` method that gets called by the cursor's consumer. The
|
||||
consumer keeps pulling as long as the `Pull` returns `true` (indicating it
|
||||
successfully performed some work and might be eligible for another `Pull`).
|
||||
Most cursors will call the `Pull` function of their input provided cursor, so
|
||||
typically a cursor chain is created that is analogue to the logical operator
|
||||
chain it's created from.
|
||||
|
||||
## Frame
|
||||
|
||||
The `Frame` object contains all the data of the current `Pull` chain. It
|
||||
serves for communicating data between cursors.
|
||||
|
||||
For example, in a `MATCH (n) RETURN n` query the `ScanAllCursor` places a
|
||||
vertex on the `Frame` for each `Pull`. It places it on the place reserved for
|
||||
the `n` symbol. Then the `ProduceCursor` can take that same value from the
|
||||
`Frame` because it knows the appropriate symbol. `Frame` positions are indexed
|
||||
by `Symbol` objects.
|
||||
|
||||
## ExpressionEvaluator
|
||||
|
||||
Expressions results are not placed on the `Frame` since they do not need to be
|
||||
communicated between different `Cursors`. Instead, expressions are evaluated
|
||||
using an instance of `ExpressionEvaluator`. Since generally speaking an
|
||||
expression can be defined by a tree of subexpressions, the
|
||||
`ExpressionEvaluator` is implemented as a tree visitor. There is a performance
|
||||
sub-optimality here because a stack is used to communicate intermediary
|
||||
expression results between elements of the tree. This is one of the reasons
|
||||
why it's planned to use `Frame` for intermediary expression results as well.
|
||||
The other reason is that it might facilitate compilation later on.
|
||||
|
||||
## Cypher Execution Semantics
|
||||
|
||||
Cypher query execution has *mostly* well-defined semantics. Some are
|
||||
explicitly defined by openCypher and its TCK, while others are implicitly
|
||||
defined by Neo4j's implementation of Cypher that we want to be generally
|
||||
compatible with.
|
||||
|
||||
These semantics can in short be described as follows: a Cypher query consists
|
||||
of multiple clauses some of which modify it. Generally, every clause in the
|
||||
query, when reading it left to right, operates on a consistent state of the
|
||||
property graph, untouched by subsequent clauses. This means that a `MATCH`
|
||||
clause in the beginning operates on a graph-state in which modifications by
|
||||
the subsequent `SET` are not visible.
|
||||
|
||||
The stated semantics feel very natural to the end-user, and Neo seems to
|
||||
implement them well. For Memgraph the situation is complex because
|
||||
`LogicalOperator` execution (through a `Cursor`) happens one `Pull` at a time
|
||||
(generally meaning all the query clauses get executed for every top-level
|
||||
`Pull`). This is not inherently consistent with Cypher semantics because a
|
||||
`SET` clause can modify data, and the `MATCH` clause that precedes it might
|
||||
see the modification in a subsequent `Pull`. Also, the `RETURN` clause might
|
||||
want to stream results to the user before all `SET` clauses have been
|
||||
executed, so the user might see some intermediate graph state. There are many
|
||||
edge-cases that Memgraph does its best to avoid to stay true to Cypher
|
||||
semantics, while at the same time using a high-performance streaming approach.
|
||||
The edge-cases are enumerated in this document along with the implementation
|
||||
details they imply.
|
||||
|
||||
## Implementation Peculiarities
|
||||
|
||||
### Once
|
||||
|
||||
An operator that does nothing but whose `Cursor::Pull` returns `true` on the
|
||||
first `Pull` and `false` on subsequent ones. This operator is used when
|
||||
another operator has an optional input, because in Cypher a clause will
|
||||
typically execute once for every input from the preceding clauses, or just
|
||||
once if there was no preceding input. For example, consider the `CREATE`
|
||||
clause. In the query `CREATE (n)` only one node is created, while in the query
|
||||
`MATCH (n) CREATE (m)` a node is created for each existing node. Thus in our
|
||||
`CreateNode` logical operator the input is either a `ScanAll` operator, or a
|
||||
`Once` operator.
|
||||
|
||||
### storage::View
|
||||
|
||||
In the previous section, [Cypher Execution
|
||||
Semantics](#cypher-execution-semantics), we mentioned how the preceding
|
||||
clauses should not see changes made in subsequent ones. For that reason, some
|
||||
operators take a `storage::View` enum value. This value determines which state of
|
||||
the graph an operator sees.
|
||||
|
||||
Consider the query `MATCH (n)--(m) WHERE n.x = 0 SET m.x = 1`. Naive streaming
|
||||
could match a vertex `n` on the given criteria, expand to `m`, update it's
|
||||
property, and in the next iteration consider the vertex previously matched to
|
||||
`m` and skip it because it's newly set property value does not qualify. This
|
||||
is not how Cypher works. To handle this issue properly, Memgraph designed the
|
||||
`VertexAccessor` class that tracks two versions of data: one that was visible
|
||||
before the current transaction+command, and the optional other that was
|
||||
created in the current transaction+command. The `MATCH` clause will be planned
|
||||
as `ScanAll` and `Expand` operations using `storage::View::OLD` value. This
|
||||
will ensure modifications performed in the same query do not affect it. The
|
||||
same applies to edges and the `EdgeAccessor` class.
|
||||
|
||||
### Existing Record Detection
|
||||
|
||||
It's possible that a pattern element has already been declared in the same
|
||||
pattern, or a preceding pattern. For example `MATCH (n)--(m), (n)--(l)` or a
|
||||
cycle-detection match `MATCH (n)-->(n) RETURN n`. Implementation-wise,
|
||||
existing record detection just checks that the expanded record is equal to the
|
||||
one already on the frame.
|
||||
|
||||
### Why Not Use Separate Expansion Ops for Edges and Vertices?
|
||||
|
||||
Expanding an edge and a vertex in separate ops is not feasible when matching a
|
||||
cycle in bi-directional expansions. Consider the query `MATCH (n)--(n) RETURN
|
||||
n`. Let's try to expand first the edge in one op, and vertex in the next. The
|
||||
vertex expansion consumes the edge expansion input. It takes the expanded edge
|
||||
from the frame. It needs to detect a cycle by comparing the vertex existing on
|
||||
the frame with one of the edge vertices (`from` or `to`). But which one? It
|
||||
doesn't know, and can't ensure correct cycle detection.
|
||||
|
||||
### Data Visibility During and After SET
|
||||
|
||||
In Cypher, setting values always works on the latest version of data (from
|
||||
preceding or current clause). That means that within a `SET` clause all the
|
||||
changes from previous clauses must be visible, as well as changes done by the
|
||||
current `SET` clause. Also, if there is a clause after `SET` it must see *all*
|
||||
the changes performed by the preceding `SET`. Both these things are best
|
||||
illustrated with the following queries executed on an empty database:
|
||||
|
||||
CREATE (n:A {x:0})-[:EdgeType]->(m:B {x:0})
|
||||
MATCH (n)--(m) SET m.x = n.x + 1 RETURN labels(n), n.x, labels(m), m.x
|
||||
|
||||
This returns:
|
||||
|
||||
+---------+---+---------+---+
|
||||
|labels(n)|n.x|labels(m)|m.x|
|
||||
+:=======:+:=:+:=======:+:=:+
|
||||
|[A] |2 |[B] |1 |
|
||||
+---------+---+---------+---+
|
||||
|[B] |1 |[A] |2 |
|
||||
+---------+---+---------+---+
|
||||
|
||||
The obtained result implies the following operations:
|
||||
|
||||
1. In the first iteration set the value of the `B.x` to 1
|
||||
2. In the second iteration the we observe `B.x` with the value of 1 and set
|
||||
`A.x` to 2
|
||||
3. In `RETURN` we see all the changes made in both iterations
|
||||
|
||||
To implement the desired behavior Memgraph utilizes two techniques. First is
|
||||
the already mentioned tracking of two versions of data in vertex accessors.
|
||||
Using this approach ensures that the second iteration in the example query
|
||||
sees the data modification performed by the preceding iteration. The second
|
||||
technique is the `Accumulate` operation that accumulates all the iterations
|
||||
from the preceding logical op before passing them to the next logical op. In
|
||||
the example query, `Accumulate` ensures that the results returned to the user
|
||||
reflect changes performed in all iterations of the query (naive streaming
|
||||
could stream results at the end of first iteration producing inconsistent
|
||||
results). Note that `Accumulate` is demanding regarding memory and slows down
|
||||
query execution. For that reason it should be used only when necessary, for
|
||||
example it does not have to be used in a query that has `MATCH` and `SET` but
|
||||
no `RETURN`.
|
||||
|
||||
### Neo4j Inconsistency on Multiple SET Clauses
|
||||
|
||||
Considering the preceding example it could be expected that when a query has
|
||||
multiple `SET` clauses all the changes from those preceding one are visible.
|
||||
This is not the case in Neo4j's implementation. Consider the following queries
|
||||
executed on an empty database:
|
||||
|
||||
CREATE (n:A {x:0})-[:EdgeType]->(m:B {x:0})
|
||||
MATCH (n)--(m) SET n.x = n.x + 1 SET m.x = m.x * 2
|
||||
RETURN labels(n), n.x, labels(m), m.x
|
||||
|
||||
This returns:
|
||||
|
||||
+---------+---+---------+---+
|
||||
|labels(n)|n.x|labels(m)|m.x|
|
||||
+:=======:+:=:+:=======:+:=:+
|
||||
|[A] |2 |[B] |1 |
|
||||
+---------+---+---------+---+
|
||||
|[B] |1 |[A] |2 |
|
||||
+---------+---+---------+---+
|
||||
|
||||
If all the iterations of the first `SET` clause were executed before executing
|
||||
the second, all the resulting values would be 2. This not being the case, we
|
||||
conclude that Neo4j does not use a barrier-like mechanism between `SET`
|
||||
clauses. It is Memgraph's current vision that this is inconsistent and we
|
||||
plan to reduce Neo4j compliance in favour of operation consistency.
|
||||
|
||||
### Double Deletion
|
||||
|
||||
It's possible to match the same graph element multiple times in a single query
|
||||
and delete it. Neo supports this, and so do we. The relevant implementation
|
||||
detail is in the `GraphDbAccessor` class, where the record deletion functions
|
||||
reside, and not in the logical plan execution. It comes down to checking if a
|
||||
record has already been deleted in the current transaction+command and not
|
||||
attempting to do it again (results in a crash).
|
||||
|
||||
### Set + Delete Edge-case
|
||||
|
||||
It's legal for a query to combine `SET` and `DELETE` clauses. Consider the
|
||||
following queries executed on an empty database:
|
||||
|
||||
|
||||
CREATE ()-[:T]->()
|
||||
MATCH (n)--(m) SET n.x = 42 DETACH DELETE m
|
||||
|
||||
Due to the `MATCH` being undirected the second pull will attempt to set data
|
||||
on a deleted vertex. This is not a legal operation in Memgraph storage
|
||||
implementation. For that reason the logical operator for `SET` must check if
|
||||
the record it's trying to set something on has been deleted by the current
|
||||
transaction+command. If so, the modification is not executed.
|
||||
|
||||
### Deletion Accumulation
|
||||
|
||||
Sometimes it's necessary to accumulate deletions of all the matches before
|
||||
attempting to execute them. Consider this the following. Start with an empty
|
||||
database and execute queries:
|
||||
|
||||
CREATE ()-[:T]->()-[:T]->()
|
||||
MATCH (a)-[r1]-(b)-[r2]-(c) DELETE r1, b, c
|
||||
|
||||
Note that the `DELETE` clause attempts to delete node `c`, but it does not
|
||||
detach it by deleting edge `r2`. However, due to undirected edge in the
|
||||
`MATCH`, both edges get pulled and deleted.
|
||||
|
||||
Currently Memgraph does not support this behavior, Neo does. There are a few
|
||||
ways that we could do this.
|
||||
|
||||
* Accumulate on deletion (that sucks because we have to keep track of
|
||||
everything that gets returned after the deletion).
|
||||
* Maybe we could stream through the deletion op, but defer actual deletion
|
||||
until plan-execution end.
|
||||
* Ignore this because it's very edgy (this is the currently selected option).
|
||||
|
||||
### Aggregation Without Input
|
||||
|
||||
It is necessary to define what aggregation ops return when they receive no
|
||||
input. Following is a table that shows what Neo4j's Cypher implementation and
|
||||
SQL produce.
|
||||
|
||||
|
||||
+-------------+------------------------+---------------------+---------------------+------------------+
|
||||
| \<OP\> | 1. Cypher, no group-by | 2. Cypher, group-by | 3. SQL, no group-by | 4. SQL, group-by |
|
||||
+=============+:======================:+:===================:+:===================:+:================:+
|
||||
| Count(\*) | 0 | \<NO\_ROWS> | 0 | \<NO\_ROWS> |
|
||||
+-------------+------------------------+---------------------+---------------------+------------------+
|
||||
| Count(prop) | 0 | \<NO\_ROWS> | 0 | \<NO\_ROWS> |
|
||||
+-------------+------------------------+---------------------+---------------------+------------------+
|
||||
| Sum | 0 | \<NO\_ROWS> | NULL | \<NO\_ROWS> |
|
||||
+-------------+------------------------+---------------------+---------------------+------------------+
|
||||
| Avg | NULL | \<NO\_ROWS> | NULL | \<NO\_ROWS> |
|
||||
+-------------+------------------------+---------------------+---------------------+------------------+
|
||||
| Min | NULL | \<NO\_ROWS> | NULL | \<NO\_ROWS> |
|
||||
+-------------+------------------------+---------------------+---------------------+------------------+
|
||||
| Max | NULL | \<NO\_ROWS> | NULL | \<NO\_ROWS> |
|
||||
+-------------+------------------------+---------------------+---------------------+------------------+
|
||||
| Collect | [] | \<NO\_ROWS> | N/A | N/A |
|
||||
+-------------+------------------------+---------------------+---------------------+------------------+
|
||||
|
||||
Where:
|
||||
|
||||
1. `MATCH (n) RETURN <OP>(n.prop)`
|
||||
2. `MATCH (n) RETURN <OP>(n.prop), (n.prop2)`
|
||||
3. `SELECT <OP>(prop) FROM Table`
|
||||
4. `SELECT <OP>(prop), prop2 FROM Table GROUP BY prop2`
|
||||
|
||||
Neo's Cypher implementation diverges from SQL only when performing `SUM`.
|
||||
Memgraph implements SQL-like behavior. It is considered that `SUM` of
|
||||
arbitrary elements should not be implicitly 0, especially in a property graph
|
||||
without a strict schema (the property in question can contain values of
|
||||
arbitrary types, or no values at all).
|
||||
|
||||
### OrderBy
|
||||
|
||||
The `OrderBy` logical operator sorts the results in the desired order. It
|
||||
occurs in Cypher as part of a `WITH` or `RETURN` clause. Both the concept and
|
||||
the implementation are straightforward. It's necessary for the logical op to
|
||||
`Pull` everything from its input so it can be sorted. It's not necessary to
|
||||
keep the whole `Frame` state of each input, it is sufficient to keep a list of
|
||||
`TypedValues` on which the results will be sorted, and another list of values
|
||||
that need to be remembered and recreated on the `Frame` when yielding.
|
||||
|
||||
The sorting itself is made to reflect that of Neo's implementation which comes
|
||||
down to these points.
|
||||
|
||||
* `Null` comes last (as if it's greater than anything).
|
||||
* Primitive types compare naturally, with no implicit casting except from
|
||||
`int` to `double`.
|
||||
* Complex types are not comparable.
|
||||
* Every unsupported comparison results in an exception that gets propagated
|
||||
to the end user.
|
||||
|
||||
### Limit in Write Queries
|
||||
|
||||
`Limit` can be used as part of a write query, in which case it will *not*
|
||||
reduce the amount of performed updates. For example, consider a database that
|
||||
has 10 vertices. The query `MATCH (n) SET n.x = 1 RETURN n LIMIT 3` will
|
||||
result in all vertices having their property value changed, while returning
|
||||
only the first to the client. This makes sense from the implementation
|
||||
standpoint, because `Accumulate` is planned after `SetProperty` but before
|
||||
`Produce` and `Limit` operations. Note that this behavior can be
|
||||
non-deterministic in some queries, since it relies on the order of iteration
|
||||
over nodes which is undefined when not explicitly specified.
|
||||
|
||||
### Merge
|
||||
|
||||
`MERGE` in Cypher attempts to match a pattern. If it already exists, it does
|
||||
nothing and subsequent clauses like `RETURN` can use the matched pattern
|
||||
elements. If the pattern can't match to any data, it creates it. For detailed
|
||||
information see Neo4j's [merge
|
||||
documentation.](https://neo4j.com/docs/developer-manual/current/cypher/clauses/merge/)
|
||||
|
||||
An important thing about `MERGE` is visibility of modified data. `MERGE` takes
|
||||
an input (typically a `MATCH`) and has two additional *phases*: the merging
|
||||
part, and the subsequent set parts (`ON MATCH SET` and `ON CREATE SET`).
|
||||
Analysis of Neo4j's behavior indicates that each of these three phases (input,
|
||||
merge, set) does not see changes to the graph state done by subsequent phase.
|
||||
The input phase does not see data created by the merge phase, nor the set
|
||||
phase. This is consistent with what seems like the general Cypher philosophy
|
||||
that query clause effects aren't visible in the preceding clauses.
|
||||
|
||||
We define the `Merge` logical operator as a *routing* operator that uses three
|
||||
logical operator branches.
|
||||
|
||||
1. The input from a preceding clause.
|
||||
|
||||
For example in `MATCH (n), (m) MERGE (n)-[:T]-(m)`. This input is
|
||||
optional because `MERGE` is allowed to be the first clause in a query.
|
||||
|
||||
2. The `merge_match` branch.
|
||||
|
||||
This logical operator branch is `Pull`-ed from until exhausted for each
|
||||
successful `Pull` from the input branch.
|
||||
|
||||
3. The `merge_create` branch.
|
||||
|
||||
This branch is `Pull`ed when the `merge_match` branch does not match
|
||||
anything (no successful `Pull`s) for an input `Pull`. It is `Pull`ed only
|
||||
once in such a situation, since only one creation needs to occur for a
|
||||
failed match.
|
||||
|
||||
The `ON MATCH SET` and `ON CREATE SET` parts of the `MERGE` clause are
|
||||
included in the `merge_match` and `merge_create` branches respectively. They
|
||||
are placed on the end of their branches so that they execute only when those
|
||||
branches succeed.
|
||||
|
||||
Memgraph strives to be consistent with Neo in its `MERGE` implementation,
|
||||
while at the same time keeping performance as good as possible. Consistency
|
||||
with Neo w.r.t. graph state visibility is not trivial. Documentation for
|
||||
`Expand` and `Set` describe how Memgraph keeps track of both the updated
|
||||
version of an edge/vertex and the old one, as it was before the current
|
||||
transaction+command. This technique is also used in `Merge`. The input
|
||||
phase/branch of `Merge` always looks at the old data. The merge phase needs to
|
||||
see the new data so it doesn't create more data then necessary.
|
||||
|
||||
For example, consider the query.
|
||||
|
||||
MATCH (p:Person) MERGE (c:City {name: p.lives_in})
|
||||
|
||||
This query needs to create a city node only once for each unique `p.lives_in`.
|
||||
Finally the set phase of a `MERGE` clause should not affect the merge phase.
|
||||
To achieve this the `merge_match` branch of the `Merge` operator should see
|
||||
the latest created nodes, but filter them on their old state (if those nodes
|
||||
were not created by the `create_branch`). Implementation-wise that means that
|
||||
`ScanAll` and `Expand` operators in the `merge_branch` need to look at the new
|
||||
graph state, while `Filter` operators the old, if available.
|
||||
@@ -1,23 +0,0 @@
|
||||
digraph interpreter {
|
||||
node [fontname="dejavusansmono"]
|
||||
edge [fontname="dejavusansmono"]
|
||||
node [shape=record]
|
||||
edge [dir=back,arrowtail=empty,arrowsize=1.5]
|
||||
Interpreter [label="{\N|+ operator(query : string, ...) : Results\l|
|
||||
# MakeLogicalPlan(...) : LogicalPlan\l|
|
||||
- plan_cache_ : Map(QueryHash, CachedPlan)\l}"]
|
||||
Interpreter -> DistributedInterpreter
|
||||
Results [label="{\N|+ PullAll(stream) : void\l|- plan_ : CachedPlan\l}"]
|
||||
Interpreter -> Results
|
||||
[dir=forward,style=dashed,arrowhead=open,label="<<create>>"]
|
||||
CachedPlan -> Results
|
||||
[dir=forward,arrowhead=odiamond,taillabel="1",headlabel="*"]
|
||||
Interpreter -> CachedPlan [arrowtail=diamond,taillabel="1",headlabel="*"]
|
||||
CachedPlan -> LogicalPlan [arrowtail=diamond]
|
||||
LogicalPlan [label="{\N|+ GetRoot() : LogicalOperator
|
||||
\l+ GetCost() : double\l}"]
|
||||
LogicalPlan -> SingleNodeLogicalPlan [style=dashed]
|
||||
LogicalPlan -> DistributedLogicalPlan [style=dashed]
|
||||
DistributedInterpreter -> DistributedLogicalPlan
|
||||
[dir=forward,style=dashed,arrowhead=open,label="<<create>>"]
|
||||
}
|
||||
@@ -1,62 +0,0 @@
|
||||
# Lexical and Syntactic Analysis
|
||||
|
||||
## Antlr
|
||||
|
||||
We use Antlr for lexical and syntax analysis of Cypher queries. Antrl uses
|
||||
grammar file `Cypher.g4` downloaded from http://www.opencypher.org to generate
|
||||
the parser and the visitor for the Cypher parse tree. Even though the provided
|
||||
grammar is not very pleasant to work with we decided not to do any drastic
|
||||
changes to it so that our transition to newly published versions of
|
||||
`Cypher.g4` would be easier. Nevertheless, we had to fix some bugs and add
|
||||
features, so our version is not completely the same.
|
||||
|
||||
In addition to using `Cypher.g4`, we have `MemgraphCypher.g4`. This grammar
|
||||
file defines Memgraph specific extensions to the original grammar. Most
|
||||
notable example is the inclusion of syntax for handling authorization. At the
|
||||
moment, some extensions are also found in `Cypher.g4`. For example, the syntax
|
||||
for using a lambda function in relationship patterns. These extensions should
|
||||
be moved out of `Cypher.g4`, so that it remains as close to the original
|
||||
grammar as possible. Additionally, having `MemgraphCypher.g4` may not be
|
||||
enough if we wish to split the functionality for community and enterprise
|
||||
editions of Memgraph.
|
||||
|
||||
## Abstract Syntax Tree (AST)
|
||||
|
||||
Since Antlr generated visitor and the official openCypher grammar are not very
|
||||
practical to use, we translate the Antlr's AST to our own AST. Currently there
|
||||
are ~40 types of nodes in our AST. Their definitions can be found in
|
||||
`src/query/frontend/ast/ast.lcp`.
|
||||
|
||||
Major groups of types can be found under the following base types.
|
||||
|
||||
* `Expression` --- types corresponding to Cypher expressions.
|
||||
* `Clause` --- types corresponding to Cypher clauses.
|
||||
* `PatternAtom` --- node or edge related information.
|
||||
* `Query` --- different kinds of queries, allows extending the language with
|
||||
Memgraph specific query syntax.
|
||||
|
||||
Memory management of created AST nodes is done with `AstStorage`. Each type
|
||||
must be created by invoking `AstStorage::Create` method. This way all of the
|
||||
pointers to nodes and their children are raw pointers. The only owner of
|
||||
allocated memory is the `AstStorage`. When the storage goes out of scope, the
|
||||
pointers become invalid. It may be more natural to handle tree ownership via
|
||||
`unique_ptr`, i.e. each node owns its children. But there are some benefits to
|
||||
having a custom storage and allocation scheme.
|
||||
|
||||
The primary reason we opted for not using `unique_ptr` is the requirement of
|
||||
Antlr's base visitor class that the resulting values must by copyable. The
|
||||
result is wrapped in `antlr::Any` so that the derived visitor classes may
|
||||
return any type they wish when visiting Antlr's AST. Unfortunately,
|
||||
`antlr::Any` does not work with non-copyable types.
|
||||
|
||||
Another benefit of having `AstStorage` is that we can easily add a different
|
||||
allocation scheme for AST nodes. The interface of node creation would not
|
||||
change.
|
||||
|
||||
### AST Translation
|
||||
|
||||
The translation process is done via `CypherMainVisitor` class, which is
|
||||
derived from Antlr generated visitor. Besides instancing our AST types, a
|
||||
minimal number of syntactic checks are done on a query. These checks handle
|
||||
the cases which were valid in original openCypher grammar, but may be invalid
|
||||
when combined with other syntax elements.
|
||||
@@ -1,487 +0,0 @@
|
||||
# Logical Planning
|
||||
|
||||
After the semantic analysis and symbol generation, the AST is converted to a
|
||||
tree of logical operators. This conversion is called *planning* and the tree
|
||||
of logical operators is called a *plan*. The whole planning process is done in
|
||||
the following steps.
|
||||
|
||||
1. [AST Preprocessing](#ast-preprocessing)
|
||||
|
||||
The first step is to preprocess the AST by collecting
|
||||
information on filters, divide the query into parts, normalize patterns
|
||||
in `MATCH` clauses, etc.
|
||||
|
||||
2. [Logical Operator Planning](#logical-operator-planning)
|
||||
|
||||
After the preprocess step, the planning can be done via 2 planners:
|
||||
`VariableStartPlanner` and `RuleBasedPlanner`. The first planner will
|
||||
generate multiple plans where each plan has different starting points for
|
||||
searching the patterns in `MATCH` clauses. The second planner produces a
|
||||
single plan by mapping the query parts as they are to logical operators.
|
||||
|
||||
3. [Logical Plan Postprocessing](#logical-plan-postprocessing)
|
||||
|
||||
In this stage, we perform various transformations on the generated logical
|
||||
plan. Here we want to optimize the operations in order to improve
|
||||
performance during the execution. Naturally, transformations need to
|
||||
preserve the semantic behaviour of the original plan.
|
||||
|
||||
4. [Cost Estimation](#cost-estimation)
|
||||
|
||||
After the generation, the execution cost of each plan is estimated. This
|
||||
estimation is used to select the best plan which will be executed.
|
||||
|
||||
5. [Distributed Planning](#distributed-planning)
|
||||
|
||||
In case we are running distributed Memgraph, the final plan is adapted
|
||||
for distributed execution. NOTE: This appears to be an error in the
|
||||
workflow. Distributed planning should be moved before step 3. or
|
||||
integrated with it. With the workflow ordered as is now, cost estimation
|
||||
doesn't consider the distributed plan.
|
||||
|
||||
The implementation can be found in the `query/plan` directory, with the public
|
||||
entry point being `query/plan/planner.hpp`.
|
||||
|
||||
## AST Preprocessing
|
||||
|
||||
Each openCypher query consists of at least 1 **single query**. Multiple single
|
||||
queries are chained together using a **query combinator**. Currently, there is
|
||||
only one combinator, `UNION`. The preprocessing step starts in the
|
||||
`CollectQueryParts` function. This function will take a look at each single
|
||||
query and divide it into parts. Each part is separated with `RETURN` and
|
||||
`WITH` clauses. For example:
|
||||
|
||||
MATCH (n) CREATE (m) WITH m MATCH (l)-[]-(m) RETURN l
|
||||
| | |
|
||||
|------- part 1 -----------+-------- part 2 --------|
|
||||
| |
|
||||
|-------------------- single query -----------------|
|
||||
|
||||
Each part is created by collecting all `MATCH` clauses and *normalizing* their
|
||||
patterns. Pattern normalization is the process of converting an arbitrarily
|
||||
long pattern chain of nodes and edges into a list of triplets `(start node,
|
||||
edge, end node)`. The triplets should preserve the semantics of the match. For
|
||||
example:
|
||||
|
||||
MATCH (a)-[p]-(b)-[q]-(c)-[r]-(d)
|
||||
|
||||
is equivalent to:
|
||||
|
||||
MATCH (a)-[p]-(b), (b)-[q]-(c), (c)-[r]-(d)
|
||||
|
||||
With this representation, it becomes easier to reorder the triplets and choose
|
||||
different strategies for pattern matching.
|
||||
|
||||
In addition to normalizing patterns, all of the filter expressions in patterns
|
||||
and inside of the `WHERE` clause (of the accompanying `MATCH`) are extracted
|
||||
and stored separately. During the extraction, symbols used in the filter
|
||||
expression are collected. This allows for planning filters in a valid order,
|
||||
as the matching for triplets is being done. Another important benefit of
|
||||
having extra information on filters, is to recognize when a database index
|
||||
could be used.
|
||||
|
||||
After each `MATCH` is processed, they are all grouped, so that even the whole
|
||||
`MATCH` clauses may be reordered. The important thing is to remember which
|
||||
symbols were used to name edges in each `MATCH`. With those symbols we can
|
||||
plan for *cyphermorphism*, i.e. ensure different edges in the search pattern
|
||||
of a single `MATCH` map to different edges in the graph. This preserves the
|
||||
semantic of the query, even though we may have reordered the matching. The
|
||||
same steps are done for `OPTIONAL MATCH`.
|
||||
|
||||
Another clause which needs processing is `MERGE`. Here we normalize the
|
||||
pattern, since the `MERGE` is a bit like `MATCH` and `CREATE` in one.
|
||||
|
||||
All the other clauses are left as is.
|
||||
|
||||
In the end, each query part consists of:
|
||||
|
||||
* processed and grouped `MATCH` clauses;
|
||||
* processed and grouped `OPTIONAL MATCH` clauses;
|
||||
* processed `MERGE` matching pattern and
|
||||
* unchanged remaining clauses.
|
||||
|
||||
The last stored clause is guaranteed to be either `WITH` or `RETURN`.
|
||||
|
||||
## Logical Operator Planning
|
||||
|
||||
### Variable Start Planner
|
||||
|
||||
The `VariableStartPlanner` generates multiple plans for a single query. Each
|
||||
plan is generated by selecting a different starting point for pattern
|
||||
matching.
|
||||
|
||||
The algorithm works as follows.
|
||||
|
||||
1. For each query part:
|
||||
1. For each node in triplets of collected `MATCH` clauses:
|
||||
i. Add the node to a set of `expanded` nodes
|
||||
ii. Select a triplet `(start node, edge, end node)` whose `start node` is
|
||||
in the `expanded` set
|
||||
iii. If no triplet was selected, choose a new starting node that isn't in
|
||||
`expanded` and continue expanding
|
||||
iv. Repeat steps ii. -- iii. until all triplets have been selected
|
||||
and store that as a variation of the `MATCH` clauses
|
||||
2. Do step 1.1. for `OPTIONAL MATCH` and `MERGE` clauses
|
||||
3. Take all combinations of the generated `MATCH`, `OPTIONAL MATCH` and
|
||||
`MERGE` and store them as variations of the query part.
|
||||
2. For each combination of query part variations:
|
||||
1. Generate a plan using the rule based planner
|
||||
|
||||
### Rule Based Planner
|
||||
|
||||
The `RuleBasedPlanner` generates a single plan for a single query. A plan is
|
||||
generated by following hardcoded rules for producing logical operators. The
|
||||
following sections are an overview on how each openCypher clause is converted
|
||||
to a `LogicalOperator`.
|
||||
|
||||
#### MATCH
|
||||
|
||||
`MATCH` clause is used to specify which patterns need to be searched for in
|
||||
the database. These patterns are normalized in the preprocess step to be
|
||||
represented as triplets `(start node, edge, end node)`. When there is no edge,
|
||||
then the triplet is reduced only to the `start node`. Generating the operators
|
||||
is done by looping over these triplets.
|
||||
|
||||
##### Searching for Nodes
|
||||
|
||||
The simplest search is finding standalone nodes. For example, `MATCH (n)`
|
||||
will find all the nodes in the graph. This is accomplished by generating a
|
||||
`ScanAll` operator and forwarding the node symbol which should store the
|
||||
results. In this case, all the nodes will be referenced by `n`.
|
||||
|
||||
Multiple nodes can be specified in a single match, e.g. `MATCH (n), (m)`.
|
||||
Planning is done by repeating the same steps for each sub pattern (separated
|
||||
by a comma). In this case, we would get 2 `ScanAll` operators chained one
|
||||
after the other. An optimization can be obtained if the node in the pattern is
|
||||
already searched for. In `MATCH (n), (n)` we can drop the second `ScanAll`
|
||||
operator since we have already generated it for the first node.
|
||||
|
||||
##### Searching for Relationships
|
||||
|
||||
A more advanced search includes finding nodes with relationships. For example,
|
||||
`MATCH (n)-[r]-(m)` should find every pair of connected nodes in the database.
|
||||
This means, that if a single node has multiple connections, it will be
|
||||
repeated for each combination of pairs. The generation of operators starts
|
||||
from the first node in the pattern. If we are referencing a new starting node,
|
||||
we need to generate a `ScanAll` which finds all the nodes and stores them
|
||||
into `n`. Then, we generate an `Expand` operator which reads the `n` and
|
||||
traverses all the edges of that node. The edge is stored into `r`, while the
|
||||
destination node is stored in `m`.
|
||||
|
||||
Matching multiple relationships proceeds similarly, by repeating the same
|
||||
steps. The only difference is that we need to ensure different edges in the
|
||||
search pattern, map to different edges in the graph. This means that after each
|
||||
`Expand` operator, we need to generate an `EdgeUniquenessFilter`. We provide
|
||||
this operator with a list of symbols for the previously matched edges and the
|
||||
symbol for the current edge.
|
||||
|
||||
For example.
|
||||
|
||||
MATCH (n)-[r1]-(m)-[r2]-(l)
|
||||
|
||||
The above is preprocessed into
|
||||
|
||||
MATCH (n)-[r1]-(m), (m)-[r2]-(l)
|
||||
|
||||
Then we look at each triplet in order and perform the described steps. This
|
||||
way, we would generate:
|
||||
|
||||
ScanAll (n) > Expand (n, r1, m) > Expand (m, r2, l) >
|
||||
EdgeUniquenessFilter ([r1], r2)
|
||||
|
||||
Note that we don't need to make `EdgeUniquenessFilter` after the first
|
||||
`Expand`, since there are no edges to compare to. This filtering needs to work
|
||||
across multiple pattern, but inside a *single* `MATCH` clause.
|
||||
|
||||
Let's take a look at the following.
|
||||
|
||||
MATCH (n)-[r1]-(m), (m)-[r2]-(l)
|
||||
|
||||
We would also generate the exact same operators.
|
||||
|
||||
ScanAll (n) > Expand (n, r1, m) > Expand (m, r2, l) >
|
||||
EdgeUniquenessFilter ([r1], r2)
|
||||
|
||||
On the other hand,
|
||||
|
||||
MATCH (n)-[r1]-(m) MATCH (m)-[r2]-(l)-[r3]-(i)
|
||||
|
||||
would reset the uniqueness filtering at the start of the second match. This
|
||||
would mean that we output the following:
|
||||
|
||||
ScanAll (n) > Expand (n, r1, m) > Expand (m, r2, l) > Expand (l, r3, i) >
|
||||
EdgeUniquenessFilter ([r2], r3)
|
||||
|
||||
There is a difference in how we handle edge uniqueness compared to Neo4j.
|
||||
Neo4j does not allow searching for a single edge multiple times, but we've
|
||||
decided to support that.
|
||||
|
||||
For example, the user can say the following.
|
||||
|
||||
MATCH (n)-[r]-(m)-[r]-l
|
||||
|
||||
We would ensure that both `r` variables match to the same edge. In our
|
||||
terminology, we call this the *edge cycle*. For the above example, we would
|
||||
generate this plan.
|
||||
|
||||
ScanAll (n) > Expand (n, r, m) > Expand (m, r, l)
|
||||
|
||||
We do not put an `EdgeUniquenessFilter` operator between 2 `Expand`
|
||||
operators and we tell the 2nd `Expand` that it is an edge cycle. This, 2nd
|
||||
`Expand` will ensure we have matched both the same edges.
|
||||
|
||||
##### Filtering
|
||||
|
||||
To narrow the search down, the patterns in `MATCH` can have filtered labels
|
||||
and properties. A more general filtering is done using the accompanying
|
||||
`WHERE` clause. During the preprocess step, all filters are collected and
|
||||
extracted into expressions. Additional information on which symbols are used
|
||||
is also stored. This way, each time we generate a `ScanAll` or `Expand`, we
|
||||
look at all the filters to see if any of them can be used. I.e. if the symbols
|
||||
they use have been bound by a newly produced operator. If a filter expression
|
||||
can be used, we immediately add a `Filter` operator with that expression.
|
||||
|
||||
For example.
|
||||
|
||||
MATCH (n)-[r]-(m :label) WHERE n.prop = 42
|
||||
|
||||
We would produce:
|
||||
|
||||
ScanAll (n) > Filter (n.prop) > Expand (n, r, m) > Filter (m :label)
|
||||
|
||||
This means that the same plan is generated for the query:
|
||||
|
||||
MATCH (n {prop: 42})-[r]-(m :label)
|
||||
|
||||
#### OPTIONAL
|
||||
|
||||
If a `MATCH` clause is preceded by `OPTIONAL`, then we need to generate a plan
|
||||
such that we produce results even if we fail to match anything. This is
|
||||
accomplished by generating an `Optional` operator, which takes 2 operator
|
||||
trees:
|
||||
|
||||
* input operation and
|
||||
* optional operation.
|
||||
|
||||
The input is the operation we generated for the part of the query before
|
||||
`OPTIONAL MATCH`. For the optional operation, we simply generate the `OPTIONAL
|
||||
MATCH` part just like we would for regular `MATCH`. In addition to operations,
|
||||
we need to send the symbols which are set during optional matching to the
|
||||
`Optional` operator. The operator will reset values of those symbols to
|
||||
`null`, when the optional part fails to match.
|
||||
|
||||
#### RETURN & WITH
|
||||
|
||||
`RETURN` and `WITH` clauses are very similar to each other. The only
|
||||
difference is that `WITH` separates parts of the query and can be paired with
|
||||
`WHERE` clause.
|
||||
|
||||
The common part is generating operators for the body of the clause. Separation
|
||||
of query parts is mostly done in semantic analysis, which checks that only the
|
||||
symbols exposed through `WITH` are visible in the query parts after the
|
||||
clause. The minor part is done in planning.
|
||||
|
||||
##### Named Results
|
||||
|
||||
Both clauses contain multiple named expressions (`expr AS name`) which are
|
||||
used to generate `Produce` operator.
|
||||
|
||||
##### Aggregations
|
||||
|
||||
If an expression contains an aggregation operator (`sum`, `avg`, ...) we need
|
||||
to plan the `Aggregate` operator as input to `Produce`. This case is more
|
||||
complex, because aggregation in openCypher can perform implicit grouping of
|
||||
results used for aggregation.
|
||||
|
||||
For example, `WITH/RETURN sum(n.x) AS s, n.y AS group` will implicitly group
|
||||
by `n.y` expression.
|
||||
|
||||
Another, obscure grouping can be achieved with `RETURN sum(n.a) + n.b AS s`.
|
||||
Here, the `n.b` will be used for grouping, even though both the `sum` and
|
||||
`n.b` are in the same named expression.
|
||||
|
||||
Therefore, we need to collect all expressions which do not contain
|
||||
aggregations and use them for grouping. You may have noticed that in the last
|
||||
example `sum` is actually a sub-expression of `+`. `Aggregate` operator does
|
||||
not see that (nor it should), so the responsibility of evaluating that falls
|
||||
on `Produce`. One way is for `Aggregate` to store results of grouping
|
||||
expressions on the frame in addition to aggregation results. Unfortunately,
|
||||
this would require rewiring named expressions in `Produce` to reference
|
||||
already evaluated expressions. In the current implementation, we opted for
|
||||
`Aggregate` to store only aggregation results on the frame, while `Produce`
|
||||
will re-evaluate all the other (grouping) expressions. To handle that, symbols
|
||||
which are used in expressions are passed to `Aggregate`, so that they can be
|
||||
remembered. `Produce` will read those symbols from the frame and use it to
|
||||
re-evaluate the needed expressions.
|
||||
|
||||
##### Accumulation
|
||||
|
||||
After we have `Produce` and potentially `Aggregate`, we need to handle a
|
||||
special case when the part of the query before `RETURN` or `WITH` performs
|
||||
updates. For that, we want to run that part of the query fully, so that we get
|
||||
the latest results. This is accomplished by adding `Accumulate` operator as
|
||||
input to `Aggregate` or `Produce` (if there is no aggregation). Accumulation
|
||||
will store all the values for all the used symbols inside `RETURN` and `WITH`,
|
||||
so that they can be used in the operator which follows. This way, only parts
|
||||
of the frame are copied, instead of the whole frame. Here is a minor
|
||||
difference between planning `WITH`, compared to `RETURN`. Since `WITH` can
|
||||
separate writing from reading, we need to advance the transaction command.
|
||||
This enables the later, read parts of the query to obtain the newest changes.
|
||||
This is supported by passing `advance_command` flag to `Accumulate` operator.
|
||||
|
||||
In the simplest case, common to both clauses, we have `Accumulate > Aggregate
|
||||
> Produce` operators, where `Accumulate` and `Aggregate` may be left out.
|
||||
|
||||
##### Ordering
|
||||
|
||||
Planning `ORDER BY` is simple enough. Since it may see new symbols (filled in
|
||||
`Produce`), we add the `OrderBy` operator at the end. The operator will change
|
||||
the order of produced results, so we pass it the ordering expressions and the
|
||||
output symbols of named expressions.
|
||||
|
||||
##### Filtering
|
||||
|
||||
A final difference in `WITH`, is when it contains a `WHERE` clause. For that,
|
||||
we simply generate the `Filter` operator, appended after `Produce` or
|
||||
`OrderBy` (depending which operator is last).
|
||||
|
||||
##### Skipping and Limiting
|
||||
|
||||
If we have `SKIP` or `LIMIT`, we generate `Skip` or `Limit` operators,
|
||||
respectively. These operators are put at the end of the clause.
|
||||
|
||||
This placement may have some unexpected behaviour when combined with
|
||||
operations that update the graph. For example.
|
||||
|
||||
MATCH (n) SET n.x = n.x + 1 RETURN n LIMIT 1
|
||||
|
||||
The above query may be interpreted as if the `SET` will be done only once.
|
||||
Since this is a write query, we need to accumulate results, so the part before
|
||||
`RETURN` will execute completely. The accumulated results will be yielded up
|
||||
to the given limit, and the user would get only the first `n` that was
|
||||
updated. This may confuse the user because in reality, every node in the
|
||||
database had been updated.
|
||||
|
||||
Note that `Skip` always comes before `Limit`. In the current implementation,
|
||||
they are generated directly one after the other.
|
||||
|
||||
#### CREATE
|
||||
|
||||
`CREATE` clause is used to create nodes and edges (relationships).
|
||||
|
||||
For multiple `CREATE` clauses or multiple creation patterns in a single
|
||||
clause, we perform the same, following steps.
|
||||
|
||||
##### Creating a Single Node
|
||||
|
||||
A node is created by simply specifying a node pattern.
|
||||
|
||||
For example `CREATE (n :label {property: "value"}), ()` would create 2 nodes.
|
||||
The 1st one would be created with a label and a property. This node could be
|
||||
referenced later in the query, by using the variable `n`. The 2nd node cannot
|
||||
be referenced and it would be created without any labels nor properties. For
|
||||
node creation, we generate a `CreateNode` operator and pass it all the details
|
||||
of node creation: variable symbol, labels and properties. In the mentioned
|
||||
example, we would have `CreateNode > CreateNode`.
|
||||
|
||||
##### Creating a Relationship
|
||||
|
||||
To create a relationship, the `CREATE` clause must contain a pattern with a
|
||||
directed edge. Compared to creating a single node, this case is a bit more
|
||||
complicated, because either side of the edge may not exist. By exist, we mean
|
||||
that the endpoint is a variable which already references a node.
|
||||
|
||||
For example, `MATCH (n) CREATE (n)-[r]->(m)` would create an edge `r` and a
|
||||
node `m` for each matched node `n`. If we focus on the `CREATE` part, we
|
||||
generate `CreateExpand (n, r, m)` where `n` already exists (refers to matched
|
||||
node) and `m` would be newly created along with edge `r`. If we had only
|
||||
`CREATE (n)-[r]->(m)`, then we would need to create both nodes of the edge
|
||||
`r`. This is done by generating `CreateNode (n) > CreateExpand(n, r, m)`. The
|
||||
final case is when both endpoints refer to an existing node. For example, when
|
||||
adding a node with a cyclical connection `CREATE (n)-[r]->(n)`. In this case,
|
||||
we would generate `CreateNode (n) > CreateExpand (n, r, n)`. We would tell
|
||||
`CreateExpand` to only create the edge `r` between the already created `n`.
|
||||
|
||||
#### MERGE
|
||||
|
||||
Although the merge operation is complex, planning turns out to be relatively
|
||||
simple. The pattern inside the `MERGE` clause is used for both matching and
|
||||
creating. Therefore, we create 2 operator trees, one for each action.
|
||||
|
||||
For example.
|
||||
|
||||
MERGE (n)-[r:r]-(m)
|
||||
|
||||
We would generate a single `Merge` operator which has the following.
|
||||
|
||||
* No input operation (since it is not preceded by any other clause).
|
||||
|
||||
* On match operation
|
||||
|
||||
`ScanAll (n) > Expand (n, r, m) > Filter (r)`
|
||||
|
||||
* On create operation
|
||||
|
||||
`CreateNode (n) > CreateExpand (n, r, m)`
|
||||
|
||||
In cases when `MERGE` contains `ON MATCH` and `ON CREATE` parts, we simply
|
||||
append their operations to the respective operator trees.
|
||||
|
||||
Observe the following example.
|
||||
|
||||
MERGE (n)-[r:r]-(m) ON MATCH SET n.x = 42 ON CREATE SET m :label
|
||||
|
||||
The `Merge` would be generated with the following.
|
||||
|
||||
* No input operation (again, since there is no clause preceding it).
|
||||
|
||||
* On match operation
|
||||
|
||||
`ScanAll (n) > Expand (n, r, m) > Filter (r) > SetProperty (n.x, 42)`
|
||||
|
||||
* On create operation
|
||||
|
||||
`CreateNode (n) > CreateExpand (n, r, m) > SetLabels (n, :label)`
|
||||
|
||||
When we have preceding clauses, we simply put their operator as input to
|
||||
`Merge`.
|
||||
|
||||
MATCH (n) MERGE (n)-[r:r]-(m)
|
||||
|
||||
The above would be generated as
|
||||
|
||||
ScanAll (n) > Merge (on_match_operation, on_create_operation)
|
||||
|
||||
Here we need to be careful to recognize which symbols are already declared.
|
||||
But, since the `on_match_operation` uses the same algorithm for generating a
|
||||
`Match`, that problem is handled there. The same should hold for
|
||||
`on_create_operation`, which uses the process of generating a `Create`. So,
|
||||
finally for this example, the `Merge` would have:
|
||||
|
||||
* Input operation
|
||||
|
||||
`ScanAll (n)`
|
||||
|
||||
* On match operation
|
||||
|
||||
`Expand (n, r, m) > Filter (r)`
|
||||
|
||||
Note that `ScanAll` is not needed since we get the nodes from input.
|
||||
|
||||
* On create operation
|
||||
|
||||
`CreateExpand (n, r, m)`
|
||||
|
||||
Note that `CreateNode` is dropped, since we want to expand the existing one.
|
||||
|
||||
## Logical Plan Postprocessing
|
||||
|
||||
NOTE: TODO
|
||||
|
||||
## Cost Estimation
|
||||
|
||||
NOTE: TODO
|
||||
|
||||
## Distributed Planning
|
||||
|
||||
NOTE: TODO
|
||||
@@ -1,134 +0,0 @@
|
||||
# Semantic Analysis and Symbol Generation
|
||||
|
||||
In this phase, various semantic and variable type checks are performed.
|
||||
Additionally, we generate symbols which map AST nodes to stored values
|
||||
computed from evaluated expressions.
|
||||
|
||||
## Symbol Generation
|
||||
|
||||
Implementation can be found in `query/frontend/semantic/symbol_generator.cpp`.
|
||||
|
||||
Symbols are generated for each AST node that represents data that needs to
|
||||
have storage. Currently, these are:
|
||||
|
||||
* `NamedExpression`
|
||||
* `CypherUnion`
|
||||
* `Identifier`
|
||||
* `Aggregation`
|
||||
|
||||
You may notice that the above AST nodes may not correspond to something named
|
||||
by a user. For example, `Aggregation` can be a part of larger expression and
|
||||
thus remain unnamed. The reason we still generate symbols is to have a uniform
|
||||
behaviour when executing a query as well as allow for caching the results of
|
||||
expression evaluation.
|
||||
|
||||
AST nodes do not actually store a `Symbol` instance, instead they have a
|
||||
`int32_t` index identifying the symbol in the `SymbolTable` class. This is
|
||||
done to minimize the size of AST types as well as allow easier sharing of same
|
||||
symbols with multiple instances of AST nodes.
|
||||
|
||||
The storage for evaluated data is represented by the `Frame` class. Each
|
||||
symbol determines a unique position in the frame. During interpretation,
|
||||
evaluation of expressions which have a symbol will either read or store values
|
||||
in the frame. For example, instance of an `Identifier` will use the symbol to
|
||||
find and read the value from `Frame`. On the other hand, `NamedExpression`
|
||||
will take the result of evaluating its own expression and store it in the
|
||||
`Frame`.
|
||||
|
||||
When a symbol is created, context of creation is used to assign a type to that
|
||||
symbol. This type is used for simple type checking operations. For example,
|
||||
`MATCH (n)` will create a symbol for variable `n`. Since the `MATCH (n)`
|
||||
represents finding a vertex in the graph, we can set `Symbol::Type::Vertex`
|
||||
for that symbol. Later, for example in `MATCH ()-[n]-()` we see that variable
|
||||
`n` is used as an edge. Since we already have a symbol for that variable, we
|
||||
detect this type mismatch and raise a `SemanticException`.
|
||||
|
||||
Basic rule of symbol generation, is that variables inside `MATCH`, `CREATE`,
|
||||
`MERGE`, `WITH ... AS` and `RETURN ... AS` clauses establish new symbols.
|
||||
|
||||
### Symbols in Patterns
|
||||
|
||||
Inside `MATCH`, symbols are created only if they didn't exist before. For
|
||||
example, patterns in `MATCH (n {a: 5})--(m {b: 5}) RETURN n, m` will create 2
|
||||
symbols: one for `n` and one for `m`. `RETURN` clause will, in turn, reference
|
||||
those symbols. Symbols established in a part of pattern are immediately bound
|
||||
and visible in later parts. For example, `MATCH (n)--(n)` will create a symbol
|
||||
for variable `n` for 1st `(n)`. That symbol is referenced in 2nd `(n)`. Note
|
||||
that the symbol is not bound inside 1st `(n)` itself. What this means is that,
|
||||
for example, `MATCH (n {a: n.b})` should raise an error, because `n` is not
|
||||
yet bound when encountering `n.b`. On the other hand,
|
||||
`MATCH (n)--(n {a: n.b})` is fine.
|
||||
|
||||
The `CREATE` is similar to `MATCH`, but it *always* establishes symbols for
|
||||
variables which create graph elements. What this means is that, for example
|
||||
`MATCH (n) CREATE (n)` is not allowed. `CREATE` wants to create a new node,
|
||||
for which we already have a symbol. In such a case, we need to throw an error
|
||||
that the variable `n` is being redeclared. On the other hand `MATCH (n) CREATE
|
||||
(n)-[r :r]->(n)` is fine, because `CREATE` will only create the edge `r`,
|
||||
connecting the already existing node `n`. Remaining behaviour is the same as
|
||||
in `MATCH`. This means that we can simplify `CREATE` to be like `MATCH` with 2
|
||||
special cases.
|
||||
|
||||
1. Are we creating a node, i.e. `CREATE (n)`? If yes, then the symbol for
|
||||
`n` must not have been created before. Otherwise, we reference the
|
||||
existing symbol.
|
||||
2. Are we creating an edge, i.e. we encounter a variable for an edge inside
|
||||
`CREATE`? If yes, then that variable must not reference a symbol.
|
||||
|
||||
The `MERGE` clause is treated the same as `CREATE` with regards to symbol
|
||||
generation. The only difference is that we allow bidirectional edges in the
|
||||
pattern. When creating such a pattern, the direction of the created edge is
|
||||
arbitrarily determined.
|
||||
|
||||
### Symbols in WITH and RETURN
|
||||
|
||||
In addition to patterns, new symbols are established in the `WITH` clause.
|
||||
This clause makes the new symbols visible *only* to the rest of the query.
|
||||
For example, `MATCH (old) WITH old AS new RETURN new, old` should raise an
|
||||
error that `old` is unbound inside `RETURN`.
|
||||
|
||||
There is a special case with symbol visibility in `WHERE` and `ORDER BY`. They
|
||||
need to see both the old and the new symbols. Therefore `MATCH (old) RETURN
|
||||
old AS new ORDER BY old.prop` needs to work. On the other hand, if we perform
|
||||
aggregations inside `WITH` or `RETURN`, then the old symbols should not be
|
||||
visible neither in `WHERE` nor in `ORDER BY`. Since the aggregation has to go
|
||||
through all the results in order to generate the final value, it makes no
|
||||
sense to store old symbols and their values. A query like `MATCH (old) WITH
|
||||
SUM(old.prop) AS sum WHERE old.prop = 42 RETURN sum` needs to raise an error
|
||||
that `old` is unbound inside `WHERE`.
|
||||
|
||||
For cases when `SKIP` and `LIMIT` appear, we disallow any identifiers from
|
||||
appearing in their expressions. Basically, `SKIP` and `LIMIT` can only be
|
||||
constant expressions[^1]. For example, `MATCH (old) RETURN old AS new SKIP
|
||||
new.prop` needs to raise that variables are not allowed in `SKIP`. It makes no
|
||||
sense to allow variables, since their values may vary on each iteration. On
|
||||
the other hand, we could support variables to constant expressions, but for
|
||||
simplicity we do not. For example, `MATCH (old) RETURN old, 2 AS limit_var
|
||||
LIMIT limit_var` would still throw an error.
|
||||
|
||||
Finally, we generate symbols for names created in `RETURN` clause. These
|
||||
symbols are used for the final results of a query.
|
||||
|
||||
NOTE: New symbols in `WITH` and `RETURN` should be unique. This means that
|
||||
`WITH a AS same, b AS same` is not allowed, neither is a construct like
|
||||
`RETURN 2, 2`
|
||||
|
||||
### Symbols in Functions which Establish New Scope
|
||||
|
||||
Symbols can also be created in some functions. These functions usually take an
|
||||
expression, bind a single variable and run the expression inside the newly
|
||||
established scope.
|
||||
|
||||
The `all` function takes a list, creates a variable for list element and runs
|
||||
the predicate expression. For example:
|
||||
|
||||
MATCH (n) RETURN n, all(n IN n.prop_list WHERE n < 42)
|
||||
|
||||
We create a new symbol for use inside `all`, this means that the `WHERE n <
|
||||
42` uses the `n` which takes values from a `n.prop_list` elements. The
|
||||
original `n` bound by `MATCH` is not visible inside the `all` function, but it
|
||||
is visible outside. Therefore, the `RETURN n` and `n.prop_list` reference the
|
||||
`n` from `MATCH`.
|
||||
|
||||
[^1]: Constant expressions are expressions for which the result can be
|
||||
computed at compile time.
|
||||
@@ -1,108 +0,0 @@
|
||||
# Quick Start
|
||||
|
||||
A short chapter on downloading the Memgraph source, compiling and running.
|
||||
|
||||
## Obtaining the Source Code
|
||||
|
||||
Memgraph uses `git` for source version control. You will need to install `git`
|
||||
on your machine before you can download the source code.
|
||||
|
||||
On Debian systems, you can do it inside a terminal with the following
|
||||
command:
|
||||
|
||||
sudo apt-get install git
|
||||
|
||||
On ArchLinux or Gentoo, you probably already know what to do.
|
||||
|
||||
After installing `git`, you are now ready to fetch your own copy of Memgraph
|
||||
source code. Run the following command:
|
||||
|
||||
git clone https://phabricator.memgraph.io/diffusion/MG/memgraph.git
|
||||
|
||||
The above will create a `memgraph` directory and put all source code there.
|
||||
|
||||
## Compiling Memgraph
|
||||
|
||||
With the source code, you are now ready to compile Memgraph. Well... Not
|
||||
quite. You'll need to download Memgraph's dependencies first.
|
||||
|
||||
In your terminal, position yourself in the obtained memgraph directory.
|
||||
|
||||
cd memgraph
|
||||
|
||||
### Installing Dependencies
|
||||
|
||||
On Debian systems, dependencies that are required by the codebase should be
|
||||
setup by running the `init` script:
|
||||
|
||||
./init -s
|
||||
|
||||
Currently, other systems aren't supported in the `init` script. But you can
|
||||
issue the needed steps manually. First run the `init` script.
|
||||
|
||||
./init
|
||||
|
||||
The script will output the required packages, which you should be able to
|
||||
install via your favorite package manager. For example, `pacman` on ArchLinux.
|
||||
After installing the packages, issue the following commands:
|
||||
|
||||
mkdir -p build
|
||||
./libs/setup.sh
|
||||
|
||||
### Compiling
|
||||
|
||||
Memgraph is compiled using our own custom toolchain that can be obtained from
|
||||
[Toolchain repository](https://deps.memgraph.io/toolchain). You should read
|
||||
the `README.txt` file in the repository and install the apropriate toolchain
|
||||
for your distribution. After you have installed the toolchain you should read
|
||||
the instructions for the toolchain in the toolchain install directory
|
||||
(`/opt/toolchain-vXYZ/README.md`) and install dependencies that are necessary
|
||||
to run the toolchain.
|
||||
|
||||
When you want to compile Memgraph you should activate the toolchain using the
|
||||
prepared toolchain activation script that is also described in the toolchain
|
||||
`README`.
|
||||
|
||||
NOTE: You *must* activate the toolchain every time you want to compile
|
||||
Memgraph!
|
||||
|
||||
You should now activate the toolchain in your console.
|
||||
|
||||
source /opt/toolchain-vXYZ/activate
|
||||
|
||||
With all of the dependencies installed and the build environment set-up, you
|
||||
need to configure the build system. To do that, execute the following:
|
||||
|
||||
cd build
|
||||
cmake ..
|
||||
|
||||
If everything went OK, you can now, finally, compile Memgraph.
|
||||
|
||||
make -j$(nproc)
|
||||
|
||||
### Running
|
||||
|
||||
After the compilation verify that Memgraph works:
|
||||
|
||||
./memgraph --version
|
||||
|
||||
To make extra sure, run the unit tests:
|
||||
|
||||
ctest -R unit -j$(nproc)
|
||||
|
||||
## Problems
|
||||
|
||||
If you have any trouble running the above commands, contact your nearest
|
||||
developer who successfully built Memgraph. Ask for help and insist on getting
|
||||
this document updated with correct steps!
|
||||
|
||||
## Next Steps
|
||||
|
||||
Familiarise yourself with our code conventions and guidelines:
|
||||
|
||||
* [C++ Code](cpp-code-conventions.md)
|
||||
* [Other Code](other-code-conventions.md)
|
||||
* [Code Review Guidelines](code-review.md)
|
||||
|
||||
Take a look at the list of [required reading](required-reading.md) for
|
||||
brushing up on technical skills.
|
||||
@@ -1,129 +0,0 @@
|
||||
# Required Reading
|
||||
|
||||
This chapter lists a few books that should be read by everyone working on
|
||||
Memgraph. Since Memgraph is developed primarily with C++, Python and Common
|
||||
Lisp, books are oriented around those languages. Of course, there are plenty
|
||||
of general books which will help you improve your technical skills (such as
|
||||
"The Pragmatic Programmer", "The Mythical Man-Month", etc.), but they are not
|
||||
listed here. This way the list should be kept short and the *required* part in
|
||||
"Required Reading" more easily honored.
|
||||
|
||||
Some of these books you may find in our office, so feel free to pick them up.
|
||||
If any are missing and you would like a physical copy, don't be afraid to
|
||||
request the book for our office shelves.
|
||||
|
||||
Besides reading, don't get stuck in a rut and be a
|
||||
[Blub Programmer](http://www.paulgraham.com/avg.html).
|
||||
|
||||
## Effective C++ by Scott Meyers
|
||||
|
||||
Required for C++ developers.
|
||||
|
||||
The book is a must-read as it explains most common gotchas of using C++. After
|
||||
reading this book, you are good to write competent C++ which will pass code
|
||||
reviews easily.
|
||||
|
||||
## Effective Modern C++ by Scott Meyers
|
||||
|
||||
Required for C++ developers.
|
||||
|
||||
This is a continuation of the previous book, it covers updates to C++ which
|
||||
came with C++11 and later. The book isn't as imperative as the previous one,
|
||||
but it will make you aware of modern features we are using in our codebase.
|
||||
|
||||
## Practical Common Lisp by Peter Siebel
|
||||
|
||||
Required for Common Lisp developers.
|
||||
|
||||
Free: http://www.gigamonkeys.com/book/
|
||||
|
||||
We use Common Lisp to generate C++ code and make our lives easier.
|
||||
Unfortunately, not many developers are familiar with the language. This book
|
||||
will make you familiar very quickly as it has tons of very practical
|
||||
exercises. E.g. implementing unit testing library, serialization library and
|
||||
bundling all that to create a mp3 music server.
|
||||
|
||||
## Effective Python by Brett Slatkin
|
||||
|
||||
(Almost) required reading for Python developers.
|
||||
|
||||
Why the "almost"? Well, Python is relatively easy to pick up and you will
|
||||
probably learn all the gotchas during code review from someone more
|
||||
experienced. This makes the book less necessary for a newcomer to Memgraph,
|
||||
but the book is not advanced enough to delegate it to
|
||||
[Advanced Reading](#advanced-reading). The book is written in similar vein as
|
||||
the "Effective C++" ones and will make you familiar with nifty Python features
|
||||
that make everyone's lives easier.
|
||||
|
||||
# Advanced Reading
|
||||
|
||||
The books listed below are not required reading, but you may want to read them
|
||||
at some point when you feel comfortable enough.
|
||||
|
||||
## Design Patterns by Gamma et. al.
|
||||
|
||||
Recommended for C++ developers.
|
||||
|
||||
This book is highly divisive because it introduced a culture centered around
|
||||
design patterns. The main issues is overuse of patterns which complicates the
|
||||
code. This has made many Java programs to serve as examples of highly
|
||||
complicated, "enterprise" code.
|
||||
|
||||
Unfortunately, design patterns are pretty much missing
|
||||
language features. This is most evident in dynamic languages such as Python
|
||||
and Lisp, as demonstrated by
|
||||
[Peter Norvig](http://www.norvig.com/design-patterns/).
|
||||
|
||||
Or as [Paul Graham](http://www.paulgraham.com/icad.html) put it:
|
||||
|
||||
```
|
||||
This practice is not only common, but institutionalized. For example, in the
|
||||
OO world you hear a good deal about "patterns". I wonder if these patterns are
|
||||
not sometimes evidence of case (c), the human compiler, at work. When I see
|
||||
patterns in my programs, I consider it a sign of trouble. The shape of a
|
||||
program should reflect only the problem it needs to solve. Any other
|
||||
regularity in the code is a sign, to me at least, that I'm using abstractions
|
||||
that aren't powerful enough-- often that I'm generating by hand the expansions
|
||||
of some macro that I need to write
|
||||
```
|
||||
|
||||
After presenting the book so negatively, why you should even read it then?
|
||||
Well, it is good to be aware of those design patterns and use them when
|
||||
appropriate. They can improve modularity and reuse of the code. You will also
|
||||
find examples of such patterns in our code, primarily Strategy and Visitor
|
||||
patterns. The book is also a good stepping stone to more advanced reading
|
||||
about software design.
|
||||
|
||||
## Modern C++ Design by Andrei Alexandrescu
|
||||
|
||||
Recommended for C++ developers.
|
||||
|
||||
This book can be treated as a continuation of the previous "Design Patterns"
|
||||
book. It introduced "dark arts of template meta-programming" to the world.
|
||||
Many of the patterns are converted to use C++ templates which makes them even
|
||||
better for reuse. But, like the previous book, there are downsides if used too
|
||||
much. You should approach it with a critical eye and it will help you
|
||||
understand ideas that are used in some parts of our codebase.
|
||||
|
||||
## Large Scale C++ Software Design by John Lakos
|
||||
|
||||
Recommended for C++ developers.
|
||||
|
||||
An old book, but well worth the read. Lakos presents a very pragmatic view of
|
||||
writing modular software and how it affects both development time as well as
|
||||
program runtime. Some things are outdated or controversial, but it will help
|
||||
you understand how the whole C++ process of working in a large team, compiling
|
||||
and linking affects development.
|
||||
|
||||
## On Lisp by Paul Graham
|
||||
|
||||
Recommended for Common Lisp developers.
|
||||
|
||||
Free: http://www.paulgraham.com/onlisp.html
|
||||
|
||||
An excellent continuation to "Practical Common Lisp". It starts of slow, as if
|
||||
introducing the language, but very quickly picks up speed. The main meat of
|
||||
the book are macros and their uses. From using macros to define cooperative
|
||||
concurrency to including Prolog as if it's part of Common Lisp. The book will
|
||||
help you understand more advanced macros that are occasionally used in our
|
||||
Lisp C++ Preprocessor (LCP).
|
||||
@@ -1,110 +0,0 @@
|
||||
# DatabaseAccessor
|
||||
|
||||
A `DatabaseAccessor` actually wraps a transactional access to database
|
||||
data, for a single transaction. In that sense the naming is bad. It
|
||||
encapsulates references to the database and the transaction object.
|
||||
|
||||
It contains logic for working with database content (graph element
|
||||
data) in the context of a single transaction. All CRUD operations are
|
||||
performed within a single transaction (as Memgraph is a transactional
|
||||
database), and therefore iteration over data, finding a specific graph
|
||||
element etc are all functionalities of a `GraphDbAccessor`.
|
||||
|
||||
In single-node Memgraph the database accessor also defined the lifetime
|
||||
of a transaction. Even though a `Transaction` object was owned by the
|
||||
transactional engine, it was `GraphDbAccessor`'s lifetime that object
|
||||
was bound to (the transaction was implicitly aborted in
|
||||
`GraphDbAccessor`'s destructor, if it was not explicitly ended before
|
||||
that).
|
||||
|
||||
# RecordAccessor
|
||||
|
||||
It is important to understand data organization and access in the
|
||||
storage layer. This discussion pertains to vertices and edges as graph
|
||||
elements that the end client works with.
|
||||
|
||||
Memgraph uses MVCC (documented on it's own page). This means that for
|
||||
each graph element there could be different versions visible to
|
||||
different currently executing transactions. When we talk about a
|
||||
`Vertex` or `Edge` as a data structure we typically mean one of those
|
||||
versions. In code this semantic is implemented so that both those classes
|
||||
inherit `mvcc::Record`, which in turn inherits `mvcc::Version`.
|
||||
|
||||
Handling MVCC and visibility is not in itself trivial. Next to that,
|
||||
there is other book-keeping to be performed when working with data. For
|
||||
that reason, Memgraph uses "accessors" to define an API of working with
|
||||
data in a safe way. Most of the code in Memgraph (for example the
|
||||
interpretation code) should work with accessors. There is a
|
||||
`RecordAccessor` as a base class for `VertexAccessor` and
|
||||
`EdgeAccessor`. Following is an enumeration of their purpose.
|
||||
|
||||
### Data access
|
||||
|
||||
The client interacts with Memgraph using the Cypher query language. That
|
||||
language has certain semantics which imply that multiple versions of the
|
||||
data need to be visible during the execution of a single query. For
|
||||
example: expansion over the graph is always done over the graph state as
|
||||
it was at the beginning of the transaction.
|
||||
|
||||
The `RecordAccessor` exposes functions to switch between the old and the new
|
||||
versions of the same graph element (intelligently named `SwitchOld` and
|
||||
`SwitchNew`) within a single transaction. In that way the client code
|
||||
(mostly the interpreter) can avoid dealing with the underlying MVCC
|
||||
version concepts.
|
||||
|
||||
### Updates
|
||||
|
||||
Data updates are also done through accessors. Meaning: there are methods
|
||||
on the accessors that modify data, the client code should almost never
|
||||
interact directly with `Vertex` or `Edge` objects.
|
||||
|
||||
The accessor layer takes care of creating version in the MVCC layer and
|
||||
performing updates on appropriate versions.
|
||||
|
||||
Next, for many kinds of updates it is necessary to update the relevant
|
||||
indexes. There are implicit indexes for vertex labels, as
|
||||
well as user-created indexes for (label, property) pairs. The accessor
|
||||
layer takes care of updating the indexes when these values are changed.
|
||||
|
||||
Each update also triggers a log statement in the write-ahead log. This
|
||||
is also handled by the accessor layer.
|
||||
|
||||
### Distributed
|
||||
|
||||
In distributed Memgraph accessors also contain a lot of the remote graph
|
||||
element handling logic. More info on that is available in the
|
||||
documentation for distributed.
|
||||
|
||||
### Deferred MVCC data lookup for Edges
|
||||
|
||||
Vertices and edges are versioned using MVCC. This means that for each
|
||||
transaction an MVCC lookup needs to be done to determine which version
|
||||
is visible to that transaction. This tends to slow things down due to
|
||||
cache invalidations (version lists and versions are stored in arbitrary
|
||||
locations on the heap).
|
||||
|
||||
However, for edges, only the properties are mutable. The edge endpoints
|
||||
and type are fixed once the edge is created. For that reason both edge
|
||||
endpoints and type are available in vertex data, so that when expanding
|
||||
it is not mandatory to do MVCC lookups of versioned, mutable data. This
|
||||
logic is implemented in `RecordAccessor` and `EdgeAccessor`.
|
||||
|
||||
### Exposure
|
||||
|
||||
The original idea and implementation of graph element accessors was that
|
||||
they'd prevent client code from ever interacting with raw `Vertex` or
|
||||
`Edge` data. This however turned out to be impractical when implementing
|
||||
distributed Memgraph and the raw data members have since been exposed
|
||||
(through getters to old and new version pointers). However, refrain from
|
||||
working with that data directly whenever possible! Always consider the
|
||||
accessors to be the first go-to for interacting with data, especially
|
||||
when in the context of a transaction.
|
||||
|
||||
# Skiplist accessor
|
||||
|
||||
The term "accessor" is also used in the context of a skiplist. Every
|
||||
operation on a skiplist must be performed within on an
|
||||
accessor. The skiplist ensures that there will be no physical deletions
|
||||
of an object during the lifetime of an accessor. This mechanism is used
|
||||
to ensure deletion correctness in a highly concurrent container.
|
||||
We only mention that here to avoid confusion regarding terminology.
|
||||
@@ -1,116 +0,0 @@
|
||||
# Label indexes
|
||||
|
||||
These are unsorted indexes that contain all the vertices that have the label
|
||||
the indexes are for (one index per label). These kinds of indexes get
|
||||
automatically generated for each label used in the database.
|
||||
|
||||
### Updating the indexes
|
||||
|
||||
Whenever something gets added to the record we update the index (add that
|
||||
record to index). We keep an index which might contain garbage (not relevant
|
||||
records, because the value got removed or something similar) but we will
|
||||
filter it out when querying the index. We do it like this because we don't
|
||||
have to do bookkeeping and deciding if we update the index on the end of the
|
||||
transaction (commit/abort phase), moreover current interpreter advances the
|
||||
command in transaction and as such assumes that the indexes now contain
|
||||
objects added in the previous command inside this transaction, so we need to
|
||||
update over the whole scope of transaction (whenever something is added to the
|
||||
record).
|
||||
|
||||
### Index Entries Label
|
||||
|
||||
These kinds of indexes are internally keeping track of pair (record, vlist).
|
||||
Why do we need to keep track of exactly those two things?
|
||||
|
||||
Problems with two different approaches
|
||||
|
||||
1) Keep track of just the record:
|
||||
|
||||
- We need the `VersionList` for creating an accessor (this in itself is a
|
||||
deal-breaker).
|
||||
- Semantically it makes sense. An edge/vertex maps bijectionally to a
|
||||
`VersionList`.
|
||||
- We might try to access some members of record while the record is being
|
||||
modified from another thread.
|
||||
- A vertex/edge could get updated, thus expiring the record in the index.
|
||||
The newly created record should be present in the index, but it's not.
|
||||
Without the `VersionList` we can't reach the newly created record.
|
||||
- Probably there are even more reasons... It should be obvious by now that
|
||||
we need the `VersionList` in the index.
|
||||
|
||||
2) Keep track of just the version list:
|
||||
|
||||
- Removing from an index is a problem for two major reasons. First, if we
|
||||
only have the `VersionList`, checking if it should be removed implies
|
||||
checking all the reachable records, which is not thread-safe. Second,
|
||||
there are issues with concurrent removal and insertion. The cleanup thread
|
||||
could determine the vertex/edge should be removed from the index and
|
||||
remove it, while in between those ops another thread attempts to insert
|
||||
the `VersionList` into the index. The insertion does nothing because the
|
||||
`VersionList` is already in, but it gets removed immediately after.
|
||||
|
||||
Because of inability to keep track of just the record, or value, we need to
|
||||
keep track of both of them. Resolution of problems mentioned above, in the
|
||||
same order, with (record, vlist) pair
|
||||
|
||||
- simple `vlist.find(current transaction)` will get us the newest visible
|
||||
record
|
||||
- we'll never try to access some record if it's still being written since we
|
||||
will always operate on vlist.find returned record
|
||||
- newest record will contain that label
|
||||
- since we have (record, vlist) pair as the key in the index when we update
|
||||
and delete in the same time we will never delete the same record, vlist
|
||||
pair we are adding because the record, vlist pair we are deleting is
|
||||
already superseded by a newer record and as such won't be inserted while
|
||||
it's being deleted
|
||||
|
||||
### Querying the index
|
||||
|
||||
We run through the index for the given label and do `vlist.find` operation for
|
||||
the current transaction, and check if the newest return record has that
|
||||
label. If it has it then we return it. By now you are probably wondering
|
||||
aren't we sometimes returning duplicate vlist entries? And you are wondering
|
||||
correctly, we would be returning them, but we are making sure that the entires
|
||||
in the index are sorted by their `vlist*` and as such we can filter consecutive
|
||||
duplicate `vlist*` to only return one of those while still being able to create
|
||||
an iterator to index.
|
||||
|
||||
### Cleaning the index
|
||||
|
||||
Cleaning the index is not as straightforward as it seems as a lot of garbage
|
||||
can accumulate, but it's hard to know when exactly can we delete some (record,
|
||||
vlist) pair. First, let's assume that we are doing the cleaning process at
|
||||
some `transaction_id`, `id` such that there doesn't exist an active transaction
|
||||
with an id lower than `id`.
|
||||
|
||||
We scan through the whole index and for each (record, vlist) pair we first
|
||||
check if it was deleted before the id (i.e. no transaction with an id >= `id`
|
||||
will ever again see that record), if it was deleted before we might naively
|
||||
say that it's safe to delete it, but, we must take into account that when some
|
||||
new record is created from this record (update operation), that record still
|
||||
contains the label but by deleting this record we won't be able to see that
|
||||
vlist because that new record won't add again to index because we didn't
|
||||
explicitly add that label again to it.
|
||||
|
||||
Because of this we have to 'update' this index (record, vlist) pair. We have
|
||||
to update the record to now point to a newer record in vlist, the one that is
|
||||
not deleted yet. We can do that by querying the `version_list` for the last
|
||||
record inside (oldest it has — remember that `mvcc_gc` will re-link not
|
||||
visible records so the last record will be visible for the current GC id).
|
||||
When updating the record inside the index, it's not okay to just update the
|
||||
pointer and leave the index as it is, because with updating the `record*` we
|
||||
might change the relative order of entries inside the index. We first have to
|
||||
re-insert it with new `record*`, and then delete the old entry. And we need to
|
||||
do insertion before the remove operation! Otherwise it could happen that the
|
||||
vlist with a newer record with that label won't exist while some transaction
|
||||
is querying the index.
|
||||
|
||||
Records which we added as a consequence of deleting older records will be
|
||||
eventually removed from the index if they don't contain label because if we
|
||||
see that the record is not deleted we try to check if that record still
|
||||
contains the label. We also need to be careful here because we can't check
|
||||
that while the record is being potentially updated by some transaction (race
|
||||
condition), so we need can check if records still contain label if it's
|
||||
creation id is smaller than our `id`, as that implies that the creating
|
||||
transaction either aborted or committed as our `id` is equal to the oldest
|
||||
active transaction in time of starting the GC.
|
||||
@@ -1,131 +0,0 @@
|
||||
# Property storage
|
||||
|
||||
Although the reader is probably familiar with properties in *Memgraph*, let's
|
||||
briefly recap.
|
||||
|
||||
Both vertices and edges can store an arbitrary number of properties. Properties
|
||||
are, in essence, ordered pairs of property names and property values. Each
|
||||
property name within a single graph element (edge/node) can store a single
|
||||
property value. Property names are represented as strings, while property values
|
||||
must be one of the following types:
|
||||
|
||||
Type | Description
|
||||
-----------|------------
|
||||
`Null` | Denotes that the property has no value. This is the same as if the property does not exist.
|
||||
`String` | A character string, i.e. text.
|
||||
`Boolean` | A boolean value, either `true` or `false`.
|
||||
`Integer` | An integer number.
|
||||
`Float` | A floating-point number, i.e. a real number.
|
||||
`List` | A list containing any number of property values of any supported type. It can be used to store multiple values under a single property name.
|
||||
`Map` | A mapping of string keys to values of any supported type.
|
||||
|
||||
Property values are modeled in a class conveniently called `PropertyValue`.
|
||||
|
||||
## Mapping between property names and property keys.
|
||||
|
||||
Although users think of property names in terms of descriptive strings
|
||||
(e.g. "location" or "department"), *Memgraph* internally converts those names
|
||||
into property keys which are, essentially, unsigned 16-bit integers.
|
||||
|
||||
Property keys are modelled by a not-so-conveniently named class called
|
||||
`Property` which can be found in `storage/types.hpp`. The actual conversion
|
||||
between property names and property keys is done within the `ConcurrentIdMapper`
|
||||
but the internals of that implementation are out of scope for understanding
|
||||
property storage.
|
||||
|
||||
## PropertyValueStore
|
||||
|
||||
Both `Edge` and `Vertex` objects contain an instance of `PropertyValueStore`
|
||||
object which is responsible for storing properties of a corresponding graph
|
||||
element.
|
||||
|
||||
An interface of `PropertyValueStore` is as follows:
|
||||
|
||||
Method | Description
|
||||
-----------|------------
|
||||
`at` | Returns the `PropertyValue` for a given `Property` (key).
|
||||
`set` | Stores a given `PropertyValue` under a given `Property` (key).
|
||||
`erase` | Deletes a given `Property` (key) alongside its corresponding `PropertyValue`.
|
||||
`clear` | Clears the storage.
|
||||
`iterator`| Provides an extension of `std::input_iterator` that iterates over storage.
|
||||
|
||||
## Storage location
|
||||
|
||||
By default, *Memgraph* is an in-memory database and all properties are therefore
|
||||
stored in working memory unless specified otherwise by the user. User has an
|
||||
option to specify via the command line which properties they wish to be stored
|
||||
on disk.
|
||||
|
||||
Storage location of each property is encapsulated within a `Property` object
|
||||
which is ensured by the `ConcurrentIdMapper`. More precisely, the unsigned 16-bit
|
||||
property key has the following format:
|
||||
|
||||
```
|
||||
|---location--|------id------|
|
||||
|-Memory|Disk-|-----2^15-----|
|
||||
```
|
||||
|
||||
In other words, the most significant bit determines the location where the
|
||||
property will be stored.
|
||||
|
||||
### In-memory storage
|
||||
|
||||
The underlying implementation of in-memory storage for the time being is
|
||||
`std::vector<std::pair<Property, PropertyValue>>`. Implementations of`at`, `set`
|
||||
and `erase` are linear in time. This implementation is arguably more efficient
|
||||
than `std::map` or `std::unordered_map` when the average number of properties of
|
||||
a record is relatively small (up to 10) which seems to be the case.
|
||||
|
||||
### On-disk storage
|
||||
|
||||
#### KVStore
|
||||
|
||||
Disk storage is modeled by an abstraction of key-value storage as implemented in
|
||||
`storage/kvstore.hpp'. An interface of this abstraction is as follows:
|
||||
|
||||
Method | Description
|
||||
----------------|------------
|
||||
`Put` | Stores the given value under the given key.
|
||||
`Get` | Obtains the given value stored under the given key.
|
||||
`Delete` | Deletes a given (key, value) pair from storage..
|
||||
`DeletePrefix` | Deletes all (key, value) pairs where key begins with a given prefix.
|
||||
`Size` | Returns the size of the storage or, optionally, the number of stored pairs that begin with a given prefix.
|
||||
`iterator` | Provides an extension of `std::input_iterator` that iterates over storage.
|
||||
|
||||
Keys and values in this context are of type `std::string`.
|
||||
|
||||
The actual underlying implementation of this abstraction uses
|
||||
[RocksDB]{https://rocksdb.org} — a persistent key-value store for fast
|
||||
storage.
|
||||
|
||||
It is worthy to note that the custom iterator implementation allows the user
|
||||
to iterate over a given prefix. Otherwise, the implementation follows familiar
|
||||
c++ constructs and can be used as follows:
|
||||
|
||||
```
|
||||
KVStore storage = ...;
|
||||
for (auto it = storage.begin(); it != storage.end(); ++it) {}
|
||||
for (auto kv : storage) {}
|
||||
for (auto it = storage.begin("prefix"); it != storage.end("prefix"); ++it) {}
|
||||
```
|
||||
|
||||
Note that it is not possible to scan over multiple prefixes. For instance, one
|
||||
might assume that you can scan over all keys that fall in a certain
|
||||
lexicographical range. Unfortunately, that is not the case and running the
|
||||
following code will result in an infinite loop with a touch of undefined
|
||||
behavior.
|
||||
|
||||
```
|
||||
KVStore storage = ...;
|
||||
for (auto it = storage.begin("alpha"); it != storage.end("omega"); ++it) {}
|
||||
```
|
||||
|
||||
#### Data organization on disk
|
||||
|
||||
Each `PropertyValueStore` instance can access a static `KVStore` object that can
|
||||
store `(key, value)` pairs on disk. The key of each property on disk consists of
|
||||
two parts — a unique identifier (unsigned 64-bit integer) of the current
|
||||
record version (see mvcc docummentation for further clarification) and a
|
||||
property key as described above. The actual value of the property is serialized
|
||||
into a bytestring using bolt `BaseEncoder`. Similarly, deserialization is
|
||||
performed by bolt `Decoder`.
|
||||
@@ -1,152 +0,0 @@
|
||||
# Bootstrapping Compilation Toolchain for Memgraph
|
||||
|
||||
Requirements:
|
||||
|
||||
* libstdc++ shipped with gcc-6.3 or gcc-6.4
|
||||
* cmake >= 3.1, Debian Stretch uses cmake-3.7.2
|
||||
* clang-3.9
|
||||
|
||||
## Installing gcc-6.4
|
||||
|
||||
gcc-6.3 has a bug, so use the 6.4 version which is just a bugfix release.
|
||||
|
||||
Requirements on CentOS 7:
|
||||
|
||||
* wget
|
||||
* make
|
||||
* gcc (bootstrap)
|
||||
* gcc-c++ (bootstrap)
|
||||
* gmp-devel (bootstrap)
|
||||
* mpfr-devel (bootstrap)
|
||||
* libmpc-devel (bootstrap)
|
||||
* zip
|
||||
* perl
|
||||
* dejagnu (testing)
|
||||
* expect (testing)
|
||||
* tcl (testing)
|
||||
|
||||
```
|
||||
wget ftp://ftp.mpi-sb.mpg.de/pub/gnu/mirror/gcc.gnu.org/pub/gcc/releases/gcc-6.4.0/gcc-6.4.0.tar.gz
|
||||
tar xf gcc-6.4.0.tar.gz
|
||||
cd gcc-6.4.0
|
||||
mkdir build
|
||||
cd build
|
||||
../configure --disable-multilib --prefix=<install-dst>
|
||||
make
|
||||
# Testing
|
||||
make -k check
|
||||
make install
|
||||
```
|
||||
|
||||
*Do not put gcc + libs on PATH* (unless you know what you are doing).
|
||||
|
||||
## Installing cmake-3.7.2
|
||||
|
||||
Requirements on CentOS 7:
|
||||
|
||||
* wget
|
||||
* make
|
||||
* gcc
|
||||
* gcc-c++
|
||||
* ncurses-devel (optional, for ccmake)
|
||||
|
||||
```
|
||||
wget https://cmake.org/files/v3.7/cmake-3.7.2.tar.gz
|
||||
tar xf cmake-3.7.2.tar.gz
|
||||
cd cmake-3.7.2.tar.gz
|
||||
./bootstrap --prefix<install-dst>
|
||||
make
|
||||
make install
|
||||
```
|
||||
|
||||
Put cmake on PATH (if appropriate)
|
||||
|
||||
**Fix the bug in CpackRPM**
|
||||
|
||||
`"<path-to-cmake>/share/cmake-3.7/Modules/CPackRPM.cmake" line 2273 of 2442`
|
||||
|
||||
The line
|
||||
|
||||
```
|
||||
set(RPMBUILD_FLAGS "-bb")
|
||||
```
|
||||
needs to be before
|
||||
|
||||
```
|
||||
if(CPACK_RPM_GENERATE_USER_BINARY_SPECFILE_TEMPLATE OR NOT CPACK_RPM_USER_BINARY_SPECFILE)
|
||||
```
|
||||
|
||||
It was probably accidentally placed after, and is fixed in later cmake
|
||||
releases.
|
||||
|
||||
## Installing clang-3.9
|
||||
|
||||
Requirements on CentOS 7:
|
||||
|
||||
* wget
|
||||
* make
|
||||
* cmake
|
||||
|
||||
```
|
||||
wget http://releases.llvm.org/3.9.1/llvm-3.9.1.src.tar.xz
|
||||
tar xf llvm-3.9.1.src.tar.xz
|
||||
mv llvm-3.9.1.src llvm
|
||||
|
||||
wget http://releases.llvm.org/3.9.1/cfe-3.9.1.src.tar.xz
|
||||
tar xf cfe-3.9.1.src.tar.xz
|
||||
mv cfe-3.9.1.src llvm/tools/clang
|
||||
|
||||
cd llvm
|
||||
mkdir build
|
||||
cd build
|
||||
cmake -DCMAKE_BUILD_TYPE="Release" -DGCC_INSTALL_PREFIX=<gcc-dir> \
|
||||
-DCMAKE_C_COMPILER=<gcc> -DCMAKE_CXX_COMPILER=<g++> \
|
||||
-DCMAKE_CXX_LINK_FLAGS="-L<gcc-dir>/lib64 -Wl,-rpath,<gcc-dir>/lib64" \
|
||||
-DCMAKE_INSTALL_PREFIX=<install-dst> ..
|
||||
make
|
||||
# Testing
|
||||
make check-clang
|
||||
make install
|
||||
```
|
||||
|
||||
Put clang on PATH (if appropriate)
|
||||
|
||||
## Memgraph
|
||||
|
||||
Requirements on CentOS 7:
|
||||
|
||||
* libuuid-devel (antlr4)
|
||||
* java-1.8.0-openjdk (antlr4)
|
||||
* boost-static (too low version --- compile manually)
|
||||
* rpm-build (RPM)
|
||||
* python3 (tests, ...)
|
||||
* which (required for rocksdb)
|
||||
* sbcl (lisp C++ preprocessing)
|
||||
|
||||
### Boost 1.62
|
||||
|
||||
```
|
||||
wget https://netix.dl.sourceforge.net/project/boost/boost/1.62.0/boost_1_62_0.tar.gz
|
||||
tar xf boost_1_62_0.tar.gz
|
||||
cd boost_1_62_0
|
||||
./bootstrap.sh --with-toolset=clang --with-libraries=iostreams,serialization --prefix=<install-dst>
|
||||
./b2
|
||||
# Default installs to /usr/local/
|
||||
./b2 install
|
||||
```
|
||||
|
||||
### Building Memgraph
|
||||
|
||||
clang is *required* to be findable by cmake, i.e. it should be on PATH.
|
||||
cmake isn't required to be on the path, since you run it manually, so can use
|
||||
the full path to executable in order to run it. Obviously, it is convenient to
|
||||
put cmake also on PATH.
|
||||
|
||||
Building is done as explained in [Quick Start](quick-start.md), but each
|
||||
`make` invocation needs to be prepended with:
|
||||
|
||||
`LD_RUN_PATH=<gcc-dir>/lib64 make ...`
|
||||
|
||||
### RPM
|
||||
|
||||
Name format: `memgraph-<version>-<pkg-version>.<arch>.rpm`
|
||||
@@ -1,177 +0,0 @@
|
||||
# Memgraph Workflow
|
||||
|
||||
This chapter describes the usual workflow for working on Memgraph.
|
||||
|
||||
## Git
|
||||
|
||||
Memgraph uses [git](https://git-scm.com/) for source version control. If you
|
||||
obtained the source, you probably already have it installed. Before you can
|
||||
track new changes, you need to setup some basic information.
|
||||
|
||||
First, tell git your name:
|
||||
|
||||
git config --global user.name "FirstName LastName"
|
||||
|
||||
Then, set your Memgraph email:
|
||||
|
||||
git config --global user.email "my.email@memgraph.com"
|
||||
|
||||
Finally, make git aware of your favourite editor:
|
||||
|
||||
git config --global core.editor "vim"
|
||||
|
||||
## Phabricator
|
||||
|
||||
All of the code in Memgraph needs to go through code review before it can be
|
||||
accepted in the codebase. This is done through
|
||||
[Phabricator](https://phacility.com/phabricator/). The command line tool for
|
||||
interfacing with Phabricator is
|
||||
[arcanist](https://phacility.com/phabricator/arcanist/). You should already
|
||||
have it installed if you followed the steps in [Quick Start](quick-start.md).
|
||||
|
||||
The only required setup is to go in the root of Memgraph's project and run:
|
||||
|
||||
arc install-certificate
|
||||
|
||||
## Working on Your Feature Branch
|
||||
|
||||
Git has a concept of source code *branches*. The `master` branch contains all
|
||||
of the changes which were reviewed and accepted in Memgraph's code base. The
|
||||
`master` branch is selected by default.
|
||||
|
||||
### Creating a Branch
|
||||
|
||||
When working on a new feature or fixing a bug, you should create a new branch
|
||||
out of the `master` branch. For example, let's say you are adding static type
|
||||
checking to the query language compiler. You would create a branch called
|
||||
`mg_query_static_typing` with the following command:
|
||||
|
||||
git branch mg_query_static_typing
|
||||
|
||||
To switch to that branch, type:
|
||||
|
||||
git checkout mg_query_static_typing
|
||||
|
||||
Since doing these two steps will happen often, you can use a shortcut command:
|
||||
|
||||
git checkout -b mg_query_static_typing
|
||||
|
||||
Note that a branch is created from the currently selected branch. So, if you
|
||||
wish to create another branch from `master` you need to switch to `master`
|
||||
first.
|
||||
|
||||
The usual convention for naming your branches is `mg_<feature_name>`, you may
|
||||
switch underscores ('\_') for hyphens ('-').
|
||||
|
||||
Do take care not to mix the case of your branch names! Certain operating
|
||||
systems (like Windows) don't distinguish the casing in git branches. This may
|
||||
cause hard to track down issues when trying to switch branches. Therefore, you
|
||||
should always name your branches with lowercase letters.
|
||||
|
||||
### Making and Committing Changes
|
||||
|
||||
When you have a branch for your new addition, you can now actually start
|
||||
implementing it. After some amount of time, you may have created new files,
|
||||
modified others and maybe even deleted unused files. You need to tell git to
|
||||
track those changes. This is accomplished with `git add` and `git rm`
|
||||
commands.
|
||||
|
||||
git add path-to-new-file path-to-modified-file
|
||||
git rm path-to-deleted-file
|
||||
|
||||
To check that everything is correctly tracked, you may use the `git status`
|
||||
command. It will also print the name of the currently selected branch.
|
||||
|
||||
If everything seems OK, you should commit these changes to git.
|
||||
|
||||
git commit
|
||||
|
||||
You will be presented with an editor where you need to type the commit
|
||||
message. Writing a good commit message is an art in itself. You should take a
|
||||
look at the links below. We try to follow these conventions as much as
|
||||
possible.
|
||||
|
||||
* [How to Write a Git Commit Message](http://chris.beams.io/posts/git-commit/)
|
||||
* [A Note About Git Commit Messages](http://tbaggery.com/2008/04/19/a-note-about-git-commit-messages.html)
|
||||
* [stopwritingramblingcommitmessages](http://stopwritingramblingcommitmessages.com/)
|
||||
|
||||
### Sending Changes on a Review
|
||||
|
||||
After finishing your work on your feature branch, you will want to send it on
|
||||
code review. This is done through Arcanist. To do that, run the following
|
||||
command:
|
||||
|
||||
arc diff
|
||||
|
||||
You will, once again, be presented with an editor where you need to describe
|
||||
your whole work. `arc` will by default fill that description with your commit
|
||||
messages. The title and summary of your work should also follow the
|
||||
conventions of git messages as described above. If you followed the
|
||||
guidelines, the message filled by `arc` should be fine.
|
||||
|
||||
In addition to the message, you need to fill the `Reviewers:` line with
|
||||
usernames of people who should do the code review.
|
||||
|
||||
You changes will be visible on Phabricator as a so called "diff". You can find
|
||||
the default view of active diffs
|
||||
[here](https://phabricator.memgraph.io/differential/)
|
||||
|
||||
### Updating Changes Based on Review
|
||||
|
||||
When you get comments in the code review, you will want to make additional
|
||||
modifications to your work. The same workflow as before applies: [Making and
|
||||
Committing Changes](#making-and-committing-changes)
|
||||
|
||||
After making those changes, send them back on code review:
|
||||
|
||||
arc diff
|
||||
|
||||
|
||||
### Updating From New Master
|
||||
|
||||
Let's say that, while you were working, someone else added some new features
|
||||
to the codebase that you would like to use in your current work. To obtain
|
||||
those changes you should update your `master` branch:
|
||||
|
||||
git checkout master
|
||||
git pull origin master
|
||||
|
||||
Now, these changes are on `master`, but you want them in your local branch. To
|
||||
do that, use `git rebase`:
|
||||
|
||||
git checkout mg_query_static_typing
|
||||
git rebase master
|
||||
|
||||
During `git rebase`, you may get reports that some files have conflicting
|
||||
changes. If you need help resolving them, don't be afraid to ask around! After
|
||||
you've resolved them, mark them as done with `git add` command. You may
|
||||
then continue with `git rebase --continue`.
|
||||
|
||||
After the `git rebase` is done, you will now have new changes from `master` on
|
||||
your feature branch as if you just created and started working on that branch.
|
||||
You may continue with the usual workflow of [Making and Committing
|
||||
Changes](#making-and-committing-changes) and [Sending Changes on a
|
||||
Review](#sending-changes-on-a-review).
|
||||
|
||||
### Sending Your Changes on Master Branch
|
||||
|
||||
When your changes pass the code review, you are ready to integrate them in the
|
||||
`master` branch. To do that, run the following command:
|
||||
|
||||
arc land
|
||||
|
||||
Arcanist will take care of obtaining the latest changes from `master` and
|
||||
merging your changes on top. If the `land` was successful, Arcanist will
|
||||
delete your local branch and you will be back on `master`. Continuing from the
|
||||
examples above, the deleted branch would be `mg_query_static_typing`.
|
||||
|
||||
This marks the completion of your changes, and you are ready to work on
|
||||
something else.
|
||||
|
||||
### Note For People Familiar With Git
|
||||
|
||||
Since Arcanist takes care of merging your git commits and pushing them on
|
||||
`master`, you should *never* have to call `git merge` and `git push`. If you
|
||||
find yourself typing those commands, check that you are doing the right thing.
|
||||
The most common mistake is to use `git merge` instead of `git rebase` for the
|
||||
case described in [Updating From New Master](#updating-from-new-master).
|
||||
@@ -1,7 +1,6 @@
|
||||
# Memgraph Code Documentation
|
||||
|
||||
IMPORTANT: auto-generated (run doxygen Doxyfile in the project root)
|
||||
IMPORTANT: Auto-generated (run doxygen Doxyfile in the project root).
|
||||
|
||||
* HTML - just open docs/doxygen/html/index.html
|
||||
|
||||
* Latex - run make inside docs/doxygen/latex
|
||||
* HTML - Open docs/doxygen/html/index.html.
|
||||
* Latex - Run make inside docs/doxygen/latex.
|
||||
|
||||
|
Before Width: | Height: | Size: 6.6 KiB After Width: | Height: | Size: 6.6 KiB |
@@ -1,20 +0,0 @@
|
||||
## Dynamic Graph Partitioner
|
||||
|
||||
Memgraph supports dynamic graph partitioning which dynamically improves
|
||||
performance on badly partitioned dataset over workers. To enable it, the user
|
||||
should use the following flag when firing up the *master* node:
|
||||
|
||||
```plaintext
|
||||
--dynamic_graph_partitioner_enable
|
||||
```
|
||||
|
||||
### Parameters
|
||||
|
||||
| Name | Default Value | Description | Range |
|
||||
|------|---------------|-------------|-------|
|
||||
|--dgp_improvement_threshold | 10 | How much better should specific node score
|
||||
be to consider a migration to another worker. This represents the minimal
|
||||
difference between new score that the vertex will have when migrated and the
|
||||
old one such that it's migrated. | Min: 1, Max: 100
|
||||
|--dgp_max_batch_size | 2000 | Maximal amount of vertices which should be
|
||||
migrated in one dynamic graph partitioner step. | Min: 1, Max: MaxInt32 |
|
||||
@@ -1,78 +0,0 @@
|
||||
# Distributed Memgraph specs
|
||||
This document describes reasnonings behind Memgraphs distributed concepts.
|
||||
|
||||
## Distributed state machine
|
||||
Memgraphs distributed mode introduces two states of the cluster, recovering and
|
||||
working. The change between states shouldn't happen often, but when it happens
|
||||
it can take a while to make a transition from one to another.
|
||||
|
||||
### Recovering
|
||||
This state is the default state for Memgraph when the cluster starts with
|
||||
recovery flags. If the recovery finishes successfully, the state changes to
|
||||
working. If recovery fails, the user will be presented with a message that
|
||||
explains what happened and what are the next steps.
|
||||
|
||||
Another way to enter this state is failure. If the cluster encounters a failure,
|
||||
the master will enter the Recovering mode. This time, it will wait for all
|
||||
workers to respond with a message saying they are alive and well, and making
|
||||
sure they all have consistent state.
|
||||
|
||||
### Working
|
||||
This state should be the default state of Memgraph most of the time. When in
|
||||
this state, Memgraph accepts connections from Bolt clients and allows query
|
||||
execution.
|
||||
|
||||
If distributed execution fails for a transaction, that transaction, and all
|
||||
other active transactions will be aborted and the cluster will enter the
|
||||
Recovering state.
|
||||
|
||||
## Durability
|
||||
One of the important concepts in distributed Memgraph is durability.
|
||||
|
||||
### Cluster configuration
|
||||
When running Memgraph in distributed mode, the master will store cluster
|
||||
metadata in a persistent store. If fore some reason the cluster shuts down,
|
||||
recovering Memgraph from durability files shouldn't require any additional
|
||||
flags.
|
||||
|
||||
### Database ID
|
||||
Each new and clean run of Memgraph should generate a new globally unique
|
||||
database id. This id will associate all files that have persisted with this
|
||||
run. Adding the database id to snapshots, write-ahead logs and cluster metadata
|
||||
files ties them a specific Memgraph run, and it makes recovery easier to reason
|
||||
about.
|
||||
|
||||
When recovering, the cluster won't generate a new id, but will reuse the one
|
||||
from the snapshot/wal that it was able to recover from.
|
||||
|
||||
### Durability files
|
||||
Memgraph uses snapshots and write-ahead logs for durability.
|
||||
|
||||
When Memgraph recovers it has to make sure all machines in the cluster recover
|
||||
to the same recovery point. This is done by finding a common snapshot and
|
||||
finding common transactions in per-machine available write-ahead logs.
|
||||
|
||||
Since we can not be sure that each machine persisted durability files, we need
|
||||
to be able to negotiate a common recovery point in the cluster. Possible
|
||||
durability file failures could require to start the cluster from scratch,
|
||||
purging everything from storage and recovering from existing durability files.
|
||||
|
||||
We need to ensure that we keep wal files containing information about
|
||||
transactions between all existing snapshots. This will provide better durability
|
||||
in the case of a random machine durability file failure, where the cluster can
|
||||
find a common recovery point that all machines in the cluster have.
|
||||
|
||||
Also, we should suggest and make clear docs that anything less than two
|
||||
snapshots isn't considered safe for recovery.
|
||||
|
||||
### Recovery
|
||||
The recovery happens in following steps:
|
||||
* Master enables worker registration.
|
||||
* Master recovers cluster metadata from the persisted storage.
|
||||
* Master waits all required workers to register.
|
||||
* Master broadcasts a recovery request to all workers.
|
||||
* Workers respond with with a set of possible recovery points.
|
||||
* Master finds a common recovery point for the whole cluster.
|
||||
* Master broadcasts a recovery request with the common recovery point.
|
||||
* Master waits for the cluster to recover.
|
||||
* After a successful cluster recovery, master can enter Working state.
|
||||
@@ -1,75 +0,0 @@
|
||||
# Dynamic Graph Partitioning (abbr. DGP)
|
||||
|
||||
## Implementation
|
||||
|
||||
Take a look under `dev/memgraph/distributed/dynamic_graph_partitioning.md`.
|
||||
|
||||
### Implemented parameters
|
||||
|
||||
--dynamic-graph-partitioner-enabled (If the dynamic graph partitioner should be
|
||||
enabled.) type: bool default: false (start time)
|
||||
--dgp-improvement-threshold (How much better should specific node score be
|
||||
to consider a migration to another worker. This represents the minimal
|
||||
difference between new score that the vertex will have when migrated
|
||||
and the old one such that it's migrated.) type: int32 default: 10
|
||||
(start time)
|
||||
--dgp-max-batch-size (Maximal amount of vertices which should be migrated
|
||||
in one dynamic graph partitioner step.) type: int32 default: 2000
|
||||
(start time)
|
||||
|
||||
## Planning
|
||||
|
||||
### Design decisions
|
||||
|
||||
* Each partitioning session has to be a new transaction.
|
||||
* When and how does an instance perform the moves?
|
||||
* Periodically.
|
||||
* Token sharing (round robin, exactly one instance at a time has an
|
||||
opportunity to perform the moves).
|
||||
* On server-side serialization error (when DGP receives an error).
|
||||
-> Quit partitioning and wait for the next turn.
|
||||
* On client-side serialization error (when end client receives an error).
|
||||
-> The client should never receive an error because of any
|
||||
internal operation.
|
||||
-> For the first implementation, it's good enough to wait until data becomes
|
||||
available again.
|
||||
-> It would be nice to achieve that DGP has lower priority than end client
|
||||
operations.
|
||||
|
||||
### End-user parameters
|
||||
|
||||
* --dynamic-graph-partitioner-enabled (execution time)
|
||||
* --dgp-improvement-threshold (execution time)
|
||||
* --dgp-max-batch-size (execution time)
|
||||
* --dgp-min-batch-size (execution time)
|
||||
-> Minimum number of nodes that will be moved in each step.
|
||||
* --dgp-fitness-threshold (execution time)
|
||||
-> Do not perform moves if partitioning is good enough.
|
||||
* --dgp-delta-turn-time (execution time)
|
||||
-> Time between each turn.
|
||||
* --dgp-delta-step-time (execution time)
|
||||
-> Time between each step.
|
||||
* --dgp-step-time (execution time)
|
||||
-> Time limit per each step.
|
||||
|
||||
### Testing
|
||||
|
||||
The implementation has to provide good enough results in terms of:
|
||||
* How good the partitioning is (numeric value), aka goodness.
|
||||
* Workload execution time.
|
||||
* Stress test correctness.
|
||||
|
||||
Test cases:
|
||||
* N not connected subgraphs
|
||||
-> shuffle nodes to N instances
|
||||
-> run partitioning
|
||||
-> test perfect partitioning.
|
||||
* N connected subgraph
|
||||
-> shuffle nodes to N instance
|
||||
-> run partitioning
|
||||
-> test partitioning.
|
||||
* Take realistic workload (Long Running, LDBC1, LDBC2, Card Fraud, BFS, WSP)
|
||||
-> measure exec time
|
||||
-> run partitioning
|
||||
-> test partitioning
|
||||
-> measure exec time (during and after partitioning).
|
||||
@@ -1,214 +0,0 @@
|
||||
# High Availability (abbr. HA)
|
||||
|
||||
## Introduction
|
||||
|
||||
High availability is a characteristic of a system which aims to ensure a
|
||||
certain level of operational performance for a higher-than-normal period.
|
||||
Although there are multiple ways to design highly available systems, Memgraph
|
||||
strives to achieve HA by elimination of single points of failure. In essence,
|
||||
this implies adding redundancy to the system so that a failure of a component
|
||||
does not imply the failure of the entire system.
|
||||
|
||||
## Theoretical Background
|
||||
|
||||
The following chapter serves as an introduction into some theoretical aspects
|
||||
of Memgraph's high availability implementation. If the reader is solely
|
||||
interested in design decisions around HA implementation, they can skip this
|
||||
chapter.
|
||||
|
||||
An important implication of any HA implementation stems from Eric Brewer's
|
||||
[CAP Theorem](https://fenix.tecnico.ulisboa.pt/downloadFile/1126518382178117/10.e-CAP-3.pdf)
|
||||
which states that it is impossible for a distributed system to simultaneously
|
||||
achieve:
|
||||
|
||||
* Consistency (C - every read receives the most recent write or an error)
|
||||
* Availability (A - every request receives a response that is not an error)
|
||||
* Partition tolerance (P - The system continues to operate despite an
|
||||
arbitrary number of messages being dropped by the
|
||||
network between nodes)
|
||||
|
||||
In the context of HA, Memgraph should strive to achieve CA.
|
||||
|
||||
### Consensus
|
||||
|
||||
Implications of the CAP theorem naturally lead us towards introducing a
|
||||
cluster of machines which will have identical internal states. When a designated
|
||||
machine for handling client requests fails, it can simply be replaced with
|
||||
another.
|
||||
|
||||
Well... turns out this is not as easy as it sounds :(
|
||||
|
||||
Keeping around a cluster of machines with consistent internal state is an
|
||||
inherently difficult problem. More precisely, this problem is as hard as
|
||||
getting a cluster of machines to agree on a single value, which is a highly
|
||||
researched area in distributed systems. Our research of state of the art
|
||||
consensus algorithms lead us to Diego Ongaro's
|
||||
[Raft algorithm](https://raft.github.io/raft.pdf).
|
||||
|
||||
#### Raft
|
||||
|
||||
As you might have guessed, analyzing each subtle detail of Raft goes way
|
||||
beyond the scope of this document. In the remainder of the chapter we will
|
||||
outline only the most important ideas and implications, leaving all further
|
||||
analysis to the reader. Detailed explanation can be found either in Diego's
|
||||
[dissertation](https://ramcloud.stanford.edu/~ongaro/thesis.pdf) \[1\] or the
|
||||
Raft [paper](https://raft.github.io/raft.pdf) \[2\].
|
||||
|
||||
In essence, Raft allows us to implement the previously mentioned idea of
|
||||
managing a cluster of machines with identical internal states. In other
|
||||
words, the Raft protocol allows us to manage a cluster of replicated
|
||||
state machines which is fully functional as long as the *majority* of
|
||||
the machines in the cluster operate correctly.
|
||||
|
||||
Another important fact is that those state machines must be *deterministic*.
|
||||
In other words, the same command on two different machines with the same
|
||||
internal state must yield the same result. This is important because Memgraph,
|
||||
as a black box, is not entirely deterministic. Non-determinism can easily be
|
||||
introduced by the user (e.g. by using the `rand` function) or by algorithms
|
||||
behind query execution (e.g. introducing fuzzy logic in the planner could yield
|
||||
a different order of results). Luckily, once we enter the storage level,
|
||||
everything should be fully deterministic.
|
||||
|
||||
To summarize, Raft is a protocol which achieves consensus in a cluster of
|
||||
deterministic state machines via log replication. The cluster is fully
|
||||
functional if the majority of the machines work correctly. The reader
|
||||
is strongly encouraged to gain a deeper understanding (at least read through
|
||||
the paper) of Raft before reading the rest of this document.
|
||||
|
||||
## Integration with Memgraph
|
||||
|
||||
The first thing that should be defined is a single instruction within the
|
||||
context of Raft (i.e. a single entry in a replicated log). As mentioned
|
||||
before, these instructions should be completely deterministic when applied
|
||||
to the state machine. We have therefore decided that the appropriate level
|
||||
of abstraction within Memgraph corresponds to `StateDelta`-s (data structures
|
||||
which describe a single change to the Memgraph state, used for durability
|
||||
in WAL). Moreover, a single instruction in a replicated log will consist of a
|
||||
batch of `StateDelta`s which correspond to a single **committed** transaction.
|
||||
This decision both improves performance and handles some special cases that
|
||||
present themselves otherwise by leveraging the knowledge that the transaction
|
||||
should be committed.
|
||||
|
||||
"What happens with aborted transactions?"
|
||||
|
||||
A great question, they are handled solely by the leader which is the only
|
||||
machine that communicates with the client. Aborted transactions do not alter
|
||||
the state of the database and there is no need to replicate it to other machines
|
||||
in the cluster. If, for instance, the leader dies before returning the result
|
||||
of some read operation in an aborted transaction, the client will notice that
|
||||
the leader has crashed. A new leader will be elected in the next term and the
|
||||
client should retry the transaction.
|
||||
|
||||
"OK, that makes sense! But, wait a minute, this is broken by design! Merely
|
||||
generating `StateDelta`s on the leader for any transaction will taint its
|
||||
internal storage before sending the first RPC to some follower. This deviates
|
||||
from Raft and will crash the universe!"
|
||||
|
||||
Another great observation. It is indeed true that applying `StateDelta`s makes
|
||||
changes to local storage, but only a single type of `StateDelta` makes that
|
||||
change durable. That `StateDelta` type is called `TRANSACTION_COMMIT` and we
|
||||
will change its behaviour when working as a HA instance. More precisely, we
|
||||
must not allow the transaction engine to modify the commit log saying that
|
||||
the transaction has been committed. That action should be delayed until those
|
||||
`StateDelta`s have been applied to the majority of the cluster. At that point
|
||||
the commit log can be safely modified leaving it up to Raft to ensure the
|
||||
durability of the transaction.
|
||||
|
||||
We should also address one subtle detail that arises in this case. Consider
|
||||
the following scenario:
|
||||
|
||||
* The leader starts working on a transaction which creates a new record in the
|
||||
database. Suppose that record is stored in the leader's internal storage
|
||||
but the transaction was not committed (i.e. no such entry in the commit log).
|
||||
* The leader should start replicating those `StateDelta`s to its followers
|
||||
but, suddenly, it's cut off from the rest of the cluster.
|
||||
* Due to timeout, a new election is held and a new leader has been elected.
|
||||
* Our old leader comes back to life and becomes a follower.
|
||||
* The new leader receives a transaction which creates that same record, but
|
||||
this transaction is successfully replicated and committed by the new leader.
|
||||
|
||||
The problem lies in the fact that there is still a record within the internal
|
||||
storage of our old leader with the same transaction ID and GID as the recently
|
||||
committed record by the new leader. Obviously, this is broken. As a solution, on
|
||||
each transition from `Leader` to `Follower`, we will reinitialize storage, reset
|
||||
the transaction engine and recover data from the Raft log. This will ensure all
|
||||
ongoing transactions which have "polluted" the storage will be gone.
|
||||
|
||||
"When will followers append that transaction to their commit logs?"
|
||||
|
||||
When the leader deduces that the transaction is safe to commit, it will include
|
||||
the relevant information in all further heartbeats which will alert the
|
||||
followers that it is safe to commit those entries from their raft logs.
|
||||
Naturally, the followers need not to delay appending data to the commit log
|
||||
as they know that the transaction has already been committed (from the clusters
|
||||
point of view). If this sounds really messed up, seriously, read the Raft paper.
|
||||
|
||||
"How does the raft log differ from WAL"
|
||||
|
||||
Conceptually, it doesn't. When operating in HA, we don't really need the
|
||||
recovery mechanisms implemented in Memgraph thus far. When a dead machine
|
||||
comes back to life, it will eventually come in sync with the rest of the
|
||||
cluster and everything will be done using the machine's raft log as well
|
||||
as the messages received from the cluster leader.
|
||||
|
||||
"Those logs will become huge, isn't that recovery going to be painfully slow?"
|
||||
|
||||
True, but there are mechanisms for making raft logs more compact. The most
|
||||
popular method is, wait for it, making snapshots :)
|
||||
Although the process of bringing an old machine back to life is a long one,
|
||||
it doesn't really affect the performance of the cluster in a great degree.
|
||||
The cluster will work perfectly fine with that machine being way out of sync.
|
||||
|
||||
"I don't know, everything seems to be a lot slower than before!"
|
||||
|
||||
Absolutely true, the user should be aware that they will suffer dire
|
||||
consequences on the performance side if they choose to be highly available.
|
||||
As Frankie says, "That's life!".
|
||||
|
||||
"Also, I didn't really care about most of the things you've said. I'm
|
||||
not a part of the storage team and couldn't care less about the issues you
|
||||
face, how does HA affect 'my part of the codebase'?"
|
||||
|
||||
Answer for query execution: That's ok, you'll be able to use the same beloved
|
||||
API (when we implement it, he he :) towards storage and continue to
|
||||
make fun of us when you find a bug.
|
||||
|
||||
Answer for infrastructure: We'll talk. Some changes will surely need to
|
||||
be made on the Memgraph client. There is a chapter in Diego's dissertation
|
||||
called 'Client interaction', but we'll cross that bridge when we get there.
|
||||
There will also be the whole 'integration with Jepsen tests' thing going on.
|
||||
|
||||
Answer for analytics: I'm astonished you've read this article. Wanna join
|
||||
storage?
|
||||
|
||||
### Subtlety Regarding Reads
|
||||
|
||||
As we have hinted in the previous chapter, we would like to bypass log
|
||||
replication for operations which do not alter the internal state of Memgraph.
|
||||
Those operations should therefore be handled only by the leader, which is not
|
||||
as trivial as it seems. The subtlety arises from the fact that a (newly-elected)
|
||||
leader can have an entry in its log which was committed by the previous leader
|
||||
that has crashed but that entry is not yet committed in its internal storage
|
||||
by the current leader. Moreover, the rule about safely committing logs that are
|
||||
replicated on the majority of the cluster only applies for entries replicated in
|
||||
the leaders current term. Therefore, we are faced with two issues:
|
||||
|
||||
* We cannot simply perform read operations if the leader has a non-committed
|
||||
entry in its log (breaks consistency).
|
||||
* Replicating those entries onto the majority of the cluster is not enough
|
||||
to guarantee that they can be safely committed.
|
||||
|
||||
This can be solved by introducing a blank no-op operation which the new leader
|
||||
will try to replicate at the start of its term. Once that operation is
|
||||
replicated and committed, the leader can safely perform those non-altering
|
||||
operations on its own.
|
||||
|
||||
For further information about these issues, you should check out section
|
||||
5.4.2 from the raft paper \[1\] which hints as to why its not safe to commit
|
||||
entries from previous terms. Also, you should check out section 6.4 from
|
||||
the thesis \[2\] which goes into more details around efficiently processing
|
||||
read-only queries.
|
||||
|
||||
## How do we test HA
|
||||
|
||||
[Check this out](https://jepsen.io/analyses/dgraph-1-0-2)
|
||||
@@ -1,80 +0,0 @@
|
||||
# Kafka - openCypher clause
|
||||
|
||||
One must be able to specify the following when importing data from Kafka:
|
||||
|
||||
* Kafka URI
|
||||
* Kafka topic
|
||||
* Transform [script](transform.md) URI
|
||||
|
||||
|
||||
Minimum required syntax looks like:
|
||||
```opencypher
|
||||
CREATE STREAM stream_name AS LOAD DATA KAFKA 'URI'
|
||||
WITH TOPIC 'topic'
|
||||
WITH TRANSFORM 'URI';
|
||||
```
|
||||
|
||||
|
||||
The full openCypher clause for creating a stream is:
|
||||
```opencypher
|
||||
CREATE STREAM stream_name AS
|
||||
LOAD DATA KAFKA 'URI'
|
||||
WITH TOPIC 'topic'
|
||||
WITH TRANSFORM 'URI'
|
||||
[BATCH_INTERVAL milliseconds]
|
||||
[BATCH_SIZE count]
|
||||
```
|
||||
The `CREATE STREAM` clause happens in a transaction.
|
||||
|
||||
`WITH TOPIC` parameter specifies the Kafka topic from which we'll stream
|
||||
data.
|
||||
|
||||
`WITH TRANSFORM` parameter should contain a URI of the transform script.
|
||||
|
||||
`BATCH_INTERVAL` parameter defines the time interval in milliseconds
|
||||
which is the time between two successive stream importing operations.
|
||||
|
||||
`BATCH_SIZE` parameter defines the count of Kafka messages that will be
|
||||
batched together before import.
|
||||
|
||||
If both `BATCH_INTERVAL` and `BATCH_SIZE` parameters are given, the condition
|
||||
that is satisfied first will trigger the batched import.
|
||||
|
||||
Default value for `BATCH_INTERVAL` is 100 milliseconds, and the default value
|
||||
for `BATCH_SIZE` is 10;
|
||||
|
||||
The `DROP` clause deletes a stream:
|
||||
```opencypher
|
||||
DROP STREAM stream_name;
|
||||
```
|
||||
|
||||
The `SHOW` clause enables you to see all configured streams:
|
||||
```opencypher
|
||||
SHOW STREAMS;
|
||||
```
|
||||
|
||||
You can also start/stop streams with the `START` and `STOP` clauses:
|
||||
```opencypher
|
||||
START STREAM stream_name [LIMIT count BATCHES];
|
||||
STOP STREAM stream_name;
|
||||
```
|
||||
A stream needs to be stopped in order to start it and it needs to be started in
|
||||
order to stop it. Starting a started or stopping a stopped stream will not
|
||||
affect that stream.
|
||||
|
||||
There are also convenience clauses to start and stop all streams:
|
||||
```opencypher
|
||||
START ALL STREAMS;
|
||||
STOP ALL STREAMS;
|
||||
```
|
||||
|
||||
Before the actual import, you can also test the stream with the `TEST
|
||||
STREAM` clause:
|
||||
```opencypher
|
||||
TEST STREAM stream_name [LIMIT count BATCHES];
|
||||
```
|
||||
When a stream is tested, data extraction and transformation occurs, but no
|
||||
output is inserted in the graph.
|
||||
|
||||
A stream needs to be stopped in order to test it. When the batch limit is
|
||||
omitted, `TEST STREAM` will run for only one batch by default.
|
||||
@@ -1,34 +0,0 @@
|
||||
# Kafka - data transform
|
||||
|
||||
The transform script is a user defined script written in Python. The script
|
||||
should be aware of the data format in the Kafka message.
|
||||
|
||||
Each Kafka message is byte length encoded, which means that the first eight
|
||||
bytes of each message contain the length of the message.
|
||||
|
||||
A sample code for a streaming transform script could look like this:
|
||||
|
||||
```python
|
||||
def create_vertex(vertex_id):
|
||||
return ("CREATE (:Node {id: $id})", {"id": vertex_id})
|
||||
|
||||
|
||||
def create_edge(from_id, to_id):
|
||||
return ("MATCH (n:Node {id: $from_id}), (m:Node {id: $to_id}) "\
|
||||
"CREATE (n)-[:Edge]->(m)", {"from_id": from_id, "to_id": to_id})
|
||||
|
||||
|
||||
def stream(batch):
|
||||
result = []
|
||||
for item in batch:
|
||||
message = item.decode('utf-8').strip().split()
|
||||
if len(message) == 1:
|
||||
result.append(create_vertex(message[0])))
|
||||
else:
|
||||
result.append(create_edge(message[0], message[1]))
|
||||
return result
|
||||
|
||||
```
|
||||
|
||||
The script should output openCypher query strings based on the type of the
|
||||
records.
|
||||
@@ -1,61 +0,0 @@
|
||||
# Tensorflow Op - Technicalities
|
||||
|
||||
The final result should be a shared object (".so") file that can be
|
||||
dynamically loaded by the Tensorflow runtime in order to directly
|
||||
access the bolt client.
|
||||
|
||||
## About Tensorflow
|
||||
|
||||
Tensorflow is usually used with Python such that the Python code is used
|
||||
to define a directed acyclic computation graph. Basically no computation
|
||||
is done in Python. Instead, values from Python are copied into the graph
|
||||
structure as constants to be used by other Ops. The directed acyclic graph
|
||||
naturally ends up with two sets of border nodes, one for inputs, one for
|
||||
outputs. These are sometimes called "feeds".
|
||||
|
||||
Following the Python definition of the graph, during training, the entire
|
||||
data processing graph/pipeline is called from Python as a single expression.
|
||||
This leads to lazy evaluation since the called result has already been
|
||||
defined for a while.
|
||||
|
||||
Tensorflow internally works with tensors, i.e. n-dimensional arrays. That
|
||||
means all of its inputs need to be matrices as well as its outputs. While
|
||||
it is possible to feed data directly from Python's numpy matrices straight
|
||||
into Tensorflow, this is less desirable than using the Tensorflow data API
|
||||
(which defines data input and processing as a Tensorflow graph) because:
|
||||
|
||||
1. The data API is written in C++ and entirely avoids Python and as such
|
||||
is faster
|
||||
2. The data API, unlike Python is available in "Tensorflow serving". The
|
||||
default way to serve Tensorflow models in production.
|
||||
|
||||
Once the entire input pipeline is defined via the tf.data API, its input
|
||||
is basically a list of node IDs the model is supposed to work with. The
|
||||
model, through the data API knows how to connect to Memgraph and execute
|
||||
openCypher queries in order to get the remaining data it needs.
|
||||
(For example features of neighbouring nodes.)
|
||||
|
||||
## The Interface
|
||||
|
||||
I think it's best you read the official guide...
|
||||
<https://www.tensorflow.org/extend/adding_an_op>
|
||||
And especially the addition that specifies how data ops are special
|
||||
<https://www.tensorflow.org/extend/new_data_formats>
|
||||
|
||||
## Compiling the TF Op
|
||||
|
||||
There are two options for compiling a custom op.
|
||||
One of them involves pulling the TF source, adding your code to it and
|
||||
compiling via bazel.
|
||||
This is probably awkward to do for us and would
|
||||
significantly slow down compilation.
|
||||
|
||||
The other method involves installing Tensorflow as a Python package and
|
||||
pulling the required headers from for example:
|
||||
`/usr/local/lib/python3.6/site-packages/tensorflow/include`
|
||||
We can then compile our Op with our regular build system.
|
||||
|
||||
This is practical since we can copy the required headers to our repo.
|
||||
If necessary, we can have several versions of the headers to build several
|
||||
versions of our Op for every TF version which we want to support.
|
||||
(But this is unlikely to be required as the API should be stable).
|
||||
@@ -1,142 +0,0 @@
|
||||
# Example for Using the Bolt Client Tensorflow Op
|
||||
|
||||
## Dynamic Loading
|
||||
|
||||
``` python3
|
||||
import tensorflow as tf
|
||||
|
||||
mg_ops = tf.load_op_library('/usr/bin/memgraph/tensorflow_ops.so')
|
||||
```
|
||||
|
||||
## Basic Usage
|
||||
|
||||
``` python3
|
||||
dataset = mg_ops.OpenCypherDataset(
|
||||
# This is probably unfortunate as the username and password
|
||||
# get hardcoded into the graph, but for the simple case it's fine
|
||||
"hostname:7687", auth=("user", "pass"),
|
||||
|
||||
# Our query
|
||||
'''
|
||||
MATCH (n:Train) RETURN n.id, n.features
|
||||
''',
|
||||
|
||||
# Cast return values to these types
|
||||
(tf.string, tf.float32))
|
||||
|
||||
# Some Tensorflow data api boilerplate
|
||||
iterator = dataset.make_one_shot_iterator()
|
||||
next_element = iterator.get_next()
|
||||
|
||||
# Up to now we have only defined our computation graph which basically
|
||||
# just connects to Memgraph
|
||||
# `next_element` is not really data but a handle to a node in the Tensorflow
|
||||
# graph, which we can and do evaluate
|
||||
# It is a Tensorflow tensor with shape=(None, 2)
|
||||
# and dtype=(tf.string, tf.float)
|
||||
# shape `None` means the shape of the tensor is unknown at definition time
|
||||
# and is dynamic and will only be known once the tensor has been evaluated
|
||||
|
||||
with tf.Session() as sess:
|
||||
node_ids = sess.run(next_element)
|
||||
# `node_ids` contains IDs and features of all the nodes
|
||||
# in the graph with the label "Train"
|
||||
# It is a numpy.ndarray with a shape ($n_matching_nodes, 2)
|
||||
```
|
||||
|
||||
## Memgraph Client as a Generic Tensorflow Op
|
||||
|
||||
Other than the Tensorflow Data Op, we'll want to support a generic Tensorflow
|
||||
Op which can be put anywhere in the Tensorflow computation Graph. It takes in
|
||||
an arbitrary tensor and produces a tensor. This would be used in the GraphSage
|
||||
algorithm to fetch the lowest level features into Tensorflow
|
||||
|
||||
```python3
|
||||
requested_ids = np.array([1, 2, 3])
|
||||
ids_placeholder = tf.placeholder(tf.int32)
|
||||
|
||||
model = mg_ops.OpenCypher()
|
||||
"hostname:7687", auth=("user", "pass"),
|
||||
"""
|
||||
UNWIND $node_ids as nid
|
||||
MATCH (n:Train {id: nid})
|
||||
RETURN n.features
|
||||
""",
|
||||
|
||||
# What to call the input tensor as an openCypher parameter
|
||||
parameter_name="node_ids",
|
||||
|
||||
# Type of our resulting tensor
|
||||
dtype=(tf.float32)
|
||||
)
|
||||
|
||||
features = model(ids_placeholder)
|
||||
|
||||
with tf.Session() as sess:
|
||||
result = sess.run(features,
|
||||
feed_dict={ids_placeholder: requested_ids})
|
||||
```
|
||||
|
||||
This is probably easier to implement than the Data Op, so it might be a good
|
||||
idea to start with.
|
||||
|
||||
## Production Usage
|
||||
|
||||
During training, in the GraphSage algorithm at least, Memgraph is at the
|
||||
beginning and at the end of the Tensorflow computation graph.
|
||||
At the beginning, the Data Op provides the node IDs which are fed into the
|
||||
generic Tensorflow Op to find their neighbours and their neighbours and
|
||||
their features.
|
||||
|
||||
Production usage differs in that we don't use the Data Op. The Data Op is
|
||||
effectively cut off and the initial input is fed by Tensorflow serving,
|
||||
with the data found in the request.
|
||||
|
||||
For example a JSON request to classify a node might look like:
|
||||
|
||||
`POST http://host:port/v1/models/GraphSage/versions/v1:classify`
|
||||
|
||||
With the contents:
|
||||
|
||||
```json
|
||||
{
|
||||
"examples": [
|
||||
{"node_id": 1},
|
||||
{"node_id": 2}
|
||||
],
|
||||
}
|
||||
```
|
||||
|
||||
Every element of the "examples" list is an example to be computed. Each is
|
||||
represented by a dict with keys matching names of feeds in the Tensorflow
|
||||
graph and values being the values we want fed in for each example
|
||||
|
||||
The REST API then replies in kind with the classification result in JSON
|
||||
|
||||
Note about adding our custom Op to Tensorflow serving.
|
||||
Our Ops .so can be added into the Bazel build to link with Tensorflow serving
|
||||
or it can be dynamically loaded by starting Tensorflow serving with a flag
|
||||
`--custom_op_paths`
|
||||
|
||||
## Considerations
|
||||
|
||||
There might be issues here that the url to connect to Memgraph is
|
||||
hardcoded into the op and would thus be wrong when moved to production,
|
||||
requiring some type of a hack to make work. We probably want to solve
|
||||
this by having the client op take in another tf.Variable as an input
|
||||
which would contain a connection url and username/password.
|
||||
We have to research whether this makes it easy enough to move to
|
||||
production, as the connection string variable is still a part of the
|
||||
graph, but maybe easier to replace.
|
||||
|
||||
It is probably the best idea to utilize openCypher parameters to make
|
||||
our queries flexible. The exact API as to how to declare the parameters
|
||||
in Python is open to discussion.
|
||||
|
||||
The Data Op might not even be necessary to implement as it is not
|
||||
key for production use. It can be replaced in training mode with
|
||||
feed dicts and either
|
||||
|
||||
1. Getting the initial list of nodes via a Python Bolt client
|
||||
2. Creating a separate Tensorflow computation graph that gets all the
|
||||
relevant node IDs into Python
|
||||
@@ -1,22 +0,0 @@
|
||||
# Memgraph LaTeX Beamer Template
|
||||
|
||||
This folder contains all of the needed files for creating a presentation with
|
||||
Memgraph styling. You should use this style for any public presentations.
|
||||
|
||||
Feel free to improve it according to style guidelines and raise issues if you
|
||||
find any.
|
||||
|
||||
## Usage
|
||||
|
||||
Copy the contents of this folder (excluding this README file) to where you
|
||||
want to write your own presentation. After copying, you can start editing the
|
||||
`template.tex` with your content.
|
||||
|
||||
To compile the presentation to a PDF, run `latexmk -pdf -xelatex`. Some
|
||||
directives require XeLaTeX, so you need to pass `-xelatex` as the final option
|
||||
of `latexmk`. You may also need to install some packages if the compilation
|
||||
complains about missing packages.
|
||||
|
||||
To clean up the generated files, use `latexmk -C`. This will also delete the
|
||||
generated PDF. If you wish to remove generated files except the PDF, use
|
||||
`latexmk -c`.
|
||||
@@ -1,82 +0,0 @@
|
||||
\NeedsTeXFormat{LaTeX2e}
|
||||
\ProvidesClass{mg-beamer}[2018/03/26 Memgraph Beamer]
|
||||
|
||||
\DeclareOption*{\PassOptionsToClass{\CurrentOption}{beamer}}
|
||||
|
||||
\ProcessOptions \relax
|
||||
|
||||
\LoadClass{beamer}
|
||||
|
||||
\usetheme{Pittsburgh}
|
||||
|
||||
% Memgraph color palette
|
||||
\definecolor{mg-purple}{HTML}{720096}
|
||||
\definecolor{mg-red}{HTML}{DD2222}
|
||||
\definecolor{mg-orange}{HTML}{FB6E00}
|
||||
\definecolor{mg-yellow}{HTML}{FFC500}
|
||||
\definecolor{mg-gray}{HTML}{857F87}
|
||||
\definecolor{mg-black}{HTML}{231F20}
|
||||
|
||||
\RequirePackage{fontspec}
|
||||
% Title fonts
|
||||
\setbeamerfont{frametitle}{family={\fontspec[Path = ./mg-style/fonts/]{EncodeSansSemiCondensed-Regular.ttf}}}
|
||||
\setbeamerfont{title}{family={\fontspec[Path = ./mg-style/fonts/]{EncodeSansSemiCondensed-Regular.ttf}}}
|
||||
% Body font
|
||||
\RequirePackage[sfdefault,light]{roboto}
|
||||
% Roboto is pretty bad for monospace font. We will find a replacement.
|
||||
% \setmonofont{RobotoMono-Regular.ttf}[Path = ./mg-style/fonts/]
|
||||
|
||||
% Title slide styles
|
||||
% \setbeamerfont{frametitle}{size=\huge}
|
||||
% \setbeamerfont{title}{size=\huge}
|
||||
% \setbeamerfont{date}{size=\tiny}
|
||||
|
||||
% Other typography styles
|
||||
\setbeamertemplate{frametitle}[default][center]
|
||||
\setbeamercolor{frametitle}{fg=mg-black}
|
||||
\setbeamercolor{title}{fg=mg-black}
|
||||
\setbeamercolor{section in toc}{fg=mg-black}
|
||||
\setbeamercolor{local structure}{fg=mg-orange}
|
||||
\setbeamercolor{alert text}{fg=mg-red}
|
||||
|
||||
% Commands
|
||||
\newcommand{\mgalert}[1]{{\usebeamercolor[fg]{alert text}#1}}
|
||||
\newcommand{\titleframe}{\frame[plain]{\titlepage}}
|
||||
\newcommand{\mgtexttt}[1]{{\textcolor{mg-gray}{\texttt{#1}}}}
|
||||
|
||||
% Title slide background
|
||||
\RequirePackage{tikz,calc}
|
||||
% Use title-slide-169 if aspect ration is 16:9
|
||||
\pgfdeclareimage[interpolate=true,width=\paperwidth,height=\paperheight]{logo}{mg-style/title-slide-169}
|
||||
\setbeamertemplate{background}{
|
||||
\begin{tikzpicture}
|
||||
\useasboundingbox (0,0) rectangle (\the\paperwidth,\the\paperheight);
|
||||
\pgftext[at=\pgfpoint{0}{0},left,base]{\pgfuseimage{logo}};
|
||||
\ifnum\thepage>1\relax
|
||||
\useasboundingbox (0,0) rectangle (\the\paperwidth,\the\paperheight);
|
||||
\fill[white, opacity=1](0,\the\paperheight)--(\the\paperwidth,\the\paperheight)--(\the\paperwidth,0)--(0,0)--(0,\the\paperheight);
|
||||
\fi
|
||||
\end{tikzpicture}
|
||||
}
|
||||
|
||||
% Footline content
|
||||
\setbeamertemplate{navigation symbols}{}%remove navigation symbols
|
||||
\setbeamertemplate{footline}{
|
||||
\begin{beamercolorbox}[ht=1.6cm,wd=\paperwidth]{footlinecolor}
|
||||
\vspace{0.1cm}
|
||||
\hfill
|
||||
\begin{minipage}[c]{3cm}
|
||||
\begin{center}
|
||||
\includegraphics[height=0.8cm]{mg-style/memgraph-logo.png}
|
||||
\end{center}
|
||||
\end{minipage}
|
||||
\begin{minipage}[c]{7cm}
|
||||
\insertshorttitle\ --- \insertsection
|
||||
\end{minipage}
|
||||
\begin{minipage}[c]{2cm}
|
||||
\tiny{\insertframenumber{} of \inserttotalframenumber}
|
||||
\end{minipage}
|
||||
\end{beamercolorbox}
|
||||
}
|
||||
|
||||
\endinput
|
||||
Binary file not shown.
Binary file not shown.
|
Before Width: | Height: | Size: 26 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 185 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 189 KiB |
@@ -1,40 +0,0 @@
|
||||
% Set 16:9 aspect ratio
|
||||
\documentclass[aspectratio=169]{mg-beamer}
|
||||
% Default directive sets the regular 4:3 aspect ratio
|
||||
% \documentclass{mg-beamer}
|
||||
\mode<presentation>
|
||||
|
||||
% requires xelatex
|
||||
\usepackage{ccicons}
|
||||
|
||||
\title{Insert Presentation Title}
|
||||
\titlegraphic{\ccbyncnd}
|
||||
\author{Insert Name}
|
||||
|
||||
% Institute doesn't look good in our current styling class.
|
||||
% \institute[Memgraph Ltd.]{\pgfimage[height=1.5cm]{mg-logo.png}}
|
||||
|
||||
% Date is autogenerated on compilation, so no need to set it explicitly,
|
||||
% unless you wish to override it with a different date.
|
||||
% \date{March 23, 2018}
|
||||
|
||||
\begin{document}
|
||||
|
||||
\titleframe
|
||||
|
||||
\section{Intro}
|
||||
|
||||
\begin{frame}{Contents}
|
||||
\tableofcontents
|
||||
\end{frame}
|
||||
|
||||
\begin{frame}{Memgraph Markup Test}
|
||||
\begin{itemize}
|
||||
\item \mgtexttt{Prefer \\mgtexttt for monospace}
|
||||
\item Replace this slide with your own
|
||||
\item Add even more slides in different sections
|
||||
\item Make sure you spellcheck your presentation
|
||||
\end{itemize}
|
||||
\end{frame}
|
||||
|
||||
\end{document}
|
||||
10
environment/README.md
Normal file
10
environment/README.md
Normal file
@@ -0,0 +1,10 @@
|
||||
# Memgraph Operating Environments
|
||||
|
||||
## os
|
||||
|
||||
Under the `os` directory, you can find scripts to install all required system
|
||||
dependencies on operating systems where Memgraph natively builds. The testing
|
||||
script helps to see how to install all packages (in the case of a new package),
|
||||
or make any adjustments in the overall system setup. Also, the testing script
|
||||
helps check if Memgraph runs on a freshly installed operating system (with no
|
||||
packages installed).
|
||||
6
environment/os/.gitignore
vendored
Normal file
6
environment/os/.gitignore
vendored
Normal file
@@ -0,0 +1,6 @@
|
||||
*.deb
|
||||
*.deb.*
|
||||
*.rpm
|
||||
*.rpm.*
|
||||
*.tar.gz
|
||||
*.tar.gz.*
|
||||
196
environment/os/amzn-2.sh
Executable file
196
environment/os/amzn-2.sh
Executable file
@@ -0,0 +1,196 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
DIR="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )"
|
||||
source "$DIR/../util.sh"
|
||||
|
||||
check_operating_system "amzn-2"
|
||||
check_architecture "x86_64"
|
||||
|
||||
TOOLCHAIN_BUILD_DEPS=(
|
||||
gcc gcc-c++ make # generic build tools
|
||||
wget # used for archive download
|
||||
gnupg2 # used for archive signature verification
|
||||
tar gzip bzip2 xz unzip # used for archive unpacking
|
||||
zlib-devel # zlib library used for all builds
|
||||
expat-devel xz-devel python3-devel texinfo
|
||||
curl libcurl-devel # for cmake
|
||||
readline-devel # for cmake and llvm
|
||||
libffi-devel libxml2-devel # for llvm
|
||||
libedit-devel pcre-devel automake bison # for swig
|
||||
file
|
||||
openssl-devel
|
||||
gmp-devel
|
||||
gperf
|
||||
diffutils
|
||||
patch
|
||||
libipt libipt-devel # intel
|
||||
perl # for openssl
|
||||
)
|
||||
|
||||
TOOLCHAIN_RUN_DEPS=(
|
||||
make # generic build tools
|
||||
tar gzip bzip2 xz # used for archive unpacking
|
||||
zlib # zlib library used for all builds
|
||||
expat xz-libs python3 # for gdb
|
||||
readline # for cmake and llvm
|
||||
libffi libxml2 # for llvm
|
||||
openssl-devel
|
||||
)
|
||||
|
||||
MEMGRAPH_BUILD_DEPS=(
|
||||
git # source code control
|
||||
make cmake # build system
|
||||
wget # for downloading libs
|
||||
libuuid-devel java-11-openjdk # required by antlr
|
||||
readline-devel # for memgraph console
|
||||
python3-devel # for query modules
|
||||
openssl-devel
|
||||
libseccomp-devel
|
||||
python3 python3-pip nmap-ncat # for tests
|
||||
#
|
||||
# IMPORTANT: python3-yaml does NOT exist on CentOS
|
||||
# Install it using `pip3 install PyYAML`
|
||||
#
|
||||
PyYAML # Package name here does not correspond to the yum package!
|
||||
libcurl-devel # mg-requests
|
||||
rpm-build rpmlint # for RPM package building
|
||||
doxygen graphviz # source documentation generators
|
||||
which nodejs golang custom-golang1.18.9 zip unzip java-11-openjdk-devel jdk-17 custom-maven3.9.3 # for driver tests
|
||||
autoconf # for jemalloc code generation
|
||||
libtool # for protobuf code generation
|
||||
cyrus-sasl-devel
|
||||
)
|
||||
|
||||
MEMGRAPH_RUN_DEPS=(
|
||||
logrotate openssl python3 libseccomp
|
||||
)
|
||||
|
||||
NEW_DEPS=(
|
||||
wget curl tar gzip
|
||||
)
|
||||
|
||||
list() {
|
||||
echo "$1"
|
||||
}
|
||||
|
||||
check() {
|
||||
local missing=""
|
||||
# On Fedora yum/dnf and python10 use newer glibc which is not compatible
|
||||
# with ours, so we need to momentarely disable env
|
||||
local OLD_LD_LIBRARY_PATH=${LD_LIBRARY_PATH:-""}
|
||||
LD_LIBRARY_PATH=""
|
||||
for pkg in $1; do
|
||||
if [ "$pkg" == custom-maven3.9.3 ]; then
|
||||
if [ ! -f "/opt/apache-maven-3.9.3/bin/mvn" ]; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == custom-golang1.18.9 ]; then
|
||||
if [ ! -f "/opt/go1.18.9/go/bin/go" ]; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == "PyYAML" ]; then
|
||||
if ! python3 -c "import yaml" >/dev/null 2>/dev/null; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if ! yum list installed "$pkg" >/dev/null 2>/dev/null; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
done
|
||||
if [ "$missing" != "" ]; then
|
||||
echo "MISSING PACKAGES: $missing"
|
||||
exit 1
|
||||
fi
|
||||
LD_LIBRARY_PATH=${OLD_LD_LIBRARY_PATH}
|
||||
}
|
||||
|
||||
install() {
|
||||
cd "$DIR"
|
||||
if [ "$EUID" -ne 0 ]; then
|
||||
echo "Please run as root."
|
||||
exit 1
|
||||
fi
|
||||
# If GitHub Actions runner is installed, append LANG to the environment.
|
||||
# Python related tests don't work without the LANG export.
|
||||
if [ -d "/home/gh/actions-runner" ]; then
|
||||
echo "LANG=en_US.utf8" >> /home/gh/actions-runner/.env
|
||||
else
|
||||
echo "NOTE: export LANG=en_US.utf8"
|
||||
fi
|
||||
|
||||
yum update -y
|
||||
for pkg in $1; do
|
||||
if [ "$pkg" == custom-maven3.9.3 ]; then
|
||||
install_custom_maven "3.9.3"
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == custom-golang1.18.9 ]; then
|
||||
install_custom_golang "1.18.9"
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == jdk-17 ]; then
|
||||
if ! yum list installed jdk-17 >/dev/null 2>/dev/null; then
|
||||
wget --no-check-certificate -c --header "Cookie: oraclelicense=accept-securebackup-cookie" https://download.oracle.com/java/17/latest/jdk-17_linux-x64_bin.rpm
|
||||
rpm -Uvh jdk-17_linux-x64_bin.rpm
|
||||
# NOTE: Set Java 11 as default.
|
||||
update-alternatives --set java java-11-openjdk.x86_64
|
||||
update-alternatives --set javac java-11-openjdk.x86_64
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == libipt ]; then
|
||||
if ! yum list installed libipt >/dev/null 2>/dev/null; then
|
||||
yum install -y http://repo.okay.com.mx/centos/8/x86_64/release/libipt-1.6.1-8.el8.x86_64.rpm
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == libipt-devel ]; then
|
||||
if ! yum list installed libipt-devel >/dev/null 2>/dev/null; then
|
||||
yum install -y http://repo.okay.com.mx/centos/8/x86_64/release/libipt-devel-1.6.1-8.el8.x86_64.rpm
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == nodejs ]; then
|
||||
curl -sL https://rpm.nodesource.com/setup_16.x | bash -
|
||||
if ! yum list installed nodejs >/dev/null 2>/dev/null; then
|
||||
yum install -y nodejs
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == PyYAML ]; then
|
||||
if [ -z ${SUDO_USER+x} ]; then # Running as root (e.g. Docker).
|
||||
pip3 install --user PyYAML
|
||||
else # Running using sudo.
|
||||
sudo -H -u "$SUDO_USER" bash -c "pip3 install --user PyYAML"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == nodejs ]; then
|
||||
curl -sL https://rpm.nodesource.com/setup_16.x | bash -
|
||||
if ! yum list installed nodejs >/dev/null 2>/dev/null; then
|
||||
yum install -y nodejs
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == java-11-openjdk ]; then
|
||||
amazon-linux-extras install -y java-openjdk11
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == java-11-openjdk-devel ]; then
|
||||
amazon-linux-extras install -y java-openjdk11
|
||||
yum install -y java-11-openjdk-devel
|
||||
continue
|
||||
fi
|
||||
yum install -y "$pkg"
|
||||
done
|
||||
}
|
||||
|
||||
deps=$2"[*]"
|
||||
"$1" "${!deps}"
|
||||
188
environment/os/centos-7.sh
Executable file
188
environment/os/centos-7.sh
Executable file
@@ -0,0 +1,188 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
DIR="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )"
|
||||
source "$DIR/../util.sh"
|
||||
|
||||
check_operating_system "centos-7"
|
||||
check_architecture "x86_64"
|
||||
|
||||
TOOLCHAIN_BUILD_DEPS=(
|
||||
coreutils gcc gcc-c++ make # generic build tools
|
||||
wget # used for archive download
|
||||
gnupg2 # used for archive signature verification
|
||||
tar gzip bzip2 xz unzip # used for archive unpacking
|
||||
zlib-devel # zlib library used for all builds
|
||||
expat-devel libipt libipt-devel libbabeltrace-devel xz-devel python3-devel # gdb
|
||||
texinfo # gdb
|
||||
libcurl-devel # cmake
|
||||
curl # snappy
|
||||
readline-devel # cmake and llvm
|
||||
libffi-devel libxml2-devel perl-Digest-MD5 # llvm
|
||||
libedit-devel pcre-devel automake bison # swig
|
||||
file
|
||||
openssl-devel
|
||||
gmp-devel
|
||||
gperf
|
||||
patch
|
||||
)
|
||||
|
||||
TOOLCHAIN_RUN_DEPS=(
|
||||
make # generic build tools
|
||||
tar gzip bzip2 xz # used for archive unpacking
|
||||
zlib # zlib library used for all builds
|
||||
expat libipt libbabeltrace xz-libs python3 # for gdb
|
||||
readline # for cmake and llvm
|
||||
libffi libxml2 # for llvm
|
||||
openssl-devel
|
||||
)
|
||||
|
||||
MEMGRAPH_BUILD_DEPS=(
|
||||
make cmake pkgconfig # build system
|
||||
curl wget # for downloading libs
|
||||
libuuid-devel java-11-openjdk # required by antlr
|
||||
readline-devel # for memgraph console
|
||||
python3-devel # for query modules
|
||||
openssl-devel
|
||||
libseccomp-devel
|
||||
python3 python-virtualenv python3-pip nmap-ncat # for qa, macro_benchmark and stress tests
|
||||
#
|
||||
# IMPORTANT: python3-yaml does NOT exist on CentOS
|
||||
# Install it using `pip3 install PyYAML`
|
||||
#
|
||||
PyYAML # Package name here does not correspond to the yum package!
|
||||
libcurl-devel # mg-requests
|
||||
sbcl # for custom Lisp C++ preprocessing
|
||||
rpm-build rpmlint # for RPM package building
|
||||
doxygen graphviz # source documentation generators
|
||||
which mono-complete dotnet-sdk-3.1 golang custom-golang1.18.9 # for driver tests
|
||||
nodejs zip unzip java-11-openjdk-devel jdk-17 custom-maven3.9.3 # for driver tests
|
||||
autoconf # for jemalloc code generation
|
||||
libtool # for protobuf code generation
|
||||
cyrus-sasl-devel
|
||||
)
|
||||
|
||||
MEMGRAPH_RUN_DEPS=(
|
||||
logrotate openssl python3 libseccomp
|
||||
)
|
||||
|
||||
NEW_DEPS=(
|
||||
wget curl tar gzip
|
||||
)
|
||||
|
||||
list() {
|
||||
echo "$1"
|
||||
}
|
||||
|
||||
check() {
|
||||
local missing=""
|
||||
for pkg in $1; do
|
||||
if [ "$pkg" == custom-maven3.9.3 ]; then
|
||||
if [ ! -f "/opt/apache-maven-3.9.3/bin/mvn" ]; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == custom-golang1.18.9 ]; then
|
||||
if [ ! -f "/opt/go1.18.9/go/bin/go" ]; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == git ]; then
|
||||
if ! which "git" >/dev/null; then
|
||||
missing="git $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == "PyYAML" ]; then
|
||||
if ! python3 -c "import yaml" >/dev/null 2>/dev/null; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if ! yum list installed "$pkg" >/dev/null 2>/dev/null; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
done
|
||||
if [ "$missing" != "" ]; then
|
||||
echo "MISSING PACKAGES: $missing"
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
install() {
|
||||
cd "$DIR"
|
||||
if [ "$EUID" -ne 0 ]; then
|
||||
echo "Please run as root."
|
||||
exit 1
|
||||
fi
|
||||
# If GitHub Actions runner is installed, append LANG to the environment.
|
||||
# Python related tests doesn't work the LANG export.
|
||||
if [ -d "/home/gh/actions-runner" ]; then
|
||||
echo "LANG=en_US.utf8" >> /home/gh/actions-runner/.env
|
||||
else
|
||||
echo "NOTE: export LANG=en_US.utf8"
|
||||
fi
|
||||
yum install -y epel-release
|
||||
yum remove -y ius-release
|
||||
yum install -y \
|
||||
https://repo.ius.io/ius-release-el7.rpm
|
||||
yum update -y
|
||||
yum install -y wget python3 python3-pip
|
||||
yum install -y git
|
||||
|
||||
for pkg in $1; do
|
||||
if [ "$pkg" == custom-maven3.9.3 ]; then
|
||||
install_custom_maven "3.9.3"
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == custom-golang1.18.9 ]; then
|
||||
install_custom_golang "1.18.9"
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == jdk-17 ]; then
|
||||
if ! yum list installed jdk-17 >/dev/null 2>/dev/null; then
|
||||
wget https://download.oracle.com/java/17/latest/jdk-17_linux-x64_bin.rpm
|
||||
rpm -ivh jdk-17_linux-x64_bin.rpm
|
||||
update-alternatives --set java java-11-openjdk.x86_64
|
||||
update-alternatives --set javac java-11-openjdk.x86_64
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == libipt ]; then
|
||||
if ! yum list installed libipt >/dev/null 2>/dev/null; then
|
||||
yum install -y http://repo.okay.com.mx/centos/8/x86_64/release/libipt-1.6.1-8.el8.x86_64.rpm
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == libipt-devel ]; then
|
||||
if ! yum list installed libipt-devel >/dev/null 2>/dev/null; then
|
||||
yum install -y http://repo.okay.com.mx/centos/8/x86_64/release/libipt-devel-1.6.1-8.el8.x86_64.rpm
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == dotnet-sdk-3.1 ]; then
|
||||
if ! yum list installed dotnet-sdk-3.1 >/dev/null 2>/dev/null; then
|
||||
wget -nv https://packages.microsoft.com/config/centos/7/packages-microsoft-prod.rpm -O packages-microsoft-prod.rpm
|
||||
rpm -Uvh https://packages.microsoft.com/config/centos/7/packages-microsoft-prod.rpm
|
||||
yum update -y
|
||||
yum install -y dotnet-sdk-3.1
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == PyYAML ]; then
|
||||
if [ -z ${SUDO_USER+x} ]; then # Running as root (e.g. Docker).
|
||||
pip3 install --user PyYAML
|
||||
else # Running using sudo.
|
||||
sudo -H -u "$SUDO_USER" bash -c "pip3 install --user PyYAML"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
yum install -y "$pkg"
|
||||
done
|
||||
}
|
||||
|
||||
deps=$2"[*]"
|
||||
"$1" "${!deps}"
|
||||
195
environment/os/centos-9.sh
Executable file
195
environment/os/centos-9.sh
Executable file
@@ -0,0 +1,195 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
DIR="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )"
|
||||
source "$DIR/../util.sh"
|
||||
|
||||
check_operating_system "centos-9"
|
||||
check_architecture "x86_64"
|
||||
|
||||
TOOLCHAIN_BUILD_DEPS=(
|
||||
coreutils-common gcc gcc-c++ make # generic build tools
|
||||
wget # used for archive download
|
||||
gnupg2 # used for archive signature verification
|
||||
tar gzip bzip2 xz unzip # used for archive unpacking
|
||||
zlib-devel # zlib library used for all builds
|
||||
expat-devel xz-devel python3-devel texinfo libbabeltrace-devel # for gdb
|
||||
readline-devel # for cmake and llvm
|
||||
libffi-devel libxml2-devel # for llvm
|
||||
libedit-devel pcre-devel automake bison # for swig
|
||||
file
|
||||
openssl-devel
|
||||
gmp-devel
|
||||
gperf
|
||||
diffutils
|
||||
libipt libipt-devel # intel
|
||||
patch
|
||||
)
|
||||
|
||||
TOOLCHAIN_RUN_DEPS=(
|
||||
make # generic build tools
|
||||
tar gzip bzip2 xz # used for archive unpacking
|
||||
zlib # zlib library used for all builds
|
||||
expat xz-libs python3 # for gdb
|
||||
readline # for cmake and llvm
|
||||
libffi libxml2 # for llvm
|
||||
openssl-devel
|
||||
perl # for openssl
|
||||
)
|
||||
|
||||
MEMGRAPH_BUILD_DEPS=(
|
||||
git # source code control
|
||||
make cmake pkgconf-pkg-config # build system
|
||||
wget # for downloading libs
|
||||
libuuid-devel java-11-openjdk # required by antlr
|
||||
readline-devel # for memgraph console
|
||||
python3-devel # for query modules
|
||||
openssl-devel
|
||||
libseccomp-devel
|
||||
python3 python3-pip python3-virtualenv nmap-ncat # for qa, macro_benchmark and stress tests
|
||||
#
|
||||
# IMPORTANT: python3-yaml does NOT exist on CentOS
|
||||
# Install it manually using `pip3 install PyYAML`
|
||||
#
|
||||
PyYAML # Package name here does not correspond to the yum package!
|
||||
libcurl-devel # mg-requests
|
||||
rpm-build rpmlint # for RPM package building
|
||||
doxygen graphviz # source documentation generators
|
||||
which nodejs golang custom-golang1.18.9 # for driver tests
|
||||
zip unzip java-11-openjdk-devel java-17-openjdk java-17-openjdk-devel custom-maven3.9.3 # for driver tests
|
||||
sbcl # for custom Lisp C++ preprocessing
|
||||
autoconf # for jemalloc code generation
|
||||
libtool # for protobuf code generation
|
||||
cyrus-sasl-devel
|
||||
)
|
||||
|
||||
MEMGRAPH_RUN_DEPS=(
|
||||
logrotate openssl python3 libseccomp
|
||||
)
|
||||
|
||||
NEW_DEPS=(
|
||||
wget curl tar gzip
|
||||
)
|
||||
|
||||
list() {
|
||||
echo "$1"
|
||||
}
|
||||
|
||||
check() {
|
||||
local missing=""
|
||||
for pkg in $1; do
|
||||
if [ "$pkg" == custom-maven3.9.3 ]; then
|
||||
if [ ! -f "/opt/apache-maven-3.9.3/bin/mvn" ]; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == custom-golang1.18.9 ]; then
|
||||
if [ ! -f "/opt/go1.18.9/go/bin/go" ]; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == "PyYAML" ]; then
|
||||
if ! python3 -c "import yaml" >/dev/null 2>/dev/null; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == "python3-virtualenv" ]; then
|
||||
continue
|
||||
fi
|
||||
if ! yum list installed "$pkg" >/dev/null 2>/dev/null; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
done
|
||||
if [ "$missing" != "" ]; then
|
||||
echo "MISSING PACKAGES: $missing"
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
install() {
|
||||
cd "$DIR"
|
||||
if [ "$EUID" -ne 0 ]; then
|
||||
echo "Please run as root."
|
||||
exit 1
|
||||
fi
|
||||
# If GitHub Actions runner is installed, append LANG to the environment.
|
||||
# Python related tests doesn't work the LANG export.
|
||||
if [ -d "/home/gh/actions-runner" ]; then
|
||||
echo "LANG=en_US.utf8" >> /home/gh/actions-runner/.env
|
||||
else
|
||||
echo "NOTE: export LANG=en_US.utf8"
|
||||
fi
|
||||
yum update -y
|
||||
yum install -y wget git python3 python3-pip
|
||||
|
||||
for pkg in $1; do
|
||||
if [ "$pkg" == custom-maven3.9.3 ]; then
|
||||
install_custom_maven "3.9.3"
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == custom-golang1.18.9 ]; then
|
||||
install_custom_golang "1.18.9"
|
||||
continue
|
||||
fi
|
||||
# Since there is no support for libipt-devel for CentOS 9 we install
|
||||
# Fedoras version of same libs, they are the same version but released
|
||||
# for different OS
|
||||
# TODO Update when libipt-devel releases for CentOS 9
|
||||
if [ "$pkg" == libipt ]; then
|
||||
if ! dnf list installed libipt >/dev/null 2>/dev/null; then
|
||||
dnf install -y http://repo.okay.com.mx/centos/8/x86_64/release/libipt-1.6.1-8.el8.x86_64.rpm
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == libipt-devel ]; then
|
||||
if ! dnf list installed libipt-devel >/dev/null 2>/dev/null; then
|
||||
dnf install -y http://repo.okay.com.mx/centos/8/x86_64/release/libipt-devel-1.6.1-8.el8.x86_64.rpm
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == libbabeltrace-devel ]; then
|
||||
if ! dnf list installed libbabeltrace-devel >/dev/null 2>/dev/null; then
|
||||
dnf install -y http://mirror.stream.centos.org/9-stream/CRB/x86_64/os/Packages/libbabeltrace-devel-1.5.8-10.el9.x86_64.rpm
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == sbcl ]; then
|
||||
if ! dnf list installed cl-asdf >/dev/null 2>/dev/null; then
|
||||
dnf install -y https://pkgs.dyn.su/el8/base/x86_64/cl-asdf-20101028-18.el8.noarch.rpm
|
||||
fi
|
||||
if ! dnf list installed common-lisp-controller >/dev/null 2>/dev/null; then
|
||||
dnf install -y https://pkgs.dyn.su/el8/base/x86_64/common-lisp-controller-7.4-20.el8.noarch.rpm
|
||||
fi
|
||||
if ! dnf list installed sbcl >/dev/null 2>/dev/null; then
|
||||
dnf install -y https://pkgs.dyn.su/el8/base/x86_64/sbcl-2.0.1-4.el8.x86_64.rpm
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == PyYAML ]; then
|
||||
if [ -z ${SUDO_USER+x} ]; then # Running as root (e.g. Docker).
|
||||
pip3 install --user PyYAML
|
||||
else # Running using sudo.
|
||||
sudo -H -u "$SUDO_USER" bash -c "pip3 install --user PyYAML"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == python3-virtualenv ]; then
|
||||
if [ -z ${SUDO_USER+x} ]; then # Running as root (e.g. Docker).
|
||||
pip3 install virtualenv
|
||||
pip3 install virtualenvwrapper
|
||||
else # Running using sudo.
|
||||
sudo -H -u "$SUDO_USER" bash -c "pip3 install virtualenv"
|
||||
sudo -H -u "$SUDO_USER" bash -c "pip3 install virtualenvwrapper"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
yum install -y "$pkg"
|
||||
done
|
||||
}
|
||||
|
||||
deps=$2"[*]"
|
||||
"$1" "${!deps}"
|
||||
159
environment/os/debian-10.sh
Executable file
159
environment/os/debian-10.sh
Executable file
@@ -0,0 +1,159 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
DIR="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )"
|
||||
source "$DIR/../util.sh"
|
||||
|
||||
check_operating_system "debian-10"
|
||||
check_architecture "x86_64"
|
||||
|
||||
TOOLCHAIN_BUILD_DEPS=(
|
||||
coreutils gcc g++ build-essential make # generic build tools
|
||||
wget # used for archive download
|
||||
gnupg # used for archive signature verification
|
||||
tar gzip bzip2 xz-utils unzip # used for archive unpacking
|
||||
zlib1g-dev # zlib library used for all builds
|
||||
libexpat1-dev libipt-dev libbabeltrace-dev liblzma-dev python3-dev texinfo # for gdb
|
||||
libcurl4-openssl-dev # for cmake
|
||||
libreadline-dev # for cmake and llvm
|
||||
libffi-dev libxml2-dev # for llvm
|
||||
curl # snappy
|
||||
file # for libunwind
|
||||
libssl-dev # for libevent
|
||||
libgmp-dev # for gdb
|
||||
gperf # for proxygen
|
||||
git # for fbthrift
|
||||
libedit-dev libpcre3-dev automake bison # for swig
|
||||
)
|
||||
|
||||
TOOLCHAIN_RUN_DEPS=(
|
||||
make # generic build tools
|
||||
tar gzip bzip2 xz-utils # used for archive unpacking
|
||||
zlib1g # zlib library used for all builds
|
||||
libexpat1 libipt2 libbabeltrace1 liblzma5 python3 # for gdb
|
||||
libcurl4 # for cmake
|
||||
libreadline7 # for cmake and llvm
|
||||
libffi6 libxml2 # for llvm
|
||||
libssl-dev # for libevent
|
||||
)
|
||||
|
||||
MEMGRAPH_BUILD_DEPS=(
|
||||
git # source code control
|
||||
make cmake pkg-config # build system
|
||||
curl wget # for downloading libs
|
||||
uuid-dev default-jre-headless # required by antlr
|
||||
libreadline-dev # for memgraph console
|
||||
libpython3-dev python3-dev # for query modules
|
||||
libssl-dev
|
||||
libseccomp-dev
|
||||
netcat # tests are using nc to wait for memgraph
|
||||
python3 virtualenv python3-virtualenv python3-pip # for qa, macro_benchmark and stress tests
|
||||
python3-yaml # for the configuration generator
|
||||
libcurl4-openssl-dev # mg-requests
|
||||
sbcl # for custom Lisp C++ preprocessing
|
||||
doxygen graphviz # source documentation generators
|
||||
mono-runtime mono-mcs zip unzip default-jdk-headless oracle-java17-installer custom-maven3.9.3 # for driver tests
|
||||
dotnet-sdk-3.1 golang custom-golang1.18.9 nodejs npm # for driver tests
|
||||
autoconf # for jemalloc code generation
|
||||
libtool # for protobuf code generation
|
||||
libsasl2-dev
|
||||
)
|
||||
|
||||
MEMGRAPH_RUN_DEPS=(
|
||||
logrotate openssl python3 libseccomp
|
||||
)
|
||||
|
||||
NEW_DEPS=(
|
||||
wget curl tar gzip
|
||||
)
|
||||
|
||||
list() {
|
||||
echo "$1"
|
||||
}
|
||||
|
||||
check() {
|
||||
local missing=""
|
||||
for pkg in $1; do
|
||||
if [ "$pkg" == custom-maven3.9.3 ]; then
|
||||
if [ ! -f "/opt/apache-maven-3.9.3/bin/mvn" ]; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == custom-golang1.18.9 ]; then
|
||||
if [ ! -f "/opt/go1.18.9/go/bin/go" ]; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if ! dpkg -s "$pkg" >/dev/null 2>/dev/null; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
done
|
||||
if [ "$missing" != "" ]; then
|
||||
echo "MISSING PACKAGES: $missing"
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
install() {
|
||||
cat >/etc/apt/sources.list <<EOF
|
||||
deb http://deb.debian.org/debian/ buster main non-free contrib
|
||||
deb-src http://deb.debian.org/debian/ buster main non-free contrib
|
||||
deb http://deb.debian.org/debian/ buster-updates main contrib non-free
|
||||
deb-src http://deb.debian.org/debian/ buster-updates main contrib non-free
|
||||
deb http://security.debian.org/debian-security buster/updates main contrib non-free
|
||||
deb-src http://security.debian.org/debian-security buster/updates main contrib non-free
|
||||
EOF
|
||||
apt --allow-releaseinfo-change update
|
||||
cat >/etc/apt/sources.list.d/java.list << EOF
|
||||
deb http://ppa.launchpad.net/linuxuprising/java/ubuntu bionic main
|
||||
deb-src http://ppa.launchpad.net/linuxuprising/java/ubuntu bionic main
|
||||
EOF
|
||||
cd "$DIR"
|
||||
apt install -y gnupg
|
||||
apt-key adv --keyserver keyserver.ubuntu.com --recv-keys EA8CACC073C3DB2A
|
||||
apt --allow-releaseinfo-change update
|
||||
# If GitHub Actions runner is installed, append LANG to the environment.
|
||||
# Python related tests doesn't work the LANG export.
|
||||
if [ -d "/home/gh/actions-runner" ]; then
|
||||
echo "LANG=en_US.utf8" >> /home/gh/actions-runner/.env
|
||||
else
|
||||
echo "NOTE: export LANG=en_US.utf8"
|
||||
fi
|
||||
|
||||
for pkg in $1; do
|
||||
if [ "$pkg" == custom-maven3.9.3 ]; then
|
||||
install_custom_maven "3.9.3"
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == custom-golang1.18.9 ]; then
|
||||
install_custom_golang "1.18.9"
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == oracle-java17-installer ]; then
|
||||
if ! dpkg -s "$pkg" 2>/dev/null >/dev/null; then
|
||||
echo oracle-java17-installer shared/accepted-oracle-license-v1-3 select true | /usr/bin/debconf-set-selections
|
||||
echo oracle-java17-installer shared/accepted-oracle-license-v1-3 seen true | /usr/bin/debconf-set-selections
|
||||
apt install -y "$pkg"
|
||||
update-alternatives --set java /usr/lib/jvm/java-11-openjdk-amd64/bin/java
|
||||
update-alternatives --set javac /usr/lib/jvm/java-11-openjdk-amd64/bin/javac
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == dotnet-sdk-3.1 ]; then
|
||||
if ! dpkg -s "$pkg" 2>/dev/null >/dev/null; then
|
||||
wget -nv https://packages.microsoft.com/config/debian/10/packages-microsoft-prod.deb -O packages-microsoft-prod.deb
|
||||
dpkg -i packages-microsoft-prod.deb
|
||||
apt-get update
|
||||
apt-get install -y apt-transport-https dotnet-sdk-3.1
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
apt install -y "$pkg"
|
||||
done
|
||||
}
|
||||
|
||||
deps=$2"[*]"
|
||||
"$1" "${!deps}"
|
||||
146
environment/os/debian-11-arm.sh
Executable file
146
environment/os/debian-11-arm.sh
Executable file
@@ -0,0 +1,146 @@
|
||||
#!/bin/bash
|
||||
|
||||
set -Eeuo pipefail
|
||||
|
||||
DIR="$( cd "$( dirname "${BASH_SOURCE[0]}" )" >/dev/null 2>&1 && pwd )"
|
||||
source "$DIR/../util.sh"
|
||||
|
||||
check_operating_system "debian-11"
|
||||
check_architecture "arm64" "aarch64"
|
||||
|
||||
TOOLCHAIN_BUILD_DEPS=(
|
||||
coreutils gcc g++ build-essential make # generic build tools
|
||||
wget # used for archive download
|
||||
gnupg # used for archive signature verification
|
||||
tar gzip bzip2 xz-utils unzip # used for archive unpacking
|
||||
zlib1g-dev # zlib library used for all builds
|
||||
libexpat1-dev liblzma-dev python3-dev texinfo # for gdb
|
||||
libcurl4-openssl-dev # for cmake
|
||||
libreadline-dev # for cmake and llvm
|
||||
libffi-dev libxml2-dev # for llvm
|
||||
libedit-dev libpcre3-dev automake bison # for swig
|
||||
curl # snappy
|
||||
file # for libunwind
|
||||
libssl-dev # for libevent
|
||||
libgmp-dev
|
||||
gperf # for proxygen
|
||||
git # for fbthrift
|
||||
)
|
||||
|
||||
TOOLCHAIN_RUN_DEPS=(
|
||||
make # generic build tools
|
||||
tar gzip bzip2 xz-utils # used for archive unpacking
|
||||
zlib1g # zlib library used for all builds
|
||||
libexpat1 liblzma5 python3 # for gdb
|
||||
libcurl4 # for cmake
|
||||
file # for CPack
|
||||
libreadline8 # for cmake and llvm
|
||||
libffi7 libxml2 # for llvm
|
||||
libssl-dev # for libevent
|
||||
)
|
||||
|
||||
MEMGRAPH_BUILD_DEPS=(
|
||||
git # source code control
|
||||
make pkg-config # build system
|
||||
curl wget # for downloading libs
|
||||
uuid-dev default-jre-headless # required by antlr
|
||||
libreadline-dev # for memgraph console
|
||||
libpython3-dev python3-dev # for query modules
|
||||
libssl-dev
|
||||
libseccomp-dev
|
||||
netcat # tests are using nc to wait for memgraph
|
||||
python3 virtualenv python3-virtualenv python3-pip # for qa, macro_benchmark and stress tests
|
||||
python3-yaml # for the configuration generator
|
||||
libcurl4-openssl-dev # mg-requests
|
||||
sbcl # for custom Lisp C++ preprocessing
|
||||
doxygen graphviz # source documentation generators
|
||||
mono-runtime mono-mcs zip unzip default-jdk-headless openjdk-17-jdk custom-maven3.9.3 # for driver tests
|
||||
golang custom-golang1.18.9 nodejs npm
|
||||
autoconf # for jemalloc code generation
|
||||
libtool # for protobuf code generation
|
||||
libsasl2-dev
|
||||
)
|
||||
|
||||
MEMGRAPH_RUN_DEPS=(
|
||||
logrotate openssl python3 libseccomp
|
||||
)
|
||||
|
||||
NEW_DEPS=(
|
||||
wget curl tar gzip
|
||||
)
|
||||
|
||||
list() {
|
||||
echo "$1"
|
||||
}
|
||||
|
||||
check() {
|
||||
local missing=""
|
||||
for pkg in $1; do
|
||||
if [ "$pkg" == custom-maven3.9.3 ]; then
|
||||
if [ ! -f "/opt/apache-maven-3.9.3/bin/mvn" ]; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == custom-golang1.18.9 ]; then
|
||||
if [ ! -f "/opt/go1.18.9/go/bin/go" ]; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
if ! dpkg -s "$pkg" >/dev/null 2>/dev/null; then
|
||||
missing="$pkg $missing"
|
||||
fi
|
||||
done
|
||||
if [ "$missing" != "" ]; then
|
||||
echo "MISSING PACKAGES: $missing"
|
||||
exit 1
|
||||
fi
|
||||
}
|
||||
|
||||
install() {
|
||||
cat >/etc/apt/sources.list <<EOF
|
||||
deb http://deb.debian.org/debian bullseye main
|
||||
deb-src http://deb.debian.org/debian bullseye main
|
||||
|
||||
deb http://deb.debian.org/debian-security/ bullseye-security main
|
||||
deb-src http://deb.debian.org/debian-security/ bullseye-security main
|
||||
|
||||
deb http://deb.debian.org/debian bullseye-updates main
|
||||
deb-src http://deb.debian.org/debian bullseye-updates main
|
||||
EOF
|
||||
cd "$DIR"
|
||||
apt update
|
||||
# If GitHub Actions runner is installed, append LANG to the environment.
|
||||
# Python related tests doesn't work the LANG export.
|
||||
if [ -d "/home/gh/actions-runner" ]; then
|
||||
echo "LANG=en_US.utf8" >> /home/gh/actions-runner/.env
|
||||
else
|
||||
echo "NOTE: export LANG=en_US.utf8"
|
||||
fi
|
||||
apt install -y wget
|
||||
|
||||
for pkg in $1; do
|
||||
if [ "$pkg" == custom-maven3.9.3 ]; then
|
||||
install_custom_maven "3.9.3"
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == custom-golang1.18.9 ]; then
|
||||
install_custom_golang "1.18.9"
|
||||
continue
|
||||
fi
|
||||
if [ "$pkg" == openjdk-17-jdk ]; then
|
||||
if ! dpkg -s "$pkg" 2>/dev/null >/dev/null; then
|
||||
apt install -y "$pkg"
|
||||
# The default Java version should be Java 11
|
||||
update-alternatives --set java /usr/lib/jvm/java-11-openjdk-arm64/bin/java
|
||||
update-alternatives --set javac /usr/lib/jvm/java-11-openjdk-arm64/bin/javac
|
||||
fi
|
||||
continue
|
||||
fi
|
||||
apt install -y "$pkg"
|
||||
done
|
||||
}
|
||||
|
||||
deps=$2"[*]"
|
||||
"$1" "${!deps}"
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user