Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b0403f1fa2 | ||
|
|
ce8c3993c5 | ||
|
|
5582a41bb6 | ||
|
|
fd4b5cb3fa | ||
|
|
3cdb8c6046 | ||
|
|
385823ea49 | ||
|
|
5e7333d2dd | ||
|
|
41b1b5df18 | ||
|
|
78e0d87177 | ||
|
|
c6db0a7c20 | ||
|
|
c6e5d1d5fe | ||
|
|
8ea8f4220c | ||
|
|
1c646662e9 | ||
|
|
366c6aff81 | ||
|
|
5d887c58ae | ||
|
|
aa8e2d1712 | ||
|
|
452b5b8a3b | ||
|
|
3dd48b5b45 | ||
|
|
057f039c4b | ||
|
|
4dca45ad24 | ||
|
|
b17499f907 | ||
|
|
29c27bc13e | ||
|
|
c61c535c32 | ||
|
|
2f17e4fb04 | ||
|
|
63057253d8 | ||
|
|
3d31fc3bee | ||
|
|
87d8e71708 | ||
|
|
f70dc8acb2 | ||
|
|
9180659f8b | ||
|
|
c2d80e8ced | ||
|
|
e3243819ef | ||
|
|
a6c8a15cad | ||
|
|
3e2649f1f1 | ||
|
|
a0da8390a2 | ||
|
|
8dfc501fb8 | ||
|
|
9d4325ee25 | ||
|
|
707c132392 | ||
|
|
08e3f958fa | ||
|
|
23b3e21817 | ||
|
|
981aa5c12f | ||
|
|
16e3c5a8f9 | ||
|
|
adfd2dc7c0 | ||
|
|
8bf9b8abc1 | ||
|
|
2a189709e0 | ||
|
|
958ebee091 | ||
|
|
319bbcc1a7 | ||
|
|
87b7c3ac1a | ||
|
|
8007ccd51b | ||
|
|
9cc750fd66 | ||
|
|
aa92b37589 | ||
|
|
8f479b22b9 | ||
|
|
854c7fdddb | ||
|
|
31bc07955c | ||
|
|
f330d6175a | ||
|
|
427c36888e | ||
|
|
cb02bd190b | ||
|
|
951ec79654 | ||
|
|
3e012c9260 | ||
|
|
758e963a4e | ||
|
|
26dcec4812 | ||
|
|
3424757f4d | ||
|
|
70ffa8ce5c | ||
|
|
99176b3e04 | ||
|
|
22ce9f3fad | ||
|
|
a5a3afd923 | ||
|
|
095c131fbb | ||
|
|
89eef40ca2 | ||
|
|
752576ce47 | ||
|
|
706721f8c8 | ||
|
|
8a5cf17cb2 | ||
|
|
a363e5fe6d | ||
|
|
6e434bcaaf | ||
|
|
68d3067125 | ||
|
|
d94058fad9 | ||
|
|
c1c7eeaa69 | ||
|
|
542736ce25 | ||
|
|
13a0a63bef | ||
|
|
d996eb82ef | ||
|
|
4e57d3f76f | ||
|
|
2fcf389f2a | ||
|
|
9500539c55 | ||
|
|
095842a748 | ||
|
|
63ae981599 | ||
|
|
cc3874ab87 | ||
|
|
f05912dea2 | ||
|
|
84471e238e | ||
|
|
d1df881ec5 | ||
|
|
53949521de | ||
|
|
b704179f15 | ||
|
|
557e0b1c07 | ||
|
|
a39ffc1fe9 | ||
|
|
f829d46535 | ||
|
|
0258e85186 | ||
|
|
f364dcca2d | ||
|
|
2114c65012 | ||
|
|
ed7c539303 | ||
|
|
1d09d67909 | ||
|
|
1f92040fcf | ||
|
|
0f2c356b07 | ||
|
|
0e3ee9afb4 | ||
|
|
ab5e01d6bc | ||
|
|
b49bc14f96 | ||
|
|
883d9e3a75 | ||
|
|
07fd2fa8a6 | ||
|
|
afcc2ff6e8 | ||
|
|
4b0bd5b0bd | ||
|
|
1ad503001f | ||
|
|
6c95ec1d6c | ||
|
|
abe33257d9 | ||
|
|
c8b6cbc6e1 | ||
|
|
1cb927aef6 | ||
|
|
b417685430 | ||
|
|
89ef4c0702 | ||
|
|
2d311dbb01 | ||
|
|
68dccc55ad | ||
|
|
68683e181c | ||
|
|
ef74527d92 | ||
|
|
f20684e7b5 | ||
|
|
1a2da02db6 | ||
|
|
6e09e05af5 | ||
|
|
cb261828bd | ||
|
|
9265234299 | ||
|
|
7939ba031d | ||
|
|
2a8af82f50 | ||
|
|
3abc801d7a | ||
|
|
774c05ab55 | ||
|
|
44c064c0b4 | ||
|
|
764fb8cc74 | ||
|
|
ab06a5a058 | ||
|
|
3627bbe12c | ||
|
|
f1d6542b1a | ||
|
|
33f03f6fc8 | ||
|
|
985bf68f34 | ||
|
|
0200e8ada6 | ||
|
|
de32f40b98 | ||
|
|
3b60921c53 | ||
|
|
a9e02dfd29 | ||
|
|
e66a50ec3c | ||
|
|
ef24ab7821 | ||
|
|
d3ada8090f | ||
|
|
2f1d917cf1 | ||
|
|
7ad3cea7fa | ||
|
|
5304318335 | ||
|
|
025790fc50 | ||
|
|
438adc917b | ||
|
|
e3a8921ab5 | ||
|
|
2d1642504d | ||
|
|
097f310797 | ||
|
|
832090d821 | ||
|
|
af6fa6f732 | ||
|
|
a90d2ea290 | ||
|
|
9e4413d1e3 | ||
|
|
c6bd5d3542 | ||
|
|
f9d4a5c435 | ||
|
|
56bc353717 | ||
|
|
6ad37b6550 | ||
|
|
710f70f963 | ||
|
|
90c2349b35 | ||
|
|
92dcfeae8b | ||
|
|
a5cf561288 | ||
|
|
1848809f66 | ||
|
|
d7a448f9ae | ||
|
|
658424fc83 | ||
|
|
3f06ddfb7b | ||
|
|
1cddd0d3ba | ||
|
|
ee933d9e2b | ||
|
|
1d5e13e121 | ||
|
|
032357ec0f | ||
|
|
695126ccce | ||
|
|
725cd268e6 | ||
|
|
66df58f961 | ||
|
|
c4f7efc3cd | ||
|
|
b6d129dce0 | ||
|
|
b045fe4e17 | ||
|
|
c5f91abaf7 | ||
|
|
6c202f495c | ||
|
|
bf13c84977 | ||
|
|
a266351c0a | ||
|
|
914cfec777 | ||
|
|
e2608478b6 | ||
|
|
57807cd338 | ||
|
|
7f5f588232 | ||
|
|
662cb2fe75 | ||
|
|
87124a38b6 | ||
|
|
1583d60cd6 | ||
|
|
8b4bde19b4 | ||
|
|
f69090d56b | ||
|
|
f723c65f1b | ||
|
|
9f376fb803 | ||
|
|
b08629a426 | ||
|
|
1cd622bdca | ||
|
|
d9134f8f95 | ||
|
|
7a40fd630d | ||
|
|
b9361ad5fe | ||
|
|
83c0348553 | ||
|
|
f164012c19 | ||
|
|
98be450f1d | ||
|
|
0f6e3a8273 | ||
|
|
de4e92ac39 | ||
|
|
49455c43ae | ||
|
|
c2694fb696 | ||
|
|
c88f9fe26f | ||
|
|
855ec46a6a | ||
|
|
f7353db7eb | ||
|
|
294492dbf2 | ||
|
|
192799539f | ||
|
|
a8850a8d30 | ||
|
|
f35ad82314 | ||
|
|
a034773497 | ||
|
|
fd5c325886 | ||
|
|
17eb33e0c3 | ||
|
|
0aeb86d78d | ||
|
|
8afb72a326 | ||
|
|
6c1e55d07c | ||
|
|
8206c782b5 | ||
|
|
04589f90d7 | ||
|
|
09f8a2f374 | ||
|
|
324f861f0e | ||
|
|
a50f3b517c | ||
|
|
e6f1667a3d | ||
|
|
ff20d534c6 | ||
|
|
337fc3d6fd | ||
|
|
870b6bd487 | ||
|
|
c688537d49 | ||
|
|
285134e43d | ||
|
|
daea83d2cf | ||
|
|
e3b9397dfe | ||
|
|
31f2b27a05 | ||
|
|
b61e3021b6 | ||
|
|
a71feb6dd8 | ||
|
|
6d1a15987b | ||
|
|
2bfffe85e9 | ||
|
|
182737f3cc | ||
|
|
7b5fbf7b3f | ||
|
|
26e5871c67 | ||
|
|
9771ca726a | ||
|
|
42a8981b46 | ||
|
|
5c8097c2de | ||
|
|
5c23f59ee3 | ||
|
|
7cfd894f3a | ||
|
|
31f097d418 | ||
|
|
33d653e24f | ||
|
|
f5e046a730 | ||
|
|
5dbcb3e4ab | ||
|
|
eb50eb20a5 | ||
|
|
c2d3e28540 | ||
|
|
0c1a764074 | ||
|
|
f86575f210 | ||
|
|
dcd0b3d020 | ||
|
|
d88611d36f | ||
|
|
4c5a9076d7 | ||
|
|
8aab7ca84c | ||
|
|
781ccc1bee | ||
|
|
ee96a5a6f5 | ||
|
|
9c81f8bd61 | ||
|
|
0f65806b5b | ||
|
|
5b8b58e472 | ||
|
|
342ee426ad | ||
|
|
4a95b3005a | ||
|
+3 |
73a9b916c9 | ||
|
|
dc0ee51cb1 | ||
|
|
21aee83abd | ||
|
|
08d714d0e5 | ||
|
|
4a12291765 | ||
|
|
8e9f5146dd | ||
|
|
04f63d4af7 | ||
|
|
dc57ee03b1 | ||
|
|
8144019a13 | ||
|
|
7665bdc91a | ||
|
|
64a40b20d9 | ||
|
|
08c2b276fb | ||
|
|
1f09a55eba | ||
|
|
684077682e | ||
|
|
f8942f93a6 | ||
|
|
c51c96656b | ||
|
|
0dd057222b | ||
|
|
59953d2df6 | ||
|
|
ddafac4c6c | ||
|
|
2af69a931a | ||
|
|
06b144aa09 | ||
|
|
db33b67d37 | ||
|
|
a106198878 | ||
|
|
05b99c8f4c | ||
|
|
9ebf80a28c | ||
|
|
155634502d | ||
|
|
79fd255828 | ||
|
|
5b84dc9678 | ||
|
|
701f06657d | ||
|
|
cf83803880 | ||
|
|
54038811c0 | ||
|
|
fdeb97629e | ||
|
|
9906daf5c9 | ||
|
|
ded8d993b7 | ||
|
|
6437d07b03 | ||
|
|
4b29be3f36 | ||
|
|
2ec78d262d | ||
|
|
0a24bfec13 | ||
|
|
611c950293 | ||
|
|
14c48cc5ed | ||
|
|
b08fd9f989 | ||
|
|
dd20aa5160 | ||
|
|
0a8e546957 | ||
|
|
4f8cdc2a1e | ||
|
|
3788a055fe | ||
|
|
b71a610f5c | ||
|
|
efcfffc528 | ||
|
|
99d09c824c | ||
|
|
6c5e263d7b | ||
|
|
3fc7ef8f81 | ||
|
|
a38c29b6c8 | ||
|
|
472d8a0df5 | ||
|
|
8ec5da785b | ||
|
|
3a2669e8ae | ||
|
|
f39e51178e | ||
|
|
5f7d7ce8b0 | ||
|
|
d092d64d7a | ||
|
|
30eb6c3210 | ||
|
|
80458d0490 | ||
|
|
99c40ab53d | ||
|
|
26ef28467a | ||
|
|
5f28bd96db | ||
|
|
438cfed70a | ||
|
|
3a3cacfefa | ||
|
|
617909e943 | ||
|
|
e9cbfccec6 | ||
|
|
7b69183a8d | ||
|
|
af7fca797a | ||
|
|
9146ac050b | ||
|
|
ef3a5f58a8 | ||
|
|
f2f85ba354 | ||
|
|
c115c5230e | ||
|
|
65a87cb773 | ||
|
|
9a89e32a95 | ||
|
|
5fe3e74a38 | ||
|
|
610eb26c11 | ||
|
|
4955375ec7 | ||
|
|
d265a61438 | ||
|
|
406a365f44 | ||
|
|
2dd4cca363 | ||
|
|
bcae770819 | ||
|
|
bf802b0764 | ||
|
|
ac03e3721d | ||
|
|
31227f4faf | ||
|
|
f6f11f3ef1 | ||
|
|
569584d463 | ||
|
|
ea3e8b79a1 | ||
|
|
74609d44cd | ||
|
|
ea05c6ac47 | ||
|
|
05aed4cab9 | ||
|
|
de7f2f87f7 | ||
|
|
fb8755a636 | ||
|
|
d3d94ccf2e | ||
|
|
00a8e72cfc | ||
|
|
a31b516e25 | ||
|
|
c77b8f45e9 | ||
|
|
a7afd1d2b2 | ||
|
|
3fcddfb61f | ||
|
|
3c9f5954b5 | ||
|
|
3b1b1d1486 | ||
|
|
60f22ca830 | ||
|
|
6ceadfb580 | ||
|
|
7f0a7f0a69 | ||
|
|
3264deb24e | ||
|
|
6c6489280c | ||
|
|
2b88db90aa | ||
|
|
b94b714f81 | ||
|
|
806459f481 | ||
|
|
3a08819f51 | ||
|
|
6f0ddc9d92 | ||
|
|
731f2dc5c7 | ||
|
|
bf643a63c8 | ||
|
|
e4ddc34463 | ||
|
|
6d5d754119 | ||
|
|
e3e631f394 | ||
|
|
6263823e54 | ||
|
|
4dd8b1faa9 | ||
|
|
8038eb3147 | ||
|
|
89742a95db | ||
|
|
60e9e630bd | ||
|
|
e750c619b2 | ||
|
|
93fb83b4cb |
@@ -62,7 +62,7 @@ jobs:
|
||||
- uses: actions/checkout@v4
|
||||
- name: make
|
||||
run: |
|
||||
sudo apt-get update && sudo apt-get install libc6-dev-i386
|
||||
sudo apt-get update && sudo apt-get install libc6-dev-i386 gcc-multilib g++-multilib
|
||||
make REDIS_CFLAGS='-Werror' 32bit
|
||||
|
||||
build-libc-malloc:
|
||||
@@ -79,7 +79,7 @@ jobs:
|
||||
- uses: actions/checkout@v4
|
||||
- name: make
|
||||
run: |
|
||||
dnf -y install which gcc make
|
||||
dnf -y install which gcc gcc-c++ make
|
||||
make REDIS_CFLAGS='-Werror'
|
||||
|
||||
build-old-chain-jemalloc:
|
||||
@@ -96,6 +96,7 @@ jobs:
|
||||
apt-key adv --keyserver keyserver.ubuntu.com --recv-keys 40976EAF437D05B5
|
||||
apt-key adv --keyserver keyserver.ubuntu.com --recv-keys 3B4FE6ACC0B21F32
|
||||
apt-get update
|
||||
apt-get install -y make gcc-4.8
|
||||
apt-get install -y make gcc-4.8 g++-4.8
|
||||
update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-4.8 100
|
||||
update-alternatives --install /usr/bin/g++ g++ /usr/bin/g++-4.8 100
|
||||
make CC=gcc REDIS_CFLAGS='-Werror'
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
name: "Codecov"
|
||||
|
||||
# Enabling on each push is to display the coverage changes in every PR,
|
||||
# where each PR needs to be compared against the coverage of the head commit
|
||||
on: [push, pull_request]
|
||||
|
||||
jobs:
|
||||
code-coverage:
|
||||
runs-on: ubuntu-22.04
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
|
||||
- name: Install lcov and run test
|
||||
run: |
|
||||
sudo apt-get install lcov
|
||||
make lcov
|
||||
|
||||
- name: Upload coverage reports to Codecov
|
||||
uses: codecov/codecov-action@v4
|
||||
with:
|
||||
token: ${{ secrets.CODECOV_TOKEN }}
|
||||
file: ./src/redis.info
|
||||
+22
-18
@@ -76,7 +76,6 @@ jobs:
|
||||
if: |
|
||||
(github.event_name == 'workflow_dispatch' || (github.event_name != 'workflow_dispatch' && github.repository == 'redis/redis')) &&
|
||||
!contains(github.event.inputs.skipjobs, 'fortify')
|
||||
container: ubuntu:lunar
|
||||
timeout-minutes: 14400
|
||||
steps:
|
||||
- name: prep
|
||||
@@ -94,11 +93,10 @@ jobs:
|
||||
ref: ${{ env.GITHUB_HEAD_REF }}
|
||||
- name: make
|
||||
run: |
|
||||
apt-get update && apt-get install -y make gcc-13
|
||||
update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-13 100
|
||||
apt-get update && apt-get install -y make gcc g++
|
||||
make CC=gcc REDIS_CFLAGS='-Werror -DREDIS_TEST -U_FORTIFY_SOURCE -D_FORTIFY_SOURCE=3'
|
||||
- name: testprep
|
||||
run: apt-get install -y tcl8.6 tclx procps
|
||||
run: sudo apt-get install -y tcl8.6 tclx procps
|
||||
- name: test
|
||||
if: true && !contains(github.event.inputs.skiptests, 'redis')
|
||||
run: ./runtest --accurate --verbose --dump-logs ${{github.event.inputs.test_args}}
|
||||
@@ -211,7 +209,7 @@ jobs:
|
||||
ref: ${{ env.GITHUB_HEAD_REF }}
|
||||
- name: make
|
||||
run: |
|
||||
sudo apt-get update && sudo apt-get install libc6-dev-i386
|
||||
sudo apt-get update && sudo apt-get install libc6-dev-i386 g++ gcc-multilib g++-multilib
|
||||
make 32bit REDIS_CFLAGS='-Werror -DREDIS_TEST'
|
||||
- name: testprep
|
||||
run: sudo apt-get install tcl8.6 tclx
|
||||
@@ -456,7 +454,7 @@ jobs:
|
||||
- name: testprep
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install tcl8.6 tclx valgrind -y
|
||||
sudo apt-get install tcl8.6 tclx valgrind g++ -y
|
||||
- name: test
|
||||
if: true && !contains(github.event.inputs.skiptests, 'redis')
|
||||
run: ./runtest --valgrind --no-latency --verbose --clients 1 --timeout 2400 --dump-logs ${{github.event.inputs.test_args}}
|
||||
@@ -521,7 +519,7 @@ jobs:
|
||||
- name: testprep
|
||||
run: |
|
||||
sudo apt-get update
|
||||
sudo apt-get install tcl8.6 tclx valgrind -y
|
||||
sudo apt-get install tcl8.6 tclx valgrind g++ -y
|
||||
- name: test
|
||||
if: true && !contains(github.event.inputs.skiptests, 'redis')
|
||||
run: ./runtest --valgrind --no-latency --verbose --clients 1 --timeout 2400 --dump-logs ${{github.event.inputs.test_args}}
|
||||
@@ -634,7 +632,7 @@ jobs:
|
||||
repository: ${{ env.GITHUB_REPOSITORY }}
|
||||
ref: ${{ env.GITHUB_HEAD_REF }}
|
||||
- name: make
|
||||
run: make SANITIZER=undefined REDIS_CFLAGS='-DREDIS_TEST -Werror' LUA_DEBUG=yes # we (ab)use this flow to also check Lua C API violations
|
||||
run: make SANITIZER=undefined REDIS_CFLAGS='-DREDIS_TEST -Werror' SKIP_VEC_SETS=yes LUA_DEBUG=yes # we (ab)use this flow to also check Lua C API violations
|
||||
- name: testprep
|
||||
run: |
|
||||
sudo apt-get update
|
||||
@@ -678,7 +676,7 @@ jobs:
|
||||
ref: ${{ env.GITHUB_HEAD_REF }}
|
||||
- name: make
|
||||
run: |
|
||||
dnf -y install which gcc make
|
||||
dnf -y install which gcc make g++
|
||||
make REDIS_CFLAGS='-Werror'
|
||||
- name: testprep
|
||||
run: |
|
||||
@@ -720,7 +718,7 @@ jobs:
|
||||
ref: ${{ env.GITHUB_HEAD_REF }}
|
||||
- name: make
|
||||
run: |
|
||||
dnf -y install which gcc make openssl-devel openssl
|
||||
dnf -y install which gcc make openssl-devel openssl g++
|
||||
make BUILD_TLS=module REDIS_CFLAGS='-Werror'
|
||||
- name: testprep
|
||||
run: |
|
||||
@@ -767,7 +765,7 @@ jobs:
|
||||
ref: ${{ env.GITHUB_HEAD_REF }}
|
||||
- name: make
|
||||
run: |
|
||||
dnf -y install which gcc make openssl-devel openssl
|
||||
dnf -y install which gcc make openssl-devel openssl g++
|
||||
make BUILD_TLS=module REDIS_CFLAGS='-Werror'
|
||||
- name: testprep
|
||||
run: |
|
||||
@@ -875,7 +873,7 @@ jobs:
|
||||
build-macos:
|
||||
strategy:
|
||||
matrix:
|
||||
os: [macos-12, macos-14]
|
||||
os: [macos-13, macos-15]
|
||||
runs-on: ${{ matrix.os }}
|
||||
if: |
|
||||
(github.event_name == 'workflow_dispatch' || (github.event_name != 'workflow_dispatch' && github.repository == 'redis/redis')) &&
|
||||
@@ -902,11 +900,14 @@ jobs:
|
||||
run: make REDIS_CFLAGS='-Werror -DREDIS_TEST'
|
||||
|
||||
test-freebsd:
|
||||
runs-on: macos-12
|
||||
runs-on: macos-13
|
||||
if: |
|
||||
(github.event_name == 'workflow_dispatch' || (github.event_name != 'workflow_dispatch' && github.repository == 'redis/redis')) &&
|
||||
!contains(github.event.inputs.skipjobs, 'freebsd')
|
||||
timeout-minutes: 14400
|
||||
env:
|
||||
CC: clang
|
||||
CXX: clang++
|
||||
steps:
|
||||
- name: prep
|
||||
if: github.event_name == 'workflow_dispatch'
|
||||
@@ -925,7 +926,7 @@ jobs:
|
||||
version: 13.2
|
||||
shell: bash
|
||||
run: |
|
||||
sudo pkg install -y bash gmake lang/tcl86 lang/tclx
|
||||
sudo pkg install -y bash gmake lang/tcl86 lang/tclx gcc
|
||||
gmake
|
||||
./runtest --single unit/keyspace --single unit/auth --single unit/networking --single unit/protocol
|
||||
|
||||
@@ -1080,8 +1081,9 @@ jobs:
|
||||
apt-key adv --keyserver keyserver.ubuntu.com --recv-keys 40976EAF437D05B5
|
||||
apt-key adv --keyserver keyserver.ubuntu.com --recv-keys 3B4FE6ACC0B21F32
|
||||
apt-get update
|
||||
apt-get install -y make gcc-4.8
|
||||
apt-get install -y make gcc-4.8 g++-4.8
|
||||
update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-4.8 100
|
||||
update-alternatives --install /usr/bin/g++ g++ /usr/bin/g++-4.8 100
|
||||
make CC=gcc REDIS_CFLAGS='-Werror'
|
||||
- name: testprep
|
||||
run: apt-get install -y tcl tcltls tclx
|
||||
@@ -1128,9 +1130,10 @@ jobs:
|
||||
apt-key adv --keyserver keyserver.ubuntu.com --recv-keys 40976EAF437D05B5
|
||||
apt-key adv --keyserver keyserver.ubuntu.com --recv-keys 3B4FE6ACC0B21F32
|
||||
apt-get update
|
||||
apt-get install -y make gcc-4.8 openssl libssl-dev
|
||||
apt-get install -y make gcc-4.8 g++-4.8 openssl libssl-dev
|
||||
update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-4.8 100
|
||||
make CC=gcc BUILD_TLS=module REDIS_CFLAGS='-Werror'
|
||||
update-alternatives --install /usr/bin/g++ g++ /usr/bin/g++-4.8 100
|
||||
make CC=gcc CXX=g++ BUILD_TLS=module REDIS_CFLAGS='-Werror'
|
||||
- name: testprep
|
||||
run: |
|
||||
apt-get install -y tcl tcltls tclx
|
||||
@@ -1182,8 +1185,9 @@ jobs:
|
||||
apt-key adv --keyserver keyserver.ubuntu.com --recv-keys 40976EAF437D05B5
|
||||
apt-key adv --keyserver keyserver.ubuntu.com --recv-keys 3B4FE6ACC0B21F32
|
||||
apt-get update
|
||||
apt-get install -y make gcc-4.8 openssl libssl-dev
|
||||
apt-get install -y make gcc-4.8 g++-4.8 openssl libssl-dev
|
||||
update-alternatives --install /usr/bin/gcc gcc /usr/bin/gcc-4.8 100
|
||||
update-alternatives --install /usr/bin/g++ g++ /usr/bin/g++-4.8 100
|
||||
make BUILD_TLS=module CC=gcc REDIS_CFLAGS='-Werror'
|
||||
- name: testprep
|
||||
run: |
|
||||
|
||||
@@ -27,7 +27,7 @@ jobs:
|
||||
--tags -slow
|
||||
- name: Archive redis log
|
||||
if: ${{ failure() }}
|
||||
uses: actions/upload-artifact@v3
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: test-external-redis-log
|
||||
path: external-redis.log
|
||||
@@ -55,7 +55,7 @@ jobs:
|
||||
--tags -slow
|
||||
- name: Archive redis log
|
||||
if: ${{ failure() }}
|
||||
uses: actions/upload-artifact@v3
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: test-external-cluster-log
|
||||
path: external-redis-cluster.log
|
||||
@@ -79,7 +79,7 @@ jobs:
|
||||
--tags "-slow -needs:debug"
|
||||
- name: Archive redis log
|
||||
if: ${{ failure() }}
|
||||
uses: actions/upload-artifact@v3
|
||||
uses: actions/upload-artifact@v4
|
||||
with:
|
||||
name: test-external-redis-nodebug-log
|
||||
path: external-redis-nodebug.log
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
name: redis_docs_sync
|
||||
|
||||
on:
|
||||
release:
|
||||
types: [published]
|
||||
|
||||
jobs:
|
||||
redis_docs_sync:
|
||||
if: github.repository == 'redis/redis'
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Generate a token
|
||||
id: generate-token
|
||||
uses: actions/create-github-app-token@v1
|
||||
with:
|
||||
app-id: ${{ secrets.DOCS_APP_ID }}
|
||||
private-key: ${{ secrets.DOCS_APP_PRIVATE_KEY }}
|
||||
|
||||
- name: Invoke workflow on redis/docs
|
||||
env:
|
||||
GH_TOKEN: ${{ steps.generate-token.outputs.token }}
|
||||
RELEASE_NAME: ${{ github.event.release.tag_name }}
|
||||
run: |
|
||||
LATEST_RELEASE=$(
|
||||
curl -Ls \
|
||||
-H "Accept: application/vnd.github+json" \
|
||||
-H "Authorization: Bearer ${GH_TOKEN}" \
|
||||
-H "X-GitHub-Api-Version: 2022-11-28" \
|
||||
https://api.github.com/repos/redis/redis/releases/latest \
|
||||
| jq -r '.tag_name'
|
||||
)
|
||||
|
||||
if [[ "${LATEST_RELEASE}" == "${RELEASE_NAME}" ]]; then
|
||||
gh workflow run -R redis/docs redis_docs_sync.yaml -f release="${RELEASE_NAME}"
|
||||
fi
|
||||
@@ -30,6 +30,7 @@ deps/lua/src/luac
|
||||
deps/lua/src/liblua.a
|
||||
deps/hdr_histogram/libhdrhistogram.a
|
||||
deps/fpconv/libfpconv.a
|
||||
deps/fast_float/libfast_float.a
|
||||
tests/tls/*
|
||||
.make-*
|
||||
.prerequisites
|
||||
|
||||
+293
-11
@@ -1,16 +1,298 @@
|
||||
Hello! This file is just a placeholder, since this is the "unstable" branch
|
||||
of Redis, the place where all the development happens.
|
||||
Redis Community Edition 8.0 release notes
|
||||
=========================================
|
||||
|
||||
There is no release notes for this branch, it gets forked into another branch
|
||||
every time there is a partial feature freeze in order to eventually create
|
||||
a new stable release.
|
||||
==========================================================
|
||||
8.0-RC1 (v7.9.240) Released Mon 7 Apr 2025 10:00:00 IST
|
||||
==========================================================
|
||||
|
||||
Usually "unstable" is stable enough for you to use it in development environments
|
||||
however you should never use it in production environments. It is possible
|
||||
to download the latest stable release here:
|
||||
This is the first Release Candidate of Redis Community Edition 8.0.
|
||||
|
||||
https://download.redis.io/redis-stable.tar.gz
|
||||
Release Candidates are feature-complete pre-releases. Pre-releases are not suitable for production use.
|
||||
|
||||
More information is available at https://redis.io
|
||||
|
||||
Happy hacking!
|
||||
### Headlines
|
||||
|
||||
8.0-RC1 includes a new beta data structure - vector set.
|
||||
|
||||
### Distributions
|
||||
|
||||
- Alpine and Debian Docker images - https://hub.docker.com/_/redis
|
||||
- Install using snap - see https://github.com/redis/redis-snap
|
||||
- Install using brew - see https://github.com/redis/homebrew-redis
|
||||
- Install using RPM and Debian APT - will be added on the GA release
|
||||
|
||||
### New Features
|
||||
|
||||
- #13915 Vector set - a new data structure [beta]:
|
||||
Vector set extends the concept of sorted sets to allow the storage and querying of
|
||||
high-dimensional vector embeddings, enhancing Redis for AI use cases that involve
|
||||
semantic search and recommendation systems. Vector sets complement the existing
|
||||
vector search capability in the Redis Query Engine. The vector set data type is
|
||||
available in beta. We may change, or even break, the features and the API in
|
||||
future versions. We are open to your feedback as you try out this new data type.
|
||||
- #13846 Allow detecting incompatibility risks before switching to cluster mode
|
||||
|
||||
### Bug fixes
|
||||
|
||||
- #13895 RDB Channel replication - replica is online after BGSAVE is done
|
||||
- #13877 Inconsistency for ShardID in case both master and replica support it
|
||||
- #13883 Defrag scan may return nothing when type/encoding changes during it
|
||||
- #13863 `RANDOMKEY` - infinite loop during client pause
|
||||
- #13853 `SLAVEOF` - crash when clients are blocked on lazy free
|
||||
- #13632 `XREAD` returns nil while stream is not empty
|
||||
|
||||
### Metrics
|
||||
|
||||
- #13846 `INFO`: `cluster_incompatible_ops` - number of cluster-incompatible commands
|
||||
|
||||
### Configuration parameters
|
||||
|
||||
- #13846 `cluster-compatibility-sample-ratio` - sampling ratio (0-100) for checking command compatibility with cluster mode
|
||||
|
||||
|
||||
============================================================
|
||||
8.0-M04 (v7.9.227) Committed Sun 16 Mar 2025 11:00:00 IST
|
||||
============================================================
|
||||
|
||||
This is the fourth Milestone of Redis Community Edition 8.0.
|
||||
|
||||
Milestones are non-feature-complete pre-releases. Pre-releases are not suitable for production use.
|
||||
Once we reach feature-completeness we will release RC1.
|
||||
|
||||
### Headlines
|
||||
|
||||
8.0-M04 includes 3 new hash commands, performance improvements, and memory defragmentation improvements.
|
||||
|
||||
### Distributions
|
||||
|
||||
- Alpine and Debian Docker images - https://hub.docker.com/_/redis
|
||||
- Install using snap - see https://github.com/redis/redis-snap
|
||||
- Install using brew - see https://github.com/redis/homebrew-redis
|
||||
- Install using RPM and Debian APT - will be added on the GA release
|
||||
|
||||
### New Features
|
||||
|
||||
- #13798 Hash - new commands:
|
||||
- `HGETDEL` Get and delete the value of one or more fields of a given hash key
|
||||
- `HGETEX` Get the value of one or more fields of a given hash key, and optionally set their expiration
|
||||
- `HSETEX` Set the value of one or more fields of a given hash key, and optionally set their expiration
|
||||
- #13773 Add replication offset to AOF, allowing more robust way to determine which AOF has a more up-to-date data during recovery
|
||||
- #13740, #13763 shared secret - new mechanism to allow sending internal commands between nodes
|
||||
|
||||
### Bug fixes
|
||||
|
||||
- #13804 Overflow on 32-bit systems when calculating idle time for eviction
|
||||
- #13793 `WAITAOF` returns prematurely
|
||||
- #13800 Remove `DENYOOM` from `HEXPIRE`, `HEXPIREAT`, `HPEXPIRE`, and `HPEXPIREAT`
|
||||
- #13632 Streams - wrong behavior of `XREAD +` after last entry
|
||||
|
||||
### Modules API
|
||||
|
||||
- #13788 `RedisModule_LoadDefaultConfigs` - load module configuration values from redis.conf
|
||||
- #13815 `RM_RegisterDefragFunc2` - support for incremental defragmentation of global module data
|
||||
- #13816 `RM_DefragRedisModuleDict` - allow modules to defrag `RedisModuleDict`
|
||||
- #13774 `RM_GetContextFlags` - add a `REDISMODULE_CTX_FLAGS_DEBUG_ENABLED` flag to execute debug commands
|
||||
|
||||
|
||||
### Performance and resource utilization improvements
|
||||
|
||||
- #13752 Reduce defrag CPU usage when defragmentation is ineffective
|
||||
- #13764 Reduce latency when a command is called consecutively
|
||||
- #13787 Optimize parsing data from clients, specifically multi-bulk (array) data
|
||||
- #13792 Optimize dictionary lookup by avoiding duplicate key length calculation during comparisons
|
||||
- #13796 Optimize expiration checks
|
||||
|
||||
|
||||
============================================================
|
||||
8.0-M03 (v7.9.226) Committed Mon 20 Jan 2025 15:00:00 IST
|
||||
============================================================
|
||||
|
||||
This is the third Milestone of Redis Community Edition 8.0.
|
||||
|
||||
Milestones are non-feature-complete pre-releases. Pre-releases are not suitable for production use.
|
||||
Once we reach feature-completeness we will release RC1.
|
||||
|
||||
### Headlines:
|
||||
|
||||
8.0-M03 introduces an improved replication mechanism which is more performant and robust, a new I/O threading implementation which enables throughput increase on multi-core environments, and many additional performance improvements. Both Alpine and Debian Docker images are now available on [Docker Hub](https://hub.docker.com/_/redis). A snap and Homebrew distributions are available as well.
|
||||
|
||||
|
||||
### Security fixes
|
||||
|
||||
- (CVE-2024-46981) Lua script may lead to remote code execution
|
||||
- (CVE-2024-51741) Denial-of-service due to malformed ACL selectors
|
||||
|
||||
### New Features
|
||||
|
||||
- #13695 New I/O threading implementation
|
||||
- #13732 New replication mechanism
|
||||
|
||||
|
||||
### Bug fixes
|
||||
|
||||
- #13653 `MODULE LOADEX` - crash on nonexistent parameter name
|
||||
- #13661 `FUNCTION FLUSH` - memory leak when using jemalloc
|
||||
- #13626 Memory leak on failed RDB loading
|
||||
|
||||
### Other general improvements
|
||||
|
||||
- #13639 When `hide-user-data-from-log` is enabled - also print command tokens on crash
|
||||
- #13660 Add the Lua VM memory to memory overhead
|
||||
|
||||
### New metrics
|
||||
|
||||
- #13592 `INFO` - new `KEYSIZES` section includes key size distributions for basic data types
|
||||
- #13695 `INFO` - new `Threads` section includes I/O threading metrics
|
||||
|
||||
### Modules API
|
||||
|
||||
- #13666 `RedisModule_ACLCheckKeyPrefixPermissions` - check access permissions to any key matching a given prefix
|
||||
- #13676 `RedisModule_HashFieldMinExpire` - query the minimum expiration time over all the hash’s fields
|
||||
- #13676 `RedisModule_HashGet` - new `REDISMODULE_HASH_EXPIRE_TIME` flag - query the field expiration time
|
||||
- #13656 `RedisModule_RegisterXXXConfig` - allow registering unprefixed configuration parameters
|
||||
|
||||
### Configuration parameters
|
||||
|
||||
|
||||
- `replica-full-sync-buffer-limit` - maximum size of accumulated replication stream data on the replica side
|
||||
- `io-threads-do-reads` is no longer effective. The new I/O threading implementation always use threads for both reads and writes
|
||||
|
||||
### Performance and resource utilization improvements
|
||||
|
||||
- #13638 Optimize CRC64 performance
|
||||
- #13521 Optimize commands with large argument count - reuse c->argv after command execution
|
||||
- #13558 Optimize `PFCOUNT` and `PFMERGE` - SIMD acceleration
|
||||
- #13644 Optimize `GET` on high pipeline use-cases
|
||||
- #13646 Optimize `EXISTS` - prefetching and branch prediction hints
|
||||
- #13652 Optimize `LRANGE` - improve listpack handling and decoding efficiency
|
||||
- #13655 Optimize `HSET` - avoid unnecessary hash field creation or deletion
|
||||
- #13721 Optimize `LRANGE` and `HGETALL` - refactor client write preparation and handling
|
||||
|
||||
|
||||
============================================================
|
||||
8.0-M02 (v7.9.225) Committed Mon 28 Oct 2024 14:00:00 IST
|
||||
============================================================
|
||||
|
||||
This is the second Milestone of Redis Community Edition 8.0.
|
||||
|
||||
Milestones are non-feature-complete pre-releases. Pre-releases are not suitable for production use.
|
||||
Once we reach feature-completeness we will release RC1.
|
||||
|
||||
### Headlines:
|
||||
|
||||
8.0-M02 introduces significant performance improvements. Both Alpine and Debian Docker images are now available on [Docker Hub](https://hub.docker.com/_/redis). Additional distributions will be introduced in upcoming pre-releases.
|
||||
|
||||
### Supported upgrade paths (by replication or persistence) to 8.0-M02
|
||||
|
||||
- From previous Redis versions, without modules
|
||||
|
||||
The following upgrade paths (by replication or persistence) to 8.0-M02 are not yet tested and will be introduced in upcoming pre-releases:
|
||||
- From previous Redis versions with modules (RediSearch, RedisJSON, RedisTimeSeries, RedisBloom)
|
||||
- From Redis Stack 7.2 or 7.4
|
||||
|
||||
### Security fixes
|
||||
|
||||
- (CVE-2024-31449) Lua library commands may lead to stack overflow and potential RCE.
|
||||
- (CVE-2024-31227) Potential Denial-of-service due to malformed ACL selectors.
|
||||
- (CVE-2024-31228) Potential Denial-of-service due to unbounded pattern matching.
|
||||
|
||||
### Bug fixes
|
||||
|
||||
- #13539 Hash: Fix key ref for a hash that no longer has fields with expiration on `RENAME`/`MOVE`/`SWAPDB`/`RESTORE`
|
||||
- #13512 Fix `TOUCH` command from a script in no-touch mode
|
||||
- #13468 Cluster: Fix cluster node config corruption caused by mixing shard-id and non-shard-id versions
|
||||
- #13608 Cluster: Fix `GET #` option in `SORT` command
|
||||
|
||||
### Modules API
|
||||
|
||||
- #13526 Extend `RedisModule_OpenKey` to read also expired keys and subkeys
|
||||
|
||||
### Performance and resource utilization improvements
|
||||
|
||||
- #11884 Optimize `ZADD` and `ZRANGE*` commands
|
||||
- #13530 Optimize `SSCAN` command in case of listpack or intset encoding
|
||||
- #13531 Optimize `HSCAN`/`ZSCAN` command in case of listpack encoding
|
||||
- #13520 Optimize commands that heavily rely on bulk/mbulk replies (example of `LRANGE`)
|
||||
- #13566 Optimize `ZUNION[STORE]` by avoiding redundant temporary dict usage
|
||||
- #13567 Optimize `SUNION`/`SDIFF` commands by avoiding redundant temporary dict usage
|
||||
- #11533 Avoid redundant `lpGet` to boost `quicklistCompare`
|
||||
- #13412 Reduce redundant call of `prepareClientToWrite` when call `addReply*` continuously
|
||||
|
||||
|
||||
===========================================================
|
||||
8.0-M01 (v7.9.224) Released Thu 12 Sep 2024 10:00:00 IST
|
||||
===========================================================
|
||||
|
||||
This is the first Milestone of Redis Community Edition 8.0.
|
||||
|
||||
Milestones are non-feature-complete pre-releases. Pre-releases are not suitable for production use.
|
||||
Once we reach feature-completeness we will release RC1.
|
||||
|
||||
### Headlines:
|
||||
|
||||
Redis 8.0 introduces new data structures: JSON, time series, and 5 probabilistic data structures (previously available as separate Redis modules) and incorporates the enhanced Redis Query Enginer (with vector search).
|
||||
|
||||
8.0-M01 is available as a Docker image and can be downloaded from [Docker Hub](https://hub.docker.com/_/redis). Additional distributions will be introduced in upcoming pre-releases.
|
||||
|
||||
### Supported upgrade paths (by replication or persistence) to 8.0-M01
|
||||
|
||||
|
||||
- From previous Redis versions, without modules
|
||||
|
||||
The following upgrade paths (by replication or persistence) to 8.0-M01 are not yet tested and will be introduced in upcoming pre-releases:
|
||||
- From previous Redis versions with modules (RediSearch, RedisJSON, RedisTimeSeries, RedisBloom)
|
||||
- From Redis Stack 7.2 or 7.4
|
||||
|
||||
### New Features in binary distributions
|
||||
|
||||
- 7 new data structures: JSON, Time series, Bloom filter, Cuckoo filter, Count-min sketch, Top-k, t-digest
|
||||
- The enhanced Redis Query Engine (with vector search)
|
||||
|
||||
### Potentially breaking changes
|
||||
|
||||
- #12272 `GETRANGE` returns an empty bulk when the negative end index is out of range
|
||||
- #12395 Optimize `SCAN` command when matching data type
|
||||
|
||||
### Bug fixes
|
||||
|
||||
- #13510 Fix `RM_RdbLoad` to enable AOF after RDB loading is completed
|
||||
- #13489 `ACL CAT` - return module commands
|
||||
- #13476 Fix a race condition in the `cache_memory` of `functionsLibCtx`
|
||||
- #13473 Fix incorrect lag due to trimming stream via `XTRIM` command
|
||||
- #13338 Fix incorrect lag field in `XINFO` when tombstone is after the `last_id` of the consume group
|
||||
- #13470 On `HDEL` of last field - update the global hash field expiration data structure
|
||||
- #13465 Cluster: Pass extensions to node if extension processing is handled by it
|
||||
- #13443 Cluster: Ensure validity of myself when loading cluster config
|
||||
- #13422 Cluster: Fix `CLUSTER SHARDS` command returns empty array
|
||||
|
||||
### Modules API
|
||||
|
||||
- #13509 New API calls: `RM_DefragAllocRaw`, `RM_DefragFreeRaw`, and `RM_RegisterDefragCallbacks` - defrag API to allocate and free raw memory
|
||||
|
||||
### Performance and resource utilization improvements
|
||||
|
||||
- #13503 Avoid overhead of comparison function pointer calls in listpack `lpFind`
|
||||
- #13505 Optimize `STRING` datatype write commands
|
||||
- #13499 Optimize `SMEMBERS` command
|
||||
- #13494 Optimize `GEO*` commands reply
|
||||
- #13490 Optimize `HELLO` command
|
||||
- #13488 Optimize client query buffer
|
||||
- #12395 Optimize `SCAN` command when matching data type
|
||||
- #13529 Optimize `LREM`, `LPOS`, `LINSERT`, and `LINDEX` commands
|
||||
- #13516 Optimize `LRANGE` and other commands that perform several writes to client buffers per call
|
||||
- #13431 Avoid `used_memory` contention when updating from multiple threads
|
||||
|
||||
### Other general improvements
|
||||
|
||||
- #13495 Reply `-LOADING` on replica while flushing the db
|
||||
|
||||
### CLI tools
|
||||
|
||||
- #13411 redis-cli: Fix wrong `dbnum` showed after the client reconnected
|
||||
|
||||
### Notes
|
||||
|
||||
- No backward compatibility for replication or persistence.
|
||||
- Additional distributions, upgrade paths, features, and improvements will be introduced in upcoming pre-releases.
|
||||
- With the GA release of 8.0 we will deprecate Redis Stack.
|
||||
|
||||
|
||||
@@ -1,11 +1,16 @@
|
||||
# Top level makefile, the real shit is at src/Makefile
|
||||
# Top level makefile, the real stuff is at ./src/Makefile and in ./modules/Makefile
|
||||
|
||||
SUBDIRS = src
|
||||
ifeq ($(BUILD_WITH_MODULES), yes)
|
||||
SUBDIRS += modules
|
||||
endif
|
||||
|
||||
default: all
|
||||
|
||||
.DEFAULT:
|
||||
cd src && $(MAKE) $@
|
||||
for dir in $(SUBDIRS); do $(MAKE) -C $$dir $@; done
|
||||
|
||||
install:
|
||||
cd src && $(MAKE) $@
|
||||
for dir in $(SUBDIRS); do $(MAKE) -C $$dir $@; done
|
||||
|
||||
.PHONY: install
|
||||
|
||||
@@ -1,3 +1,5 @@
|
||||
[](https://codecov.io/github/redis/redis)
|
||||
|
||||
This README is just a fast *quick start* document. You can find more detailed documentation at [redis.io](https://redis.io).
|
||||
|
||||
What is Redis?
|
||||
@@ -15,8 +17,8 @@ Another good example is to think of Redis as a more complex version of memcached
|
||||
|
||||
If you want to know more, this is a list of selected starting points:
|
||||
|
||||
* Introduction to Redis data types. https://redis.io/topics/data-types-intro
|
||||
* Try Redis directly inside your browser. https://try.redis.io
|
||||
* Introduction to Redis data types. https://redis.io/docs/latest/develop/data-types/
|
||||
|
||||
* The full list of Redis commands. https://redis.io/commands
|
||||
* There is much more inside the official Redis documentation. https://redis.io/documentation
|
||||
|
||||
@@ -494,7 +496,7 @@ Other C files
|
||||
* `dict.c` is an implementation of a non-blocking hash table which rehashes incrementally.
|
||||
* `cluster.c` implements the Redis Cluster. Probably a good read only after being very familiar with the rest of the Redis code base. If you want to read `cluster.c` make sure to read the [Redis Cluster specification][4].
|
||||
|
||||
[4]: https://redis.io/topics/cluster-spec
|
||||
[4]: https://redis.io/docs/latest/operate/oss_and_stack/reference/cluster-spec/
|
||||
|
||||
Anatomy of a Redis command
|
||||
---
|
||||
|
||||
+19
@@ -0,0 +1,19 @@
|
||||
coverage:
|
||||
status:
|
||||
patch:
|
||||
default:
|
||||
informational: true
|
||||
project:
|
||||
default:
|
||||
informational: true
|
||||
|
||||
comment:
|
||||
require_changes: false
|
||||
require_head: false
|
||||
require_base: false
|
||||
layout: "condensed_header, diff, files"
|
||||
hide_project_coverage: false
|
||||
behavior: default
|
||||
|
||||
github_checks:
|
||||
annotations: false
|
||||
Vendored
+7
@@ -42,6 +42,7 @@ distclean:
|
||||
-(cd jemalloc && [ -f Makefile ] && $(MAKE) distclean) > /dev/null || true
|
||||
-(cd hdr_histogram && $(MAKE) clean) > /dev/null || true
|
||||
-(cd fpconv && $(MAKE) clean) > /dev/null || true
|
||||
-(cd fast_float && $(MAKE) clean) > /dev/null || true
|
||||
-(rm -f .make-*)
|
||||
|
||||
.PHONY: distclean
|
||||
@@ -74,6 +75,12 @@ fpconv: .make-prerequisites
|
||||
|
||||
.PHONY: fpconv
|
||||
|
||||
fast_float: .make-prerequisites
|
||||
@printf '%b %b\n' $(MAKECOLOR)MAKE$(ENDCOLOR) $(BINCOLOR)$@$(ENDCOLOR)
|
||||
cd fast_float && $(MAKE) libfast_float
|
||||
|
||||
.PHONY: fast_float
|
||||
|
||||
ifeq ($(uname_S),SunOS)
|
||||
# Make isinf() available
|
||||
LUA_CFLAGS= -D__C99FEATURES__=1
|
||||
|
||||
Vendored
+24
@@ -0,0 +1,24 @@
|
||||
# Fallback to gcc/g++ when $CC or $CXX is not in $PATH.
|
||||
CC ?= gcc
|
||||
CXX ?= g++
|
||||
|
||||
CFLAGS=-Wall -O3
|
||||
# This avoids loosing the fastfloat specific compile flags when we override the CFLAGS via the main project
|
||||
FASTFLOAT_CFLAGS=-std=c++11 -DFASTFLOAT_ALLOWS_LEADING_PLUS
|
||||
LDFLAGS=
|
||||
|
||||
libfast_float: fast_float_strtod.o
|
||||
$(AR) -r libfast_float.a fast_float_strtod.o
|
||||
|
||||
32bit: CFLAGS += -m32
|
||||
32bit: LDFLAGS += -m32
|
||||
32bit: libfast_float
|
||||
|
||||
fast_float_strtod.o: fast_float_strtod.cpp
|
||||
$(CXX) $(CFLAGS) $(FASTFLOAT_CFLAGS) -c fast_float_strtod.cpp $(LDFLAGS)
|
||||
|
||||
clean:
|
||||
rm -f *.o
|
||||
rm -f *.a
|
||||
rm -f *.h.gch
|
||||
rm -rf *.dSYM
|
||||
Vendored
+21
@@ -0,0 +1,21 @@
|
||||
README for fast_float v6.1.4
|
||||
|
||||
----------------------------------------------
|
||||
|
||||
We're using the fast_float library[1] in our (compiled-in)
|
||||
floating-point fast_float_strtod implementation for faster and more
|
||||
portable parsing of 64 decimal strings.
|
||||
|
||||
The single file fast_float.h is an amalgamation of the entire library,
|
||||
which can be (re)generated with the amalgamate.py script (from the
|
||||
fast_float repository) via the command
|
||||
|
||||
```
|
||||
git clone https://github.com/fastfloat/fast_float
|
||||
cd fast_float
|
||||
git checkout v6.1.4
|
||||
python3 ./script/amalgamate.py --license=MIT \
|
||||
> $REDIS_SRC/deps/fast_float/fast_float.h
|
||||
```
|
||||
|
||||
[1]: https://github.com/fastfloat/fast_float
|
||||
Vendored
+3838
File diff suppressed because it is too large
Load Diff
+32
@@ -0,0 +1,32 @@
|
||||
#include "fast_float.h"
|
||||
#include <iostream>
|
||||
#include <string>
|
||||
#include <system_error>
|
||||
#include <cerrno>
|
||||
|
||||
/* Convert NPTR to a double using the fast_float library.
|
||||
*
|
||||
* This function behaves similarly to the standard strtod function, converting
|
||||
* the initial portion of the string pointed to by `nptr` to a `double` value,
|
||||
* using the fast_float library for high performance. If the conversion fails,
|
||||
* errno is set to EINVAL error code.
|
||||
*
|
||||
* @param nptr A pointer to the null-terminated byte string to be interpreted.
|
||||
* @param endptr A pointer to a pointer to character. If `endptr` is not NULL,
|
||||
* it will point to the character after the last character used
|
||||
* in the conversion.
|
||||
* @return The converted value as a double. If no valid conversion could
|
||||
* be performed, returns 0.0.
|
||||
* If ENDPTR is not NULL, a pointer to the character after the last one used
|
||||
* in the number is put in *ENDPTR. */
|
||||
extern "C" double fast_float_strtod(const char *nptr, char **endptr) {
|
||||
double result = 0.0;
|
||||
auto answer = fast_float::from_chars(nptr, nptr + strlen(nptr), result);
|
||||
if (answer.ec != std::errc()) {
|
||||
errno = EINVAL; // Fallback to for other errors
|
||||
}
|
||||
if (endptr != NULL) {
|
||||
*endptr = (char *)answer.ptr;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
Vendored
+15
@@ -0,0 +1,15 @@
|
||||
|
||||
#ifndef __FAST_FLOAT_STRTOD_H__
|
||||
#define __FAST_FLOAT_STRTOD_H__
|
||||
|
||||
#if defined(__cplusplus)
|
||||
extern "C"
|
||||
{
|
||||
#endif
|
||||
double fast_float_strtod(const char *in, char **out);
|
||||
|
||||
#if defined(__cplusplus)
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif /* __FAST_FLOAT_STRTOD_H__ */
|
||||
Vendored
+1
-1
@@ -478,7 +478,7 @@ static int __redisGetSubscribeCallback(redisAsyncContext *ac, redisReply *reply,
|
||||
|
||||
/* Match reply with the expected format of a pushed message.
|
||||
* The type and number of elements (3 to 4) are specified at:
|
||||
* https://redis.io/topics/pubsub#format-of-pushed-messages */
|
||||
* https://redis.io/docs/latest/develop/interact/pubsub/#format-of-pushed-messages */
|
||||
if ((reply->type == REDIS_REPLY_ARRAY && !(c->flags & REDIS_SUPPORTS_PUSH) && reply->elements >= 3) ||
|
||||
reply->type == REDIS_REPLY_PUSH) {
|
||||
assert(reply->element[0]->type == REDIS_REPLY_STRING);
|
||||
|
||||
Vendored
+1
-1
@@ -296,7 +296,7 @@ static int isUnsupportedTerm(void) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Raw mode: 1960 magic shit. */
|
||||
/* Raw mode: 1960's magic. */
|
||||
static int enableRawMode(int fd) {
|
||||
if (getenv("FAKETTY_WITH_PROMPT") != NULL) {
|
||||
return 0;
|
||||
|
||||
Vendored
+1
@@ -132,6 +132,7 @@ static int bit_tohex(lua_State *L)
|
||||
const char *hexdigits = "0123456789abcdef";
|
||||
char buf[8];
|
||||
int i;
|
||||
if (n == INT32_MIN) n = INT32_MIN+1;
|
||||
if (n < 0) { n = -n; hexdigits = "0123456789ABCDEF"; }
|
||||
if (n > 8) n = 8;
|
||||
for (i = (int)n; --i >= 0; ) { buf[i] = hexdigits[b & 15]; b >>= 4; }
|
||||
|
||||
@@ -0,0 +1,90 @@
|
||||
|
||||
SUBDIRS = redisjson redistimeseries redisbloom redisearch
|
||||
|
||||
define submake
|
||||
for dir in $(SUBDIRS); do $(MAKE) -C $$dir $(1); done
|
||||
endef
|
||||
|
||||
all: prepare_source
|
||||
$(call submake,$@)
|
||||
|
||||
get_source:
|
||||
$(call submake,$@)
|
||||
|
||||
prepare_source: get_source handle-werrors setup_environment
|
||||
|
||||
clean:
|
||||
$(call submake,$@)
|
||||
|
||||
distclean: clean_environment
|
||||
$(call submake,$@)
|
||||
|
||||
pristine:
|
||||
$(call submake,$@)
|
||||
|
||||
install:
|
||||
$(call submake,$@)
|
||||
|
||||
setup_environment: install-rust handle-werrors
|
||||
|
||||
clean_environment: uninstall-rust
|
||||
|
||||
# Keep all of the Rust stuff in one place
|
||||
install-rust:
|
||||
ifeq ($(INSTALL_RUST_TOOLCHAIN),yes)
|
||||
@RUST_VERSION=1.80.1; \
|
||||
ARCH="$$(uname -m)"; \
|
||||
if ldd --version 2>&1 | grep -q musl; then LIBC_TYPE="musl"; else LIBC_TYPE="gnu"; fi; \
|
||||
echo "Detected architecture: $${ARCH} and libc: $${LIBC_TYPE}"; \
|
||||
case "$${ARCH}" in \
|
||||
'x86_64') \
|
||||
if [ "$${LIBC_TYPE}" = "musl" ]; then \
|
||||
RUST_INSTALLER="rust-$${RUST_VERSION}-x86_64-unknown-linux-musl"; \
|
||||
RUST_SHA256="37bbec6a7b9f55fef79c451260766d281a7a5b9d2e65c348bbc241127cf34c8d"; \
|
||||
else \
|
||||
RUST_INSTALLER="rust-$${RUST_VERSION}-x86_64-unknown-linux-gnu"; \
|
||||
RUST_SHA256="85e936d5d36970afb80756fa122edcc99bd72a88155f6bdd514f5d27e778e00a"; \
|
||||
fi ;; \
|
||||
'aarch64') \
|
||||
if [ "$${LIBC_TYPE}" = "musl" ]; then \
|
||||
RUST_INSTALLER="rust-$${RUST_VERSION}-aarch64-unknown-linux-musl"; \
|
||||
RUST_SHA256="dd668c2d82f77c5458deb023932600fae633fff8d7f876330e01bc47e9976d17"; \
|
||||
else \
|
||||
RUST_INSTALLER="rust-$${RUST_VERSION}-aarch64-unknown-linux-gnu"; \
|
||||
RUST_SHA256="2e89bad7857711a1c11d017ea28fbfeec54076317763901194f8f5decbac1850"; \
|
||||
fi ;; \
|
||||
*) echo >&2 "Unsupported architecture: '$${ARCH}'"; exit 1 ;; \
|
||||
esac; \
|
||||
echo "Downloading and installing Rust standalone installer: $${RUST_INSTALLER}"; \
|
||||
wget --quiet -O $${RUST_INSTALLER}.tar.xz https://static.rust-lang.org/dist/$${RUST_INSTALLER}.tar.xz; \
|
||||
echo "$${RUST_SHA256} $${RUST_INSTALLER}.tar.xz" | sha256sum -c --quiet || { echo "Rust standalone installer checksum failed!"; exit 1; }; \
|
||||
tar -xf $${RUST_INSTALLER}.tar.xz; \
|
||||
(cd $${RUST_INSTALLER} && ./install.sh); \
|
||||
rm -rf $${RUST_INSTALLER}
|
||||
endif
|
||||
|
||||
uninstall-rust:
|
||||
ifeq ($(INSTALL_RUST_TOOLCHAIN),yes)
|
||||
@if [ -x "/usr/local/lib/rustlib/uninstall.sh" ]; then \
|
||||
echo "Uninstalling Rust using uninstall.sh script"; \
|
||||
rm -rf ~/.cargo; \
|
||||
/usr/local/lib/rustlib/uninstall.sh; \
|
||||
else \
|
||||
echo "WARNING: Rust toolchain not found or uninstall script is missing."; \
|
||||
fi
|
||||
endif
|
||||
|
||||
handle-werrors: get_source
|
||||
ifeq ($(DISABLE_WERRORS),yes)
|
||||
@echo "Disabling -Werror for all modules"
|
||||
@for dir in $(SUBDIRS); do \
|
||||
echo "Processing $$dir"; \
|
||||
find $$dir/src -type f \
|
||||
\( -name "Makefile" \
|
||||
-o -name "*.mk" \
|
||||
-o -name "CMakeLists.txt" \) \
|
||||
-exec sed -i 's/-Werror//g' {} +; \
|
||||
done
|
||||
endif
|
||||
|
||||
.PHONY: all clean distclean install $(SUBDIRS) setup_environment clean_environment install-rust uninstall-rust handle-werrors
|
||||
@@ -0,0 +1,51 @@
|
||||
PREFIX ?= /usr/local
|
||||
INSTALL_DIR ?= $(DESTDIR)$(PREFIX)/lib/redis/modules
|
||||
INSTALL ?= install
|
||||
|
||||
# This logic *partially* follows the current module build system. It is a bit awkward and
|
||||
# should be changed if/when the modules' build process is refactored.
|
||||
|
||||
ARCH_MAP_x86_64 := x64
|
||||
ARCH_MAP_i386 := x86
|
||||
ARCH_MAP_i686 := x86
|
||||
ARCH_MAP_aarch64 := arm64v8
|
||||
ARCH_MAP_arm64 := arm64v8
|
||||
|
||||
OS := $(shell uname -s | tr '[:upper:]' '[:lower:]')
|
||||
ARCH := $(ARCH_MAP_$(shell uname -m))
|
||||
ifeq ($(ARCH),)
|
||||
$(error Unrecognized CPU architecture $(shell uname -m))
|
||||
endif
|
||||
|
||||
FULL_VARIANT := $(OS)-$(ARCH)-release
|
||||
|
||||
# Common rules for all modules, based on per-module configuration
|
||||
|
||||
all: $(TARGET_MODULE)
|
||||
|
||||
$(TARGET_MODULE): get_source
|
||||
$(MAKE) -C $(SRC_DIR)
|
||||
cp ${TARGET_MODULE} ./
|
||||
|
||||
get_source: $(SRC_DIR)/.prepared
|
||||
|
||||
$(SRC_DIR)/.prepared:
|
||||
mkdir -p $(SRC_DIR)
|
||||
git clone --recursive --depth 1 --branch $(MODULE_VERSION) $(MODULE_REPO) $(SRC_DIR)
|
||||
touch $@
|
||||
|
||||
clean:
|
||||
-$(MAKE) -C $(SRC_DIR) clean
|
||||
-rm -f ./*.so
|
||||
|
||||
distclean: clean
|
||||
-$(MAKE) -C $(SRC_DIR) distclean
|
||||
|
||||
pristine:
|
||||
-rm -rf $(SRC_DIR)
|
||||
|
||||
install: $(TARGET_MODULE)
|
||||
mkdir -p $(INSTALL_DIR)
|
||||
$(INSTALL) -m 0755 -D $(TARGET_MODULE) $(INSTALL_DIR)
|
||||
|
||||
.PHONY: all clean distclean pristine install
|
||||
@@ -0,0 +1,6 @@
|
||||
SRC_DIR = src
|
||||
MODULE_VERSION = v7.99.90
|
||||
MODULE_REPO = https://github.com/redisbloom/redisbloom
|
||||
TARGET_MODULE = $(SRC_DIR)/bin/$(FULL_VARIANT)/redisbloom.so
|
||||
|
||||
include ../common.mk
|
||||
@@ -0,0 +1,7 @@
|
||||
SRC_DIR = src
|
||||
MODULE_VERSION = v7.99.90
|
||||
MODULE_REPO = https://github.com/redisearch/redisearch
|
||||
TARGET_MODULE = $(SRC_DIR)/bin/$(FULL_VARIANT)/search-community/redisearch.so
|
||||
|
||||
include ../common.mk
|
||||
|
||||
@@ -0,0 +1,11 @@
|
||||
SRC_DIR = src
|
||||
MODULE_VERSION = v7.99.90
|
||||
MODULE_REPO = https://github.com/redisjson/redisjson
|
||||
TARGET_MODULE = $(SRC_DIR)/bin/$(FULL_VARIANT)/rejson.so
|
||||
|
||||
include ../common.mk
|
||||
|
||||
$(SRC_DIR)/.cargo_fetched:
|
||||
cd $(SRC_DIR) && cargo fetch
|
||||
|
||||
get_source: $(SRC_DIR)/.cargo_fetched
|
||||
@@ -0,0 +1,6 @@
|
||||
SRC_DIR = src
|
||||
MODULE_VERSION = v7.99.90
|
||||
MODULE_REPO = https://github.com/redistimeseries/redistimeseries
|
||||
TARGET_MODULE = $(SRC_DIR)/bin/$(FULL_VARIANT)/redistimeseries.so
|
||||
|
||||
include ../common.mk
|
||||
@@ -0,0 +1,11 @@
|
||||
__pycache__
|
||||
misc
|
||||
*.so
|
||||
*.xo
|
||||
*.o
|
||||
.DS_Store
|
||||
w2v
|
||||
word2vec.bin
|
||||
TODO
|
||||
*.txt
|
||||
*.rdb
|
||||
@@ -0,0 +1,84 @@
|
||||
# Compiler settings
|
||||
CC = cc
|
||||
|
||||
ifdef SANITIZER
|
||||
ifeq ($(SANITIZER),address)
|
||||
SAN=-fsanitize=address
|
||||
else
|
||||
ifeq ($(SANITIZER),undefined)
|
||||
SAN=-fsanitize=undefined
|
||||
else
|
||||
ifeq ($(SANITIZER),thread)
|
||||
SAN=-fsanitize=thread
|
||||
else
|
||||
$(error "unknown sanitizer=${SANITIZER}")
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
CFLAGS = -O2 -Wall -Wextra -g $(SAN) -std=c11
|
||||
LDFLAGS = -lm $(SAN)
|
||||
|
||||
# Detect OS
|
||||
uname_S := $(shell sh -c 'uname -s 2>/dev/null || echo not')
|
||||
uname_M := $(shell sh -c 'uname -m 2>/dev/null || echo not')
|
||||
|
||||
# Shared library compile flags for linux / osx
|
||||
ifeq ($(uname_S),Linux)
|
||||
SHOBJ_CFLAGS ?= -W -Wall -fno-common -g -ggdb -std=c11 -O2
|
||||
SHOBJ_LDFLAGS ?= -shared
|
||||
ifneq (,$(findstring armv,$(uname_M)))
|
||||
SHOBJ_LDFLAGS += -latomic
|
||||
endif
|
||||
ifneq (,$(findstring aarch64,$(uname_M)))
|
||||
SHOBJ_LDFLAGS += -latomic
|
||||
endif
|
||||
else
|
||||
SHOBJ_CFLAGS ?= -W -Wall -dynamic -fno-common -g -ggdb -std=c11 -O3
|
||||
SHOBJ_LDFLAGS ?= -bundle -undefined dynamic_lookup
|
||||
endif
|
||||
|
||||
# OS X 11.x doesn't have /usr/lib/libSystem.dylib and needs an explicit setting.
|
||||
ifeq ($(uname_S),Darwin)
|
||||
ifeq ("$(wildcard /usr/lib/libSystem.dylib)","")
|
||||
LIBS = -L /Library/Developer/CommandLineTools/SDKs/MacOSX.sdk/usr/lib -lsystem
|
||||
endif
|
||||
endif
|
||||
|
||||
.SUFFIXES: .c .so .xo .o
|
||||
|
||||
all: vset.so
|
||||
|
||||
.c.xo:
|
||||
$(CC) -I. $(CFLAGS) $(SHOBJ_CFLAGS) -fPIC -c $< -o $@
|
||||
|
||||
vset.xo: ../../src/redismodule.h expr.c
|
||||
|
||||
vset.so: vset.xo hnsw.xo cJSON.xo
|
||||
$(CC) -o $@ $^ $(SHOBJ_LDFLAGS) $(LIBS) $(SAN) -lc
|
||||
|
||||
# Example sources / objects
|
||||
SRCS = hnsw.c w2v.c
|
||||
OBJS = $(SRCS:.c=.o)
|
||||
|
||||
TARGET = w2v
|
||||
MODULE = vset.so
|
||||
|
||||
# Default target
|
||||
all: $(TARGET) $(MODULE)
|
||||
|
||||
# Example linking rule
|
||||
$(TARGET): $(OBJS)
|
||||
$(CC) $(OBJS) $(LDFLAGS) -o $(TARGET)
|
||||
|
||||
# Compilation rule for object files
|
||||
%.o: %.c
|
||||
$(CC) $(CFLAGS) -c $< -o $@
|
||||
|
||||
# Clean rule
|
||||
clean:
|
||||
rm -f $(TARGET) $(OBJS) *.xo *.so
|
||||
|
||||
# Declare phony targets
|
||||
.PHONY: all clean
|
||||
@@ -0,0 +1,633 @@
|
||||
This module implements Vector Sets for Redis, a new Redis data type similar
|
||||
to Sorted Sets but having string elements associated to a vector instead of
|
||||
a score. The fundamental goal of Vector Sets is to make possible adding items,
|
||||
and later get a subset of the added items that are the most similar to a
|
||||
specified vector (often a learned embedding), or the most similar to the vector
|
||||
of an element that is already part of the Vector Set.
|
||||
|
||||
Moreover, Vector sets implement optional filtered search capabilities: it is possible to associate attributes to all or to a subset of elements in the set, and then, using the `FILTER` option of the `VSIM` command, to ask for items similar to a given vector but also passing a filter specified as a simple mathematical expression (Like `".year > 1950"` or similar). This means that **you can have vector similarity and scalar filters at the same time**.
|
||||
|
||||
## Installation
|
||||
|
||||
Build with:
|
||||
|
||||
make
|
||||
|
||||
Then load the module with the following command line, or by inserting the needed directives in the `redis.conf` file.
|
||||
|
||||
./redis-server --loadmodule vset.so
|
||||
|
||||
To run tests, I suggest using this:
|
||||
|
||||
./redis-server --save "" --enable-debug-command yes
|
||||
|
||||
The execute the tests with:
|
||||
|
||||
./test.py
|
||||
|
||||
## Reference of available commands
|
||||
|
||||
**VADD: add items into a vector set**
|
||||
|
||||
VADD key [REDUCE dim] FP32|VALUES vector element [CAS] [NOQUANT | Q8 | BIN]
|
||||
[EF build-exploration-factor] [SETATTR <attributes>] [M <numlinks>]
|
||||
|
||||
Add a new element into the vector set specified by the key.
|
||||
The vector can be provided as FP32 blob of values, or as floating point
|
||||
numbers as strings, prefixed by the number of elements (3 in the example):
|
||||
|
||||
VADD mykey VALUES 3 0.1 1.2 0.5 my-element
|
||||
|
||||
Meaning of the options:
|
||||
|
||||
`REDUCE` implements random projection, in order to reduce the
|
||||
dimensionality of the vector. The projection matrix is saved and reloaded
|
||||
along with the vector set. **Please note that** the `REDUCE` option must be passed immediately before the vector, like in `REDUCE 50 VALUES ...`.
|
||||
|
||||
`CAS` performs the operation partially using threads, in a
|
||||
check-and-set style. The neighbor candidates collection, which is slow, is
|
||||
performed in the background, while the command is executed in the main thread.
|
||||
|
||||
`NOQUANT` forces the vector to be created (in the first VADD call to a given key) without integer 8 quantization, which is otherwise the default.
|
||||
|
||||
`BIN` forces the vector to use binary quantization instead of int8. This is much faster and uses less memory, but has impacts on the recall quality.
|
||||
|
||||
`Q8` forces the vector to use signed 8 bit quantization. This is the default, and the option only exists in order to make sure to check at insertion time if the vector set is of the same format.
|
||||
|
||||
`EF` plays a role in the effort made to find good candidates when connecting the new node to the existing HNSW graph. The default is 200. Using a larger value, may help to have a better recall. To improve the recall it is also possible to increase `EF` during `VSIM` searches.
|
||||
|
||||
`SETATTR` associates attributes to the newly created entry or update the entry attributes (if it already exists). It is the same as calling the `VSETATTR` attribute separately, so please check the documentation of that command in the filtered search section of this documentation.
|
||||
|
||||
`M` defaults to 16 and is the HNSW famous `M` parameters. It is the maximum number of connections that each node of the graph have with other nodes: more connections mean more memory, but a better ability to explore the graph. Nodes at layer zero (every node exists at least at layer zero) have `M*2` connections, while the other layers only have `M` connections. This means that, for instance, an `M` of 64 will use at least 1024 bytes of memory for each node! That is, `64 links * 2 times * 8 bytes pointers`, and even more, since on average each node has something like 1.33 layers (but the other layers have just `M` connections, instead of `M*2`). If you don't have a recall quality problem, the default is fine, and uses a limited amount of memory.
|
||||
|
||||
**VSIM: return elements by vector similarity**
|
||||
|
||||
VSIM key [ELE|FP32|VALUES] <vector or element> [WITHSCORES] [COUNT num] [EF search-exploration-factor] [FILTER expression] [FILTER-EF max-filtering-effort] [TRUTH] [NOTHREAD]
|
||||
|
||||
The command returns similar vectors, for simplicity (and verbosity) in the following example, instead of providing a vector using FP32 or VALUES (like in `VADD`), we will ask for elements having a vector similar to a given element already in the sorted set:
|
||||
|
||||
> VSIM word_embeddings ELE apple
|
||||
1) "apple"
|
||||
2) "apples"
|
||||
3) "pear"
|
||||
4) "fruit"
|
||||
5) "berry"
|
||||
6) "pears"
|
||||
7) "strawberry"
|
||||
8) "peach"
|
||||
9) "potato"
|
||||
10) "grape"
|
||||
|
||||
It is possible to specify a `COUNT` and also to get the similarity score (from 1 to 0, where 1 is identical, 0 is opposite vector) between the query and the returned items.
|
||||
|
||||
> VSIM word_embeddings ELE apple WITHSCORES COUNT 3
|
||||
1) "apple"
|
||||
2) "0.9998867657923256"
|
||||
3) "apples"
|
||||
4) "0.8598527610301971"
|
||||
5) "pear"
|
||||
6) "0.8226882219314575"
|
||||
|
||||
The `EF` argument is the exploration factor: the higher it is, the slower the command becomes, but the better the index is explored to find nodes that are near to our query. Sensible values are from 50 to 1000.
|
||||
|
||||
The `TRUTH` option forces the command to perform a linear scan of all the entries inside the set, without using the graph search inside the HNSW, so it returns the best matching elements (the perfect result set) that can be used in order to easily calculate the recall. Of course the linear scan is `O(N)`, so it is much slower than the `log(N)` (considering a small `COUNT`) provided by the HNSW index.
|
||||
|
||||
The `NOTHREAD` option forces the command to execute the search on the data structure in the main thread. Normally `VSIM` spawns a thread instead. This may be useful for benchmarking purposes, or when we work with extremely small vector sets and don't want to pay the cost of spawning a thread. It is possible that in the future this option will be automatically used by Redis when we detect small vector sets. Note that this option blocks the server for all the time needed to complete the command, so it is a source of potential latency issues: if you are in doubt, never use it.
|
||||
|
||||
For `FILTER` and `FILTER-EF` options, please check the filtered search section of this documentation.
|
||||
|
||||
**VDIM: return the dimension of the vectors inside the vector set**
|
||||
|
||||
VDIM keyname
|
||||
|
||||
Example:
|
||||
|
||||
> VDIM word_embeddings
|
||||
(integer) 300
|
||||
|
||||
Note that in the case of vectors that were populated using the `REDUCE`
|
||||
option, for random projection, the vector set will report the size of
|
||||
the projected (reduced) dimension. Yet the user should perform all the
|
||||
queries using full-size vectors.
|
||||
|
||||
**VCARD: return the number of elements in a vector set**
|
||||
|
||||
VCARD key
|
||||
|
||||
Example:
|
||||
|
||||
> VCARD word_embeddings
|
||||
(integer) 3000000
|
||||
|
||||
|
||||
**VREM: remove elements from vector set**
|
||||
|
||||
VREM key element
|
||||
|
||||
Example:
|
||||
|
||||
> VADD vset VALUES 3 1 0 1 bar
|
||||
(integer) 1
|
||||
> VREM vset bar
|
||||
(integer) 1
|
||||
> VREM vset bar
|
||||
(integer) 0
|
||||
|
||||
VREM does not perform thumstone / logical deletion, but will actually reclaim
|
||||
the memory from the vector set, so it is save to add and remove elements
|
||||
in a vector set in the context of long running applications that continuously
|
||||
update the same index.
|
||||
|
||||
**VEMB: return the approximated vector of an element**
|
||||
|
||||
VEMB key element
|
||||
|
||||
Example:
|
||||
|
||||
> VEMB word_embeddings SQL
|
||||
1) "0.18208661675453186"
|
||||
2) "0.08535309880971909"
|
||||
3) "0.1365649551153183"
|
||||
4) "-0.16501599550247192"
|
||||
5) "0.14225517213344574"
|
||||
... 295 more elements ...
|
||||
|
||||
Because vector sets perform insertion time normalization and optional
|
||||
quantization, the returned vector could be approximated. `VEMB` will take
|
||||
care to de-quantized and de-normalize the vector before returning it.
|
||||
|
||||
It is possible to ask VEMB to return raw data, that is, the internal representation used by the vector: fp32, int8, or a bitmap for binary quantization. This behavior is triggered by the `RAW` option of of VEMB:
|
||||
|
||||
VEMB word_embedding apple RAW
|
||||
|
||||
In this case the return value of the command is an array of three or more elements:
|
||||
1. The name of the quantization used, that is one of: "fp32", "bin", "q8".
|
||||
2. The a string blob containing the raw data, 4 bytes fp32 floats for fp32, a bitmap for binary quants, or int8 bytes array for q8 quants.
|
||||
3. A float representing the l2 of the vector before normalization. You need to multiply by this vector if you want to de-normalize the value for any reason.
|
||||
|
||||
For q8 quantization, an additional elements is also returned: the quantization
|
||||
range, so the integers from -127 to 127 represent (normalized) components
|
||||
in the range `-range`, `+range`.
|
||||
|
||||
**VLINKS: introspection command that shows neighbors for a node**
|
||||
|
||||
VLINKS key element [WITHSCORES]
|
||||
|
||||
The command reports the neighbors for each level.
|
||||
|
||||
**VINFO: introspection command that shows info about a vector set**
|
||||
|
||||
VINFO key
|
||||
|
||||
Example:
|
||||
|
||||
> VINFO word_embeddings
|
||||
1) quant-type
|
||||
2) int8
|
||||
3) vector-dim
|
||||
4) (integer) 300
|
||||
5) size
|
||||
6) (integer) 3000000
|
||||
7) max-level
|
||||
8) (integer) 12
|
||||
9) vset-uid
|
||||
10) (integer) 1
|
||||
11) hnsw-max-node-uid
|
||||
12) (integer) 3000000
|
||||
|
||||
**VSETATTR: associate or remove the JSON attributes of elements**
|
||||
|
||||
VSETATTR key element "{... json ...}"
|
||||
|
||||
Each element of a vector set can be optionally associated with a JSON string
|
||||
in order to use the `FILTER` option of `VSIM` to filter elements by scalars
|
||||
(see the filtered search section for more information). This command can set,
|
||||
update (if already set) or delete (if you set to an empty string) the
|
||||
associated JSON attributes of an element.
|
||||
|
||||
The command returns 0 if the element or the key don't exist, without
|
||||
raising an error, otherwise 1 is returned, and the element attributes
|
||||
are set or updated.
|
||||
|
||||
**VGETATTR: retrieve the JSON attributes of elements**
|
||||
|
||||
VGETATTR key element
|
||||
|
||||
The command returns the JSON attribute associated with an element, or
|
||||
null if there is no element associated, or no element at all, or no key.
|
||||
|
||||
**VRANDMEMBER: return random members from a vector set**
|
||||
|
||||
VRANDMEMBER key [count]
|
||||
|
||||
Return one or more random elements from a vector set.
|
||||
|
||||
The semantics of this command are similar to Redis's native SRANDMEMBER command:
|
||||
|
||||
- When called without count, returns a single random element from the set, as a single string (no array reply).
|
||||
- When called with a positive count, returns up to count distinct random elements (no duplicates).
|
||||
- When called with a negative count, returns count random elements, potentially with duplicates.
|
||||
- If the count value is larger than the set size (and positive), only the entire set is returned.
|
||||
|
||||
If the key doesn't exist, returns a Null reply if count is not given, or an empty array if a count is provided.
|
||||
|
||||
Examples:
|
||||
|
||||
> VADD vset VALUES 3 1 0 0 elem1
|
||||
(integer) 1
|
||||
> VADD vset VALUES 3 0 1 0 elem2
|
||||
(integer) 1
|
||||
> VADD vset VALUES 3 0 0 1 elem3
|
||||
(integer) 1
|
||||
|
||||
# Return a single random element
|
||||
> VRANDMEMBER vset
|
||||
"elem2"
|
||||
|
||||
# Return 2 distinct random elements
|
||||
> VRANDMEMBER vset 2
|
||||
1) "elem1"
|
||||
2) "elem3"
|
||||
|
||||
# Return 3 random elements with possible duplicates
|
||||
> VRANDMEMBER vset -3
|
||||
1) "elem2"
|
||||
2) "elem2"
|
||||
3) "elem1"
|
||||
|
||||
# Return more elements than in the set (returns all elements)
|
||||
> VRANDMEMBER vset 10
|
||||
1) "elem1"
|
||||
2) "elem2"
|
||||
3) "elem3"
|
||||
|
||||
# When key doesn't exist
|
||||
> VRANDMEMBER nonexistent
|
||||
(nil)
|
||||
> VRANDMEMBER nonexistent 3
|
||||
(empty array)
|
||||
|
||||
This command is particularly useful for:
|
||||
|
||||
1. Selecting random samples from a vector set for testing or training.
|
||||
2. Performance testing by retrieving random elements for subsequent similarity searches.
|
||||
|
||||
When the user asks for unique elements (positev count) the implementation optimizes for two scenarios:
|
||||
- For small sample sizes (less than 20% of the set size), it uses a dictionary to avoid duplicates, and performs a real random walk inside the graph.
|
||||
- For large sample sizes (more than 20% of the set size), it starts from a random node and sequentially traverses the internal list, providing faster performances but not really "random" elements.
|
||||
|
||||
The command has `O(N)` worst-case time complexity when requesting many unique elements (it uses linear scanning), or `O(M*log(N))` complexity when the users asks for `M` random elements in a sorted set of `N` elements, with `M` much smaller than `N`.
|
||||
|
||||
# Filtered search
|
||||
|
||||
Each element of the vector set can be associated with a set of attributes specified as a JSON blob:
|
||||
|
||||
> VADD vset VALUES 3 1 1 1 a SETATTR '{"year": 1950}'
|
||||
(integer) 1
|
||||
> VADD vset VALUES 3 -1 -1 -1 b SETATTR '{"year": 1951}'
|
||||
(integer) 1
|
||||
|
||||
Specifying an attribute with the `SETATTR` option of `VADD` is exactly equivalent to adding an element and then setting (or updating, if already set) the attributes JSON string. Also the symmetrical `VGETATTR` command returns the attribute associated to a given element.
|
||||
|
||||
> VADD vset VALUES 3 0 1 0 c
|
||||
(integer) 1
|
||||
> VSETATTR vset c '{"year": 1952}'
|
||||
(integer) 1
|
||||
> VGETATTR vset c
|
||||
"{\"year\": 1952}"
|
||||
|
||||
At this point, I may use the FILTER option of VSIM to only ask for the subset of elements that are verified by my expression:
|
||||
|
||||
> VSIM vset VALUES 3 0 0 0 FILTER '.year > 1950'
|
||||
1) "c"
|
||||
2) "b"
|
||||
|
||||
The items will be returned again in order of similarity (most similar first), but only the items with the year field matching the expression is returned.
|
||||
|
||||
The expressions are similar to what you would write inside the `if` statement of JavaScript or other familiar programming languages: you can use `and`, `or`, the obvious math operators like `+`, `-`, `/`, `>=`, `<`, ... and so forth (see the expressions section for more info). The selectors of the JSON object attributes start with a dot followed by the name of the key inside the JSON objects.
|
||||
|
||||
Elements with invalid JSON or not having a given specified field **are considered as not matching** the expression, but will not generate any error at runtime.
|
||||
|
||||
## FILTER expressions capabilities
|
||||
|
||||
FILTER expressions allow you to perform complex filtering on vector similarity results using a JavaScript-like syntax. The expression is evaluated against each element's JSON attributes, with only elements that satisfy the expression being included in the results.
|
||||
|
||||
### Expression Syntax
|
||||
|
||||
Expressions support the following operators and capabilities:
|
||||
|
||||
1. **Arithmetic operators**: `+`, `-`, `*`, `/`, `%` (modulo), `**` (exponentiation)
|
||||
2. **Comparison operators**: `>`, `>=`, `<`, `<=`, `==`, `!=`
|
||||
3. **Logical operators**: `and`/`&&`, `or`/`||`, `!`/`not`
|
||||
4. **Containment operator**: `in`
|
||||
5. **Parentheses** for grouping: `(...)`
|
||||
|
||||
### Selector Notation
|
||||
|
||||
Attributes are accessed using dot notation:
|
||||
|
||||
- `.year` references the "year" attribute
|
||||
- `.movie.year` would **NOT** reference the "year" field inside a "movie" object, only keys that are at the first level of the JSON object are accessible.
|
||||
|
||||
### JSON and expressions data types
|
||||
|
||||
Expressions can work with:
|
||||
|
||||
- Numbers (dobule precision floats)
|
||||
- Strings (enclosed in single or double quotes)
|
||||
- Booleans (no native type: they are represented as 1 for true, 0 for false)
|
||||
- Arrays (for use with the `in` operator: `value in [1, 2, 3]`)
|
||||
|
||||
JSON attributes are converted in this way:
|
||||
|
||||
- Numbers will be converted to numbers.
|
||||
- Strings to strings.
|
||||
- Booleans to 0 or 1 number.
|
||||
- Arrays to tuples (for "in" operator), but only if composed of just numbers and strings.
|
||||
|
||||
Any other type is ignored, and accessig it will make the expression evaluate to false.
|
||||
|
||||
### Examples
|
||||
|
||||
```
|
||||
# Find items from the 1980s
|
||||
VSIM movies VALUES 3 0.5 0.8 0.2 FILTER '.year >= 1980 and .year < 1990'
|
||||
|
||||
# Find action movies with high ratings
|
||||
VSIM movies VALUES 3 0.5 0.8 0.2 FILTER '.genre == "action" and .rating > 8.0'
|
||||
|
||||
# Find movies directed by either Spielberg or Nolan
|
||||
VSIM movies VALUES 3 0.5 0.8 0.2 FILTER '.director in ["Spielberg", "Nolan"]'
|
||||
|
||||
# Complex condition with numerical operations
|
||||
VSIM movies VALUES 3 0.5 0.8 0.2 FILTER '(.year - 2000) ** 2 < 100 and .rating / 2 > 4'
|
||||
```
|
||||
|
||||
### Error Handling
|
||||
|
||||
Elements with any of the following conditions are considered not matching:
|
||||
- Missing the queried JSON attribute
|
||||
- Having invalid JSON in their attributes
|
||||
- Having a JSON value that cannot be converted to the expected type
|
||||
|
||||
This behavior allows you to safely filter on optional attributes without generating errors.
|
||||
|
||||
### FILTER effort
|
||||
|
||||
The `FILTER-EF` option controls the maximum effort spent when filtering vector search results.
|
||||
|
||||
When performing vector similarity search with filtering, Vector Sets perform the standard similarity search as they apply the filter expression to each node. Since many results might be filtered out, Vector Sets may need to examine a lot more candidates than the requested `COUNT` to ensure sufficient matching results are returned. Actually, if the elements matching the filter are very rare or if there are less than elements matching than the specified count, this would trigger a full scan of the HNSW graph.
|
||||
|
||||
For this reason, by default, the maximum effort is limited to a reasonable amount of nodes explored.
|
||||
|
||||
### Modifying the FILTER effort
|
||||
|
||||
1. By default, Vector Sets will explore up to `COUNT * 100` candidates to find matching results.
|
||||
2. You can control this exploration with the `FILTER-EF` parameter.
|
||||
3. A higher `FILTER-EF` value increases the chances of finding all relevant matches at the cost of increased processing time.
|
||||
4. A `FILTER-EF` of zero will explore as many nodes as needed in order to actually return the number of elements specified by `COUNT`.
|
||||
5. Even when a high `FILTER-EF` value is specified **the implementation will do a lot less work** if the elements passing the filter are very common, because of the early stop conditions of the HNSW implementation (once the specified amount of elements is reached and the quality check of the other candidates trigger an early stop).
|
||||
|
||||
```
|
||||
VSIM key [ELE|FP32|VALUES] <vector or element> COUNT 10 FILTER '.year > 2000' FILTER-EF 500
|
||||
```
|
||||
|
||||
In this example, Vector Sets will examine up to 500 potential nodes. Of course if count is reached before exploring 500 nodes, and the quality checks show that it is not possible to make progresses on similarity, the search is ended sooner.
|
||||
|
||||
### Performance Considerations
|
||||
|
||||
- If you have highly selective filters (few items match), use a higher `FILTER-EF`, or just design your application in order to handle a result set that is smaller than the requested count. Note that anyway the additional elements may be too distant than the query vector.
|
||||
- For less selective filters, the default should be sufficient.
|
||||
- Very selective filters with low `FILTER-EF` values may return fewer items than requested.
|
||||
- Extremely high values may impact performance without significantly improving results.
|
||||
|
||||
The optimal `FILTER-EF` value depends on:
|
||||
1. The selectivity of your filter.
|
||||
2. The distribution of your data.
|
||||
3. The required recall quality.
|
||||
|
||||
A good practice is to start with the default and increase if needed when you observe fewer results than expected.
|
||||
|
||||
### Testing a larg-ish data set
|
||||
|
||||
To really see how things work at scale, you can [download](https://antirez.com/word2vec_with_attribs.rdb) the following dataset:
|
||||
|
||||
wget https://antirez.com/word2vec_with_attribs.rdb
|
||||
|
||||
It contains the 3 million words in Word2Vec having as attribute a JSON with just the length of the word. Because of the length distribution of words in large amounts of texts, where longer words become less and less common, this is ideal to check how filtering behaves with a filter verifying as true with less and less elements in a vector set.
|
||||
|
||||
For instance:
|
||||
|
||||
> VSIM word_embeddings_bin ele "pasta" FILTER ".len == 6"
|
||||
1) "pastas"
|
||||
2) "rotini"
|
||||
3) "gnocci"
|
||||
4) "panino"
|
||||
5) "salads"
|
||||
6) "breads"
|
||||
7) "salame"
|
||||
8) "sauces"
|
||||
9) "cheese"
|
||||
10) "fritti"
|
||||
|
||||
This will easily retrieve the desired amount of items (`COUNT` is 10 by default) since there are many items of length 6. However:
|
||||
|
||||
> VSIM word_embeddings_bin ele "pasta" FILTER ".len == 33"
|
||||
1) "skinless_boneless_chicken_breasts"
|
||||
2) "boneless_skinless_chicken_breasts"
|
||||
3) "Boneless_skinless_chicken_breasts"
|
||||
|
||||
This time even if we asked for 10 items, we only get 3, since the default filter effort will be `10*100 = 1000`. We can tune this giving the effort in an explicit way, with the risk of our query being slower, of course:
|
||||
|
||||
> VSIM word_embeddings_bin ele "pasta" FILTER ".len == 33" FILTER-EF 10000
|
||||
1) "skinless_boneless_chicken_breasts"
|
||||
2) "boneless_skinless_chicken_breasts"
|
||||
3) "Boneless_skinless_chicken_breasts"
|
||||
4) "mozzarella_feta_provolone_cheddar"
|
||||
5) "Greatfood.com_R_www.greatfood.com"
|
||||
6) "Pepperidge_Farm_Goldfish_crackers"
|
||||
7) "Prosecuted_Mobsters_Rebuilt_Dying"
|
||||
8) "Crispy_Snacker_Sandwiches_Popcorn"
|
||||
9) "risultati_delle_partite_disputate"
|
||||
10) "Peppermint_Mocha_Twist_Gingersnap"
|
||||
|
||||
This time we get all the ten items, even if the last one will be quite far from our query vector. We encourage to experiment with this test dataset in order to understand better the dynamics of the implementation and the natural tradeoffs of filtered search.
|
||||
|
||||
**Keep in mind** that by default, Redis Vector Sets will try to avoid a likely very useless huge scan of the HNSW graph, and will be more happy to return few or no elements at all, since this is almost always what the user actually wants in the context of retrieving *similar* items to the query.
|
||||
|
||||
# Single Instance Scalability and Latency
|
||||
|
||||
Vector Sets implement a threading model that allows Redis to handle many concurrent requests: by default `VSIM` is always threaded, and `VADD` is not (but can be partially threaded using the `CAS` option). This section explains how the threading and locking mechanisms work, and what to expect in terms of performance.
|
||||
|
||||
## Threading Model
|
||||
|
||||
- The `VSIM` command runs in a separate thread by default, allowing Redis to continue serving other commands.
|
||||
- A maximum of 32 threads can run concurrently (defined by `HNSW_MAX_THREADS`).
|
||||
- When this limit is reached, additional `VSIM` requests are queued - Redis remains responsive, no latency event is generated.
|
||||
- The `VADD` command with the `CAS` option also leverages threading for the computation-heavy candidate search phase, but the insertion itself is performed in the main thread. `VADD` always runs in a sub-millisecond time, so this is not a source of latency, but having too many hundreds of writes per second can be challenging to handle with a single instance. Please, look at the next section about multiple instances scalability.
|
||||
- Commands run within Lua scripts, MULTI/EXEC blocks, or from replication are executed in the main thread to ensure consistency.
|
||||
|
||||
```
|
||||
> VSIM vset VALUES 3 1 1 1 FILTER '.year > 2000' # This runs in a thread.
|
||||
> VADD vset VALUES 3 1 1 1 element CAS # Candidate search runs in a thread.
|
||||
```
|
||||
|
||||
## Locking Mechanism
|
||||
|
||||
Vector Sets use a read/write locking mechanism to coordinate access:
|
||||
|
||||
- Reads (`VSIM`, `VEMB`, etc.) acquire a read lock, allowing multiple concurrent reads.
|
||||
- Writes (`VADD`, `VREM`, etc.) acquire a write lock, temporarily blocking all reads.
|
||||
- When a write lock is requested while reads are in progress, the write operation waits for all reads to complete.
|
||||
- Once a write lock is granted, all reads are blocked until the write completes.
|
||||
- Each thread has a dedicated slot for tracking visited nodes during graph traversal, avoiding contention. This improves performances but limits the maximum number of concurrent threads, since each node has a memory cost proportional to the number of slots.
|
||||
|
||||
## DEL latency
|
||||
|
||||
Deleting a very large vector set (millions of elements) can cause latency spikes, as deletion rebuilds connections between nodes. This may change in the future.
|
||||
The deletion latency is most noticeable when using `DEL` on a key containing a large vector set or when the key expires.
|
||||
|
||||
## Performance Characteristics
|
||||
|
||||
- Search operations (`VSIM`) scale almost linearly with the number of CPU cores available, up to the thread limit. You can expect a Vector Set composed of million of items associated with components of dimension 300, with the default int8 quantization, to deliver around 50k VSIM operations per second in a single host.
|
||||
- Insertion operations (`VADD`) are more computationally expensive than searches, and can't be threaded: expect much lower throughput, in the range of a few thousands inserts per second.
|
||||
- Binary quantization offers significantly faster search performance at the cost of some recall quality, while int8 quantization, the default, seems to have very small impacts on recall quality, while it significantly improves performances and space efficiency.
|
||||
- The `EF` parameter has a major impact on both search quality and performance - higher values mean better recall but slower searches.
|
||||
- Graph traversal time scales logarithmically with the number of elements, making Vector Sets efficient even with millions of vectors
|
||||
|
||||
## Loading / Saving performances
|
||||
|
||||
Vector Sets are able to serialize on disk the graph structure as it is in memory, so loading back the data does not need to rebuild the HNSW graph. This means that Redis can load millions of items per minute. For instance 3 million items with 300 components vectors can be loaded back into memory into around 15 seconds.
|
||||
|
||||
# Scaling vector sets to multiple instances
|
||||
|
||||
The fundamental way vector sets can be scaled to very large data sets
|
||||
and to many Redis instances is that a given very large set of vectors
|
||||
can be partitioned into N different Redis keys, that can also live into
|
||||
different Redis instances.
|
||||
|
||||
For instance, I could add my elements into `key0`, `key1`, `key2`, by hashing
|
||||
the item in some way, like doing `crc32(item)%3`, effectively splitting
|
||||
the dataset into three different parts. However once I want all the vectors
|
||||
of my dataset near to a given query vector, I could simply perform the
|
||||
`VSIM` command against all the three keys, merging the results by
|
||||
score (so the commands must be called using the `WITHSCORES` option) on
|
||||
the client side: once the union of the results are ordered by the
|
||||
similarity score, the query is equivalent to having a single key `key1+2+3`
|
||||
containing all the items.
|
||||
|
||||
There are a few interesting facts to note about this pattern:
|
||||
|
||||
1. It is possible to have a logical sorted set that is as big as the sum of all the Redis instances we are using.
|
||||
2. Deletion operations remain simple, we can hash the key and select the key where our item belongs.
|
||||
3. However, even if I use 10 different Redis instances, I'm not going to reach 10x the **read** operations per second, compared to using a single server: for each logical query, I need to query all the instances. Yet, smaller graphs are faster to navigate, so there is some win even from the point of view of CPU usage.
|
||||
4. Insertions, so **write** queries, will be scaled linearly: I can add N items against N instances at the same time, splitting the insertion load evenly. This is very important since vector sets, being based on HNSW data structures, are slower to add items than to query similar items, by a very big factor.
|
||||
5. While it cannot guarantee always the best results, with proper timeout management this system may be considered *highly available*, since if a subset of N instances are reachable, I'll be still be able to return similar items to my query vector.
|
||||
|
||||
Notably, this pattern can be implemented in a way that avoids paying the sum of the round trip time with all the servers: it is possible to send the queries at the same time to all the instances, so that latency will be equal the slower reply out of of the N servers queries.
|
||||
|
||||
# Optimizing memory usage
|
||||
|
||||
Vector Sets, or better, HNSWs, the underlying data structure used by Vector Sets, combined with the features provided by the Vector Sets themselves (quantization, random projection, filtering, ...) form an implementation that has a non-trivial space of parameters that can be tuned. Despite to the complexity of the implementation and of vector similarity problems, here there is a list of simple ideas that can drive the user to pick the best settings:
|
||||
|
||||
* 8 bit quantization (the default) is almost always a win. It reduces the memory usage of vectors by a factor of 4, yet the performance penalty in terms of recall is minimal. It also reduces insertion and search time by around 2 times or more.
|
||||
* Binary quantization is much more extreme: it makes vector sets a lot faster, but increases the recall error in a sensible way, for instance from 95% to 80% if all the parameters remain the same. Yet, the speedup is really big, and the memory usage of vectors, compaerd to full precision vectors, 32 times smaller.
|
||||
* Vectors memory usage are not the only responsible for Vector Set high memory usage per entry: nodes contain, on average `M*2 + M*0.33` pointers, where M is by default 16 (but can be tuned in `VADD`, see the `M` option). Also each node has the string item and the optional JSON attributes: those should be as small as possible in order to avoid contributing more to the memory usage.
|
||||
* The `M` parameter should be increased to 32 or more only when a near perfect recall is really needed.
|
||||
* It is possible to gain space (less memory usage) sacrificing time (more CPU time) by using a low `M` (the default of 16, for instance) and a high `EF` (the effort parameter of `VSIM`) in order to scan the graph more deeply.
|
||||
* When memory usage is seriosu concern, and there is the suspect the vectors we are storing don't contain as much information - at least for our use case - to justify the number of components they feature, random projection (the `REDUCE` option of `VADD`) could be tested to see if dimensionality reduction is possible with acceptable precision loss.
|
||||
|
||||
## Random projection tradeoffs
|
||||
|
||||
Sometimes learned vectors are not as information dense as we could guess, that
|
||||
is there are components having similar meanings in the space, and components
|
||||
having values that don't really represent features that matter in our use case.
|
||||
|
||||
At the same time, certain vectors are very big, 1024 components or more. In this cases, it is possible to use the random projection feature of Redis Vector Sets in order to reduce both space (less RAM used) and space (more operstions per second). The feature is accessible via the `REDUCE` option of the `VADD` command. However, keep in mind that you need to test how much reduction impacts the performances of your vectors in term of recall and quality of the results you get back.
|
||||
|
||||
## What is a random projection?
|
||||
|
||||
The concept of Random Projection is relatively simple to grasp. For instance, a projection that turns a 100 components vector into a 10 components vector will perform a different linear transformation between the 100 components and each of the target 10 components. Please note that *each of the target components* will get some random amount of all the 100 original components. It is mathematically proved that this process results in a vector space where elements still have similar distances among them, but still some information will get lost.
|
||||
|
||||
## Examples of projections and loss of precision
|
||||
|
||||
To show you a bit of a extreme case, let's take Word2Vec 3 million items and compress them from 300 to 100, 50 and 25 components vectors. Then, we check the recall compared to the ground truth against each of the vector sets produced in this way (using different `REDUCE` parameters of `VADD`). This is the result, obtained asking for the top 10 elements.
|
||||
|
||||
```
|
||||
----------------------------------------------------------------------
|
||||
Key Average Recall % Std Dev
|
||||
----------------------------------------------------------------------
|
||||
word_embeddings_int8 95.98 12.14
|
||||
^ This is the same key used for ground truth, but without TRUTH option
|
||||
word_embeddings_reduced_100 40.20 20.13
|
||||
word_embeddings_reduced_50 24.42 16.89
|
||||
word_embeddings_reduced_25 14.31 9.99
|
||||
```
|
||||
|
||||
Here the dimensionality reduction we are using is quite extreme: from 300 to 100 means that 66.6% of the original information is lost. The recall drops from 96% to 40%, down to 24% and 14% for even more extreme dimension reduction.
|
||||
|
||||
Reducing the dimension of vectors that are already relatively small, like the above example, of 300 components, will provide only relatively small memory savings, especially because by default Vector Sets use `int8` quantization, that will use only one byte per component:
|
||||
|
||||
```
|
||||
> MEMORY USAGE word_embeddings_int8
|
||||
(integer) 3107002888
|
||||
> MEMORY USAGE word_embeddings_reduced_100
|
||||
(integer) 2507122888
|
||||
```
|
||||
|
||||
Of course going, for example, from 2048 component vectors to 1024 would provide a much more sensible memory saving, even with the `int8` quantization used by Vector Sets, assuming the recall loss is acceptable. Other than the memory saving, there is also the reduction in CPU time, translating to more operations per second.
|
||||
|
||||
Another thing to note is that, with certain embedding models, binary quantization (that offers a 8x reduction of memory usage compared to 8 bit quants, and a very big speedup in computation) performs much better than reducing the dimension of vectors of the same amount via random projections:
|
||||
|
||||
```
|
||||
word_embeddings_bin 35.48 19.78
|
||||
```
|
||||
|
||||
Here in the same test did above: we have a 35% recall which is not too far than the 40% obtained with a random projection from 300 to 100 components. However, while the first technique reduces the size by 3 times, the size reduced of binary quantization is by 8 times.
|
||||
|
||||
```
|
||||
> memory usage word_embeddings_bin
|
||||
(integer) 2327002888
|
||||
```
|
||||
|
||||
In this specific case the key uses JSON attributes and has a graph connection overhead that is much bigger than the 300 bits each vector takes, but, as already said, for big vectors (1024 components, for instance) or for lower values of `M` (see `VADD`, the `M` parameter connects the level of connectivity, so it changes the amount of pointers used per node) the memory saving is much stronger.
|
||||
|
||||
# Vector Sets troubleshooting and understandability
|
||||
|
||||
## Debugging poor recall or unexpected results
|
||||
|
||||
Vector graphs and similarity queries pose many challenges mainly due to the following three problems:
|
||||
|
||||
1. The error due to the approximated nature of Vector Sets is hard to evaluate.
|
||||
2. The error added by the quantization is often depends on the exact vector space (the embedding we are using **and** how far apart the elements we represent into such embeddings are).
|
||||
3. We live in the illusion that learned embeddings capture the best similarity possible among elements, which is obviously not always true, and highly application dependent.
|
||||
|
||||
The only way to debug such problems, is the ability to inspect step by step what is happening inside our application, and the structure of the HNSW graph itself. To do so, we suggest to consider the following tools:
|
||||
|
||||
1. The `TRUTH` option of the `VSIM` command is able to return the ground truth of the most similar elements, without using the HNSW graph, but doing a linear scan.
|
||||
2. The `VLINKS` command allows to explore the graph to see if the connections among nodes make sense, and to investigate why a given node may be more isolated than expected. Such command can also be used in a different way, when we want very fast "similar items" without paying the HNSW traversal time. It exploits the fact that we have a direct reference from each element in our vector set to each node in our HNSW graph.
|
||||
3. The `WITHSCORES` option, in the supported commands, return a value that is directly related to the *cosine similarity* between the query and the items vectors, the interval of the similarity is simply rescaled from the -1, 1 original range to 0, 1, otherwise the metric is identical.
|
||||
|
||||
## Clients, latency and bandwidth usage
|
||||
|
||||
During Vector Sets testing, we discovered that often clients introduce considerable latecy and CPU usage (in the client side, not in Redis) for two main reasons:
|
||||
|
||||
1. Often the serialization to `VALUES ... list of floats ...` can be very slow.
|
||||
2. The vector payload of floats represented as strings is very large, resulting in high bandwidth usage and latency, compared to other Redis commands.
|
||||
|
||||
Switching from `VALUES` to `FP32` as a method for transmitting vectors may easily provide 10-20x speedups.
|
||||
|
||||
# Known bugs
|
||||
|
||||
* Replication code is pretty much untested, and very vanilla (replicating the commands verbatim).
|
||||
|
||||
# Implementation details
|
||||
|
||||
Vector sets are based on the `hnsw.c` implementation of the HNSW data structure with extensions for speed and functionality.
|
||||
|
||||
The main features are:
|
||||
|
||||
* Proper nodes deletion with relinking.
|
||||
* 8 bits and binary quantization.
|
||||
* Threaded queries.
|
||||
* Filtered search with predicate callback.
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,306 @@
|
||||
/*
|
||||
Copyright (c) 2009-2017 Dave Gamble and cJSON contributors
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
*/
|
||||
|
||||
#ifndef cJSON__h
|
||||
#define cJSON__h
|
||||
|
||||
#ifdef __cplusplus
|
||||
extern "C"
|
||||
{
|
||||
#endif
|
||||
|
||||
#if !defined(__WINDOWS__) && (defined(WIN32) || defined(WIN64) || defined(_MSC_VER) || defined(_WIN32))
|
||||
#define __WINDOWS__
|
||||
#endif
|
||||
|
||||
#ifdef __WINDOWS__
|
||||
|
||||
/* When compiling for windows, we specify a specific calling convention to avoid issues where we are being called from a project with a different default calling convention. For windows you have 3 define options:
|
||||
|
||||
CJSON_HIDE_SYMBOLS - Define this in the case where you don't want to ever dllexport symbols
|
||||
CJSON_EXPORT_SYMBOLS - Define this on library build when you want to dllexport symbols (default)
|
||||
CJSON_IMPORT_SYMBOLS - Define this if you want to dllimport symbol
|
||||
|
||||
For *nix builds that support visibility attribute, you can define similar behavior by
|
||||
|
||||
setting default visibility to hidden by adding
|
||||
-fvisibility=hidden (for gcc)
|
||||
or
|
||||
-xldscope=hidden (for sun cc)
|
||||
to CFLAGS
|
||||
|
||||
then using the CJSON_API_VISIBILITY flag to "export" the same symbols the way CJSON_EXPORT_SYMBOLS does
|
||||
|
||||
*/
|
||||
|
||||
#define CJSON_CDECL __cdecl
|
||||
#define CJSON_STDCALL __stdcall
|
||||
|
||||
/* export symbols by default, this is necessary for copy pasting the C and header file */
|
||||
#if !defined(CJSON_HIDE_SYMBOLS) && !defined(CJSON_IMPORT_SYMBOLS) && !defined(CJSON_EXPORT_SYMBOLS)
|
||||
#define CJSON_EXPORT_SYMBOLS
|
||||
#endif
|
||||
|
||||
#if defined(CJSON_HIDE_SYMBOLS)
|
||||
#define CJSON_PUBLIC(type) type CJSON_STDCALL
|
||||
#elif defined(CJSON_EXPORT_SYMBOLS)
|
||||
#define CJSON_PUBLIC(type) __declspec(dllexport) type CJSON_STDCALL
|
||||
#elif defined(CJSON_IMPORT_SYMBOLS)
|
||||
#define CJSON_PUBLIC(type) __declspec(dllimport) type CJSON_STDCALL
|
||||
#endif
|
||||
#else /* !__WINDOWS__ */
|
||||
#define CJSON_CDECL
|
||||
#define CJSON_STDCALL
|
||||
|
||||
#if (defined(__GNUC__) || defined(__SUNPRO_CC) || defined (__SUNPRO_C)) && defined(CJSON_API_VISIBILITY)
|
||||
#define CJSON_PUBLIC(type) __attribute__((visibility("default"))) type
|
||||
#else
|
||||
#define CJSON_PUBLIC(type) type
|
||||
#endif
|
||||
#endif
|
||||
|
||||
/* project version */
|
||||
#define CJSON_VERSION_MAJOR 1
|
||||
#define CJSON_VERSION_MINOR 7
|
||||
#define CJSON_VERSION_PATCH 18
|
||||
|
||||
#include <stddef.h>
|
||||
|
||||
/* cJSON Types: */
|
||||
#define cJSON_Invalid (0)
|
||||
#define cJSON_False (1 << 0)
|
||||
#define cJSON_True (1 << 1)
|
||||
#define cJSON_NULL (1 << 2)
|
||||
#define cJSON_Number (1 << 3)
|
||||
#define cJSON_String (1 << 4)
|
||||
#define cJSON_Array (1 << 5)
|
||||
#define cJSON_Object (1 << 6)
|
||||
#define cJSON_Raw (1 << 7) /* raw json */
|
||||
|
||||
#define cJSON_IsReference 256
|
||||
#define cJSON_StringIsConst 512
|
||||
|
||||
/* The cJSON structure: */
|
||||
typedef struct cJSON
|
||||
{
|
||||
/* next/prev allow you to walk array/object chains. Alternatively, use GetArraySize/GetArrayItem/GetObjectItem */
|
||||
struct cJSON *next;
|
||||
struct cJSON *prev;
|
||||
/* An array or object item will have a child pointer pointing to a chain of the items in the array/object. */
|
||||
struct cJSON *child;
|
||||
|
||||
/* The type of the item, as above. */
|
||||
int type;
|
||||
|
||||
/* The item's string, if type==cJSON_String and type == cJSON_Raw */
|
||||
char *valuestring;
|
||||
/* writing to valueint is DEPRECATED, use cJSON_SetNumberValue instead */
|
||||
int valueint;
|
||||
/* The item's number, if type==cJSON_Number */
|
||||
double valuedouble;
|
||||
|
||||
/* The item's name string, if this item is the child of, or is in the list of subitems of an object. */
|
||||
char *string;
|
||||
} cJSON;
|
||||
|
||||
typedef struct cJSON_Hooks
|
||||
{
|
||||
/* malloc/free are CDECL on Windows regardless of the default calling convention of the compiler, so ensure the hooks allow passing those functions directly. */
|
||||
void *(CJSON_CDECL *malloc_fn)(size_t sz);
|
||||
void (CJSON_CDECL *free_fn)(void *ptr);
|
||||
} cJSON_Hooks;
|
||||
|
||||
typedef int cJSON_bool;
|
||||
|
||||
/* Limits how deeply nested arrays/objects can be before cJSON rejects to parse them.
|
||||
* This is to prevent stack overflows. */
|
||||
#ifndef CJSON_NESTING_LIMIT
|
||||
#define CJSON_NESTING_LIMIT 1000
|
||||
#endif
|
||||
|
||||
/* Limits the length of circular references can be before cJSON rejects to parse them.
|
||||
* This is to prevent stack overflows. */
|
||||
#ifndef CJSON_CIRCULAR_LIMIT
|
||||
#define CJSON_CIRCULAR_LIMIT 10000
|
||||
#endif
|
||||
|
||||
/* returns the version of cJSON as a string */
|
||||
CJSON_PUBLIC(const char*) cJSON_Version(void);
|
||||
|
||||
/* Supply malloc, realloc and free functions to cJSON */
|
||||
CJSON_PUBLIC(void) cJSON_InitHooks(cJSON_Hooks* hooks);
|
||||
|
||||
/* Memory Management: the caller is always responsible to free the results from all variants of cJSON_Parse (with cJSON_Delete) and cJSON_Print (with stdlib free, cJSON_Hooks.free_fn, or cJSON_free as appropriate). The exception is cJSON_PrintPreallocated, where the caller has full responsibility of the buffer. */
|
||||
/* Supply a block of JSON, and this returns a cJSON object you can interrogate. */
|
||||
CJSON_PUBLIC(cJSON *) cJSON_Parse(const char *value);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_ParseWithLength(const char *value, size_t buffer_length);
|
||||
/* ParseWithOpts allows you to require (and check) that the JSON is null terminated, and to retrieve the pointer to the final byte parsed. */
|
||||
/* If you supply a ptr in return_parse_end and parsing fails, then return_parse_end will contain a pointer to the error so will match cJSON_GetErrorPtr(). */
|
||||
CJSON_PUBLIC(cJSON *) cJSON_ParseWithOpts(const char *value, const char **return_parse_end, cJSON_bool require_null_terminated);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_ParseWithLengthOpts(const char *value, size_t buffer_length, const char **return_parse_end, cJSON_bool require_null_terminated);
|
||||
|
||||
/* Render a cJSON entity to text for transfer/storage. */
|
||||
CJSON_PUBLIC(char *) cJSON_Print(const cJSON *item);
|
||||
/* Render a cJSON entity to text for transfer/storage without any formatting. */
|
||||
CJSON_PUBLIC(char *) cJSON_PrintUnformatted(const cJSON *item);
|
||||
/* Render a cJSON entity to text using a buffered strategy. prebuffer is a guess at the final size. guessing well reduces reallocation. fmt=0 gives unformatted, =1 gives formatted */
|
||||
CJSON_PUBLIC(char *) cJSON_PrintBuffered(const cJSON *item, int prebuffer, cJSON_bool fmt);
|
||||
/* Render a cJSON entity to text using a buffer already allocated in memory with given length. Returns 1 on success and 0 on failure. */
|
||||
/* NOTE: cJSON is not always 100% accurate in estimating how much memory it will use, so to be safe allocate 5 bytes more than you actually need */
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_PrintPreallocated(cJSON *item, char *buffer, const int length, const cJSON_bool format);
|
||||
/* Delete a cJSON entity and all subentities. */
|
||||
CJSON_PUBLIC(void) cJSON_Delete(cJSON *item);
|
||||
|
||||
/* Returns the number of items in an array (or object). */
|
||||
CJSON_PUBLIC(int) cJSON_GetArraySize(const cJSON *array);
|
||||
/* Retrieve item number "index" from array "array". Returns NULL if unsuccessful. */
|
||||
CJSON_PUBLIC(cJSON *) cJSON_GetArrayItem(const cJSON *array, int index);
|
||||
/* Get item "string" from object. Case insensitive. */
|
||||
CJSON_PUBLIC(cJSON *) cJSON_GetObjectItem(const cJSON * const object, const char * const string);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_GetObjectItemCaseSensitive(const cJSON * const object, const char * const string);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_HasObjectItem(const cJSON *object, const char *string);
|
||||
/* For analysing failed parses. This returns a pointer to the parse error. You'll probably need to look a few chars back to make sense of it. Defined when cJSON_Parse() returns 0. 0 when cJSON_Parse() succeeds. */
|
||||
CJSON_PUBLIC(const char *) cJSON_GetErrorPtr(void);
|
||||
|
||||
/* Check item type and return its value */
|
||||
CJSON_PUBLIC(char *) cJSON_GetStringValue(const cJSON * const item);
|
||||
CJSON_PUBLIC(double) cJSON_GetNumberValue(const cJSON * const item);
|
||||
|
||||
/* These functions check the type of an item */
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_IsInvalid(const cJSON * const item);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_IsFalse(const cJSON * const item);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_IsTrue(const cJSON * const item);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_IsBool(const cJSON * const item);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_IsNull(const cJSON * const item);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_IsNumber(const cJSON * const item);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_IsString(const cJSON * const item);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_IsArray(const cJSON * const item);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_IsObject(const cJSON * const item);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_IsRaw(const cJSON * const item);
|
||||
|
||||
/* These calls create a cJSON item of the appropriate type. */
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateNull(void);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateTrue(void);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateFalse(void);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateBool(cJSON_bool boolean);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateNumber(double num);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateString(const char *string);
|
||||
/* raw json */
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateRaw(const char *raw);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateArray(void);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateObject(void);
|
||||
|
||||
/* Create a string where valuestring references a string so
|
||||
* it will not be freed by cJSON_Delete */
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateStringReference(const char *string);
|
||||
/* Create an object/array that only references it's elements so
|
||||
* they will not be freed by cJSON_Delete */
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateObjectReference(const cJSON *child);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateArrayReference(const cJSON *child);
|
||||
|
||||
/* These utilities create an Array of count items.
|
||||
* The parameter count cannot be greater than the number of elements in the number array, otherwise array access will be out of bounds.*/
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateIntArray(const int *numbers, int count);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateFloatArray(const float *numbers, int count);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateDoubleArray(const double *numbers, int count);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_CreateStringArray(const char *const *strings, int count);
|
||||
|
||||
/* Append item to the specified array/object. */
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_AddItemToArray(cJSON *array, cJSON *item);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_AddItemToObject(cJSON *object, const char *string, cJSON *item);
|
||||
/* Use this when string is definitely const (i.e. a literal, or as good as), and will definitely survive the cJSON object.
|
||||
* WARNING: When this function was used, make sure to always check that (item->type & cJSON_StringIsConst) is zero before
|
||||
* writing to `item->string` */
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_AddItemToObjectCS(cJSON *object, const char *string, cJSON *item);
|
||||
/* Append reference to item to the specified array/object. Use this when you want to add an existing cJSON to a new cJSON, but don't want to corrupt your existing cJSON. */
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_AddItemReferenceToArray(cJSON *array, cJSON *item);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_AddItemReferenceToObject(cJSON *object, const char *string, cJSON *item);
|
||||
|
||||
/* Remove/Detach items from Arrays/Objects. */
|
||||
CJSON_PUBLIC(cJSON *) cJSON_DetachItemViaPointer(cJSON *parent, cJSON * const item);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_DetachItemFromArray(cJSON *array, int which);
|
||||
CJSON_PUBLIC(void) cJSON_DeleteItemFromArray(cJSON *array, int which);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_DetachItemFromObject(cJSON *object, const char *string);
|
||||
CJSON_PUBLIC(cJSON *) cJSON_DetachItemFromObjectCaseSensitive(cJSON *object, const char *string);
|
||||
CJSON_PUBLIC(void) cJSON_DeleteItemFromObject(cJSON *object, const char *string);
|
||||
CJSON_PUBLIC(void) cJSON_DeleteItemFromObjectCaseSensitive(cJSON *object, const char *string);
|
||||
|
||||
/* Update array items. */
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_InsertItemInArray(cJSON *array, int which, cJSON *newitem); /* Shifts pre-existing items to the right. */
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_ReplaceItemViaPointer(cJSON * const parent, cJSON * const item, cJSON * replacement);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_ReplaceItemInArray(cJSON *array, int which, cJSON *newitem);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_ReplaceItemInObject(cJSON *object,const char *string,cJSON *newitem);
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_ReplaceItemInObjectCaseSensitive(cJSON *object,const char *string,cJSON *newitem);
|
||||
|
||||
/* Duplicate a cJSON item */
|
||||
CJSON_PUBLIC(cJSON *) cJSON_Duplicate(const cJSON *item, cJSON_bool recurse);
|
||||
/* Duplicate will create a new, identical cJSON item to the one you pass, in new memory that will
|
||||
* need to be released. With recurse!=0, it will duplicate any children connected to the item.
|
||||
* The item->next and ->prev pointers are always zero on return from Duplicate. */
|
||||
/* Recursively compare two cJSON items for equality. If either a or b is NULL or invalid, they will be considered unequal.
|
||||
* case_sensitive determines if object keys are treated case sensitive (1) or case insensitive (0) */
|
||||
CJSON_PUBLIC(cJSON_bool) cJSON_Compare(const cJSON * const a, const cJSON * const b, const cJSON_bool case_sensitive);
|
||||
|
||||
/* Minify a strings, remove blank characters(such as ' ', '\t', '\r', '\n') from strings.
|
||||
* The input pointer json cannot point to a read-only address area, such as a string constant,
|
||||
* but should point to a readable and writable address area. */
|
||||
CJSON_PUBLIC(void) cJSON_Minify(char *json);
|
||||
|
||||
/* Helper functions for creating and adding items to an object at the same time.
|
||||
* They return the added item or NULL on failure. */
|
||||
CJSON_PUBLIC(cJSON*) cJSON_AddNullToObject(cJSON * const object, const char * const name);
|
||||
CJSON_PUBLIC(cJSON*) cJSON_AddTrueToObject(cJSON * const object, const char * const name);
|
||||
CJSON_PUBLIC(cJSON*) cJSON_AddFalseToObject(cJSON * const object, const char * const name);
|
||||
CJSON_PUBLIC(cJSON*) cJSON_AddBoolToObject(cJSON * const object, const char * const name, const cJSON_bool boolean);
|
||||
CJSON_PUBLIC(cJSON*) cJSON_AddNumberToObject(cJSON * const object, const char * const name, const double number);
|
||||
CJSON_PUBLIC(cJSON*) cJSON_AddStringToObject(cJSON * const object, const char * const name, const char * const string);
|
||||
CJSON_PUBLIC(cJSON*) cJSON_AddRawToObject(cJSON * const object, const char * const name, const char * const raw);
|
||||
CJSON_PUBLIC(cJSON*) cJSON_AddObjectToObject(cJSON * const object, const char * const name);
|
||||
CJSON_PUBLIC(cJSON*) cJSON_AddArrayToObject(cJSON * const object, const char * const name);
|
||||
|
||||
/* When assigning an integer value, it needs to be propagated to valuedouble too. */
|
||||
#define cJSON_SetIntValue(object, number) ((object) ? (object)->valueint = (object)->valuedouble = (number) : (number))
|
||||
/* helper for the cJSON_SetNumberValue macro */
|
||||
CJSON_PUBLIC(double) cJSON_SetNumberHelper(cJSON *object, double number);
|
||||
#define cJSON_SetNumberValue(object, number) ((object != NULL) ? cJSON_SetNumberHelper(object, (double)number) : (number))
|
||||
/* Change the valuestring of a cJSON_String object, only takes effect when type of object is cJSON_String */
|
||||
CJSON_PUBLIC(char*) cJSON_SetValuestring(cJSON *object, const char *valuestring);
|
||||
|
||||
/* If the object is not a boolean type this does nothing and returns cJSON_Invalid else it returns the new type*/
|
||||
#define cJSON_SetBoolValue(object, boolValue) ( \
|
||||
(object != NULL && ((object)->type & (cJSON_False|cJSON_True))) ? \
|
||||
(object)->type=((object)->type &(~(cJSON_False|cJSON_True)))|((boolValue)?cJSON_True:cJSON_False) : \
|
||||
cJSON_Invalid\
|
||||
)
|
||||
|
||||
/* Macro for iterating over an array or object */
|
||||
#define cJSON_ArrayForEach(element, array) for(element = (array != NULL) ? (array)->child : NULL; element != NULL; element = element->next)
|
||||
|
||||
/* malloc/free objects using the malloc/free functions that have been set with cJSON_InitHooks */
|
||||
CJSON_PUBLIC(void *) cJSON_malloc(size_t size);
|
||||
CJSON_PUBLIC(void) cJSON_free(void *object);
|
||||
|
||||
#ifdef __cplusplus
|
||||
}
|
||||
#endif
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1 @@
|
||||
venv
|
||||
@@ -0,0 +1,44 @@
|
||||
This tool is similar to redis-cli (but very basic) but allows
|
||||
to specify arguments that are expanded as vectors by calling
|
||||
ollama to get the embedding.
|
||||
|
||||
Whatever is passed as !"foo bar" gets expanded into
|
||||
VALUES ... embedding ...
|
||||
|
||||
You must have ollama running with the mxbai-emb-large model
|
||||
already installed for this to work.
|
||||
|
||||
Example:
|
||||
|
||||
redis> KEYS *
|
||||
1) food_items
|
||||
2) glove_embeddings_bin
|
||||
3) many_movies_mxbai-embed-large_BIN
|
||||
4) many_movies_mxbai-embed-large_NOQUANT
|
||||
5) word_embeddings
|
||||
6) word_embeddings_bin
|
||||
7) glove_embeddings_fp32
|
||||
|
||||
redis> VSIM food_items !"drinks with fruit"
|
||||
1) (Fruit)Juices,Lemonade,100ml,50 cal,210 kJ
|
||||
2) (Fruit)Juices,Limeade,100ml,128 cal,538 kJ
|
||||
3) CannedFruit,Canned Fruit Cocktail,100g,81 cal,340 kJ
|
||||
4) (Fruit)Juices,Energy-Drink,100ml,87 cal,365 kJ
|
||||
5) Fruits,Lime,100g,30 cal,126 kJ
|
||||
6) (Fruit)Juices,Coconut Water,100ml,19 cal,80 kJ
|
||||
7) Fruits,Lemon,100g,29 cal,122 kJ
|
||||
8) (Fruit)Juices,Clamato,100ml,60 cal,252 kJ
|
||||
9) Fruits,Fruit salad,100g,50 cal,210 kJ
|
||||
10) (Fruit)Juices,Capri-Sun,100ml,41 cal,172 kJ
|
||||
|
||||
redis> vsim food_items !"barilla"
|
||||
1) Pasta&Noodles,Spirelli,100g,367 cal,1541 kJ
|
||||
2) Pasta&Noodles,Farfalle,100g,358 cal,1504 kJ
|
||||
3) Pasta&Noodles,Capellini,100g,353 cal,1483 kJ
|
||||
4) Pasta&Noodles,Spaetzle,100g,368 cal,1546 kJ
|
||||
5) Pasta&Noodles,Cappelletti,100g,164 cal,689 kJ
|
||||
6) Pasta&Noodles,Penne,100g,351 cal,1474 kJ
|
||||
7) Pasta&Noodles,Shells,100g,353 cal,1483 kJ
|
||||
8) Pasta&Noodles,Linguine,100g,357 cal,1499 kJ
|
||||
9) Pasta&Noodles,Rotini,100g,353 cal,1483 kJ
|
||||
10) Pasta&Noodles,Rigatoni,100g,353 cal,1483 kJ
|
||||
Executable
+146
@@ -0,0 +1,146 @@
|
||||
#
|
||||
# Copyright (c) 2009-Present, Redis Ltd.
|
||||
# All rights reserved.
|
||||
#
|
||||
# Licensed under your choice of the Redis Source Available License 2.0
|
||||
# (RSALv2) or the Server Side Public License v1 (SSPLv1).
|
||||
#
|
||||
|
||||
#!/usr/bin/env python3
|
||||
import redis
|
||||
import requests
|
||||
import re
|
||||
import shlex
|
||||
from prompt_toolkit import PromptSession
|
||||
from prompt_toolkit.history import InMemoryHistory
|
||||
|
||||
def get_embedding(text):
|
||||
"""Get embedding from local Ollama API"""
|
||||
url = "http://localhost:11434/api/embeddings"
|
||||
payload = {
|
||||
"model": "mxbai-embed-large",
|
||||
"prompt": text
|
||||
}
|
||||
try:
|
||||
response = requests.post(url, json=payload)
|
||||
response.raise_for_status()
|
||||
return response.json()['embedding']
|
||||
except requests.exceptions.RequestException as e:
|
||||
raise Exception(f"Failed to get embedding: {str(e)}")
|
||||
|
||||
def process_embedding_patterns(text):
|
||||
"""Process !"text" and !!"text" patterns in the command"""
|
||||
|
||||
def replace_with_embedding(match):
|
||||
text = match.group(1)
|
||||
embedding = get_embedding(text)
|
||||
return f"VALUES {len(embedding)} {' '.join(map(str, embedding))}"
|
||||
|
||||
def replace_with_embedding_and_text(match):
|
||||
text = match.group(1)
|
||||
embedding = get_embedding(text)
|
||||
# Return both the embedding values and the original text as next argument
|
||||
return f'VALUES {len(embedding)} {" ".join(map(str, embedding))} "{text}"'
|
||||
|
||||
# First handle !!"text" pattern (must be done before !"text")
|
||||
text = re.sub(r'!!"([^"]*)"', replace_with_embedding_and_text, text)
|
||||
# Then handle !"text" pattern
|
||||
text = re.sub(r'!"([^"]*)"', replace_with_embedding, text)
|
||||
return text
|
||||
|
||||
def parse_command(command):
|
||||
"""Parse command respecting quoted strings"""
|
||||
try:
|
||||
# Use shlex to properly handle quoted strings
|
||||
return shlex.split(command)
|
||||
except ValueError as e:
|
||||
raise Exception(f"Invalid command syntax: {str(e)}")
|
||||
|
||||
def format_response(response):
|
||||
"""Format the response to match Redis protocol style"""
|
||||
if response is None:
|
||||
return "(nil)"
|
||||
elif isinstance(response, bool):
|
||||
return "+OK" if response else "(error) Operation failed"
|
||||
elif isinstance(response, (list, set)):
|
||||
if not response:
|
||||
return "(empty list or set)"
|
||||
return "\n".join(f"{i+1}) {item}" for i, item in enumerate(response))
|
||||
elif isinstance(response, int):
|
||||
return f"(integer) {response}"
|
||||
else:
|
||||
return str(response)
|
||||
|
||||
def main():
|
||||
# Default connection to localhost:6379
|
||||
r = redis.Redis(host='localhost', port=6379, decode_responses=True)
|
||||
|
||||
try:
|
||||
# Test connection
|
||||
r.ping()
|
||||
print("Connected to Redis. Type your commands (CTRL+D to exit):")
|
||||
print("Special syntax:")
|
||||
print(" !\"text\" - Replace with embedding")
|
||||
print(" !!\"text\" - Replace with embedding and append text as value")
|
||||
print(" \"text\" - Quote strings containing spaces")
|
||||
except redis.ConnectionError:
|
||||
print("Error: Could not connect to Redis server")
|
||||
return
|
||||
|
||||
# Setup prompt session with history
|
||||
session = PromptSession(history=InMemoryHistory())
|
||||
|
||||
# Main loop
|
||||
while True:
|
||||
try:
|
||||
# Read input with line editing support
|
||||
command = session.prompt("redis> ")
|
||||
|
||||
# Skip empty commands
|
||||
if not command.strip():
|
||||
continue
|
||||
|
||||
# Process any embedding patterns before parsing
|
||||
try:
|
||||
processed_command = process_embedding_patterns(command)
|
||||
except Exception as e:
|
||||
print(f"(error) Embedding processing failed: {str(e)}")
|
||||
continue
|
||||
|
||||
# Parse the command respecting quoted strings
|
||||
try:
|
||||
parts = parse_command(processed_command)
|
||||
except Exception as e:
|
||||
print(f"(error) {str(e)}")
|
||||
continue
|
||||
|
||||
if not parts:
|
||||
continue
|
||||
|
||||
cmd = parts[0].lower()
|
||||
args = parts[1:]
|
||||
|
||||
# Execute command
|
||||
try:
|
||||
method = getattr(r, cmd, None)
|
||||
if method is not None:
|
||||
result = method(*args)
|
||||
else:
|
||||
# Use execute_command for unknown commands
|
||||
result = r.execute_command(cmd, *args)
|
||||
print(format_response(result))
|
||||
except AttributeError:
|
||||
print(f"(error) Unknown command '{cmd}'")
|
||||
|
||||
except EOFError:
|
||||
print("\nGoodbye!")
|
||||
break
|
||||
except KeyboardInterrupt:
|
||||
continue # Allow Ctrl+C to clear current line
|
||||
except redis.RedisError as e:
|
||||
print(f"(error) {str(e)}")
|
||||
except Exception as e:
|
||||
print(f"(error) {str(e)}")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,3 @@
|
||||
wget http://ann-benchmarks.com/glove-100-angular.hdf5
|
||||
python insert.py
|
||||
python recall.py (use --k <count> optionally, default top-10)
|
||||
@@ -0,0 +1,55 @@
|
||||
#
|
||||
# Copyright (c) 2009-Present, Redis Ltd.
|
||||
# All rights reserved.
|
||||
#
|
||||
# Licensed under your choice of the Redis Source Available License 2.0
|
||||
# (RSALv2) or the Server Side Public License v1 (SSPLv1).
|
||||
#
|
||||
|
||||
import h5py
|
||||
import redis
|
||||
from tqdm import tqdm
|
||||
|
||||
# Initialize Redis connection
|
||||
redis_client = redis.Redis(host='localhost', port=6379, decode_responses=True, encoding='utf-8')
|
||||
|
||||
def add_to_redis(index, embedding):
|
||||
"""Add embedding to Redis using VADD command"""
|
||||
args = ["VADD", "glove_embeddings", "VALUES", "100"] # 100 is vector dimension
|
||||
args.extend(map(str, embedding))
|
||||
args.append(f"{index}") # Using index as identifier since we don't have words
|
||||
args.append("EF")
|
||||
args.append("200")
|
||||
# args.append("NOQUANT")
|
||||
# args.append("BIN")
|
||||
redis_client.execute_command(*args)
|
||||
|
||||
def main():
|
||||
with h5py.File('glove-100-angular.hdf5', 'r') as f:
|
||||
# Get the train dataset
|
||||
train_vectors = f['train']
|
||||
total_vectors = train_vectors.shape[0]
|
||||
|
||||
print(f"Starting to process {total_vectors} vectors...")
|
||||
|
||||
# Process in batches to avoid memory issues
|
||||
batch_size = 1000
|
||||
|
||||
for i in tqdm(range(0, total_vectors, batch_size)):
|
||||
batch_end = min(i + batch_size, total_vectors)
|
||||
batch = train_vectors[i:batch_end]
|
||||
|
||||
for j, vector in enumerate(batch):
|
||||
try:
|
||||
current_index = i + j
|
||||
add_to_redis(current_index, vector)
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error processing vector {current_index}: {str(e)}")
|
||||
continue
|
||||
|
||||
if (i + batch_size) % 10000 == 0:
|
||||
print(f"Processed {i + batch_size} vectors")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,86 @@
|
||||
#
|
||||
# Copyright (c) 2009-Present, Redis Ltd.
|
||||
# All rights reserved.
|
||||
#
|
||||
# Licensed under your choice of the Redis Source Available License 2.0
|
||||
# (RSALv2) or the Server Side Public License v1 (SSPLv1).
|
||||
#
|
||||
|
||||
import h5py
|
||||
import redis
|
||||
import numpy as np
|
||||
from tqdm import tqdm
|
||||
import argparse
|
||||
|
||||
# Initialize Redis connection
|
||||
redis_client = redis.Redis(host='localhost', port=6379, decode_responses=True, encoding='utf-8')
|
||||
|
||||
def get_redis_neighbors(query_vector, k):
|
||||
"""Get nearest neighbors using Redis VSIM command"""
|
||||
args = ["VSIM", "glove_embeddings_bin", "VALUES", "100"]
|
||||
args.extend(map(str, query_vector))
|
||||
args.extend(["COUNT", str(k)])
|
||||
args.extend(["EF", 100])
|
||||
if False:
|
||||
print(args)
|
||||
exit(1)
|
||||
results = redis_client.execute_command(*args)
|
||||
return [int(res) for res in results]
|
||||
|
||||
def calculate_recall(ground_truth, predicted, k):
|
||||
"""Calculate recall@k"""
|
||||
relevant = set(ground_truth[:k])
|
||||
retrieved = set(predicted[:k])
|
||||
return len(relevant.intersection(retrieved)) / len(relevant)
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description='Evaluate Redis VSIM recall')
|
||||
parser.add_argument('--k', type=int, default=10, help='Number of neighbors to evaluate (default: 10)')
|
||||
parser.add_argument('--batch', type=int, default=100, help='Progress update frequency (default: 100)')
|
||||
args = parser.parse_args()
|
||||
|
||||
k = args.k
|
||||
batch_size = args.batch
|
||||
|
||||
with h5py.File('glove-100-angular.hdf5', 'r') as f:
|
||||
test_vectors = f['test'][:]
|
||||
ground_truth_neighbors = f['neighbors'][:]
|
||||
|
||||
num_queries = len(test_vectors)
|
||||
recalls = []
|
||||
|
||||
print(f"Evaluating recall@{k} for {num_queries} test queries...")
|
||||
|
||||
for i in tqdm(range(num_queries)):
|
||||
try:
|
||||
# Get Redis results
|
||||
redis_neighbors = get_redis_neighbors(test_vectors[i], k)
|
||||
|
||||
# Get ground truth for this query
|
||||
true_neighbors = ground_truth_neighbors[i]
|
||||
|
||||
# Calculate recall
|
||||
recall = calculate_recall(true_neighbors, redis_neighbors, k)
|
||||
recalls.append(recall)
|
||||
|
||||
if (i + 1) % batch_size == 0:
|
||||
current_avg_recall = np.mean(recalls)
|
||||
print(f"Current average recall@{k} after {i+1} queries: {current_avg_recall:.4f}")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error processing query {i}: {str(e)}")
|
||||
continue
|
||||
|
||||
final_recall = np.mean(recalls)
|
||||
print("\nFinal Results:")
|
||||
print(f"Average recall@{k}: {final_recall:.4f}")
|
||||
print(f"Total queries evaluated: {len(recalls)}")
|
||||
|
||||
# Save detailed results
|
||||
with open(f'recall_evaluation_results_k{k}.txt', 'w') as f:
|
||||
f.write(f"Average recall@{k}: {final_recall:.4f}\n")
|
||||
f.write(f"Total queries evaluated: {len(recalls)}\n")
|
||||
f.write(f"Individual query recalls: {recalls}\n")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,2 @@
|
||||
mpst_full_data.csv
|
||||
partition.json
|
||||
@@ -0,0 +1,30 @@
|
||||
This example maps long form movies plots to movies titles.
|
||||
It will create fp32 and binary vectors (the two extremes).
|
||||
|
||||
1. Install ollama, and install the embedding model "mxbai-embed-large"
|
||||
2. Download mpst_full_data.csv from https://www.kaggle.com/datasets/cryptexcode/mpst-movie-plot-synopses-with-tags
|
||||
3. python insert.py
|
||||
|
||||
127.0.0.1:6379> VSIM many_movies_mxbai-embed-large_NOQUANT ELE "The Matrix"
|
||||
1) "The Matrix"
|
||||
2) "The Matrix Reloaded"
|
||||
3) "The Matrix Revolutions"
|
||||
4) "Commando"
|
||||
5) "Avatar"
|
||||
6) "Forbidden Planet"
|
||||
7) "Terminator Salvation"
|
||||
8) "Mandroid"
|
||||
9) "The Omega Code"
|
||||
10) "Coherence"
|
||||
|
||||
127.0.0.1:6379> VSIM many_movies_mxbai-embed-large_BIN ELE "The Matrix"
|
||||
1) "The Matrix"
|
||||
2) "The Matrix Reloaded"
|
||||
3) "The Matrix Revolutions"
|
||||
4) "The Omega Code"
|
||||
5) "Forbidden Planet"
|
||||
6) "Avatar"
|
||||
7) "John Carter"
|
||||
8) "System Shock 2"
|
||||
9) "Coherence"
|
||||
10) "Tomorrowland"
|
||||
@@ -0,0 +1,56 @@
|
||||
#
|
||||
# Copyright (c) 2009-Present, Redis Ltd.
|
||||
# All rights reserved.
|
||||
#
|
||||
# Licensed under your choice of the Redis Source Available License 2.0
|
||||
# (RSALv2) or the Server Side Public License v1 (SSPLv1).
|
||||
#
|
||||
|
||||
import csv
|
||||
import requests
|
||||
import redis
|
||||
|
||||
ModelName="mxbai-embed-large"
|
||||
|
||||
# Initialize Redis connection, setting encoding to utf-8
|
||||
redis_client = redis.Redis(host='localhost', port=6379, decode_responses=True, encoding='utf-8')
|
||||
|
||||
def get_embedding(text):
|
||||
"""Get embedding from local API"""
|
||||
url = "http://localhost:11434/api/embeddings"
|
||||
payload = {
|
||||
"model": ModelName,
|
||||
"prompt": "Represent this movie plot and genre: "+text
|
||||
}
|
||||
response = requests.post(url, json=payload)
|
||||
return response.json()['embedding']
|
||||
|
||||
def add_to_redis(title, embedding, quant_type):
|
||||
"""Add embedding to Redis using VADD command"""
|
||||
args = ["VADD", "many_movies_"+ModelName+"_"+quant_type, "VALUES", str(len(embedding))]
|
||||
args.extend(map(str, embedding))
|
||||
args.append(title)
|
||||
args.append(quant_type)
|
||||
redis_client.execute_command(*args)
|
||||
|
||||
def main():
|
||||
with open('mpst_full_data.csv', 'r', encoding='utf-8') as file:
|
||||
reader = csv.DictReader(file)
|
||||
|
||||
for movie in reader:
|
||||
try:
|
||||
text_to_embed = f"{movie['title']} {movie['plot_synopsis']} {movie['tags']}"
|
||||
|
||||
print(f"Getting embedding for: {movie['title']}")
|
||||
embedding = get_embedding(text_to_embed)
|
||||
|
||||
add_to_redis(movie['title'], embedding, "BIN")
|
||||
add_to_redis(movie['title'], embedding, "NOQUANT")
|
||||
print(f"Successfully processed: {movie['title']}")
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error processing {movie['title']}: {str(e)}")
|
||||
continue
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,999 @@
|
||||
/* Filtering of objects based on simple expressions.
|
||||
* This powers the FILTER option of Vector Sets, but it is otherwise
|
||||
* general code to be used when we want to tell if a given object (with fields)
|
||||
* passes or fails a given test for scalars, strings, ...
|
||||
*
|
||||
* Copyright (c) 2009-Present, Redis Ltd.
|
||||
* All rights reserved.
|
||||
*
|
||||
* Licensed under your choice of the Redis Source Available License 2.0
|
||||
* (RSALv2) or the Server Side Public License v1 (SSPLv1).
|
||||
* Originally authored by: Salvatore Sanfilippo.
|
||||
*/
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <ctype.h>
|
||||
#include <math.h>
|
||||
#include <string.h>
|
||||
#include "cJSON.h"
|
||||
|
||||
#ifdef TEST_MAIN
|
||||
#define RedisModule_Alloc malloc
|
||||
#define RedisModule_Realloc realloc
|
||||
#define RedisModule_Free free
|
||||
#define RedisModule_Strdup strdup
|
||||
#endif
|
||||
|
||||
#define EXPR_TOKEN_EOF 0
|
||||
#define EXPR_TOKEN_NUM 1
|
||||
#define EXPR_TOKEN_STR 2
|
||||
#define EXPR_TOKEN_TUPLE 3
|
||||
#define EXPR_TOKEN_SELECTOR 4
|
||||
#define EXPR_TOKEN_OP 5
|
||||
|
||||
#define EXPR_OP_OPAREN 0 /* ( */
|
||||
#define EXPR_OP_CPAREN 1 /* ) */
|
||||
#define EXPR_OP_NOT 2 /* ! */
|
||||
#define EXPR_OP_POW 3 /* ** */
|
||||
#define EXPR_OP_MULT 4 /* * */
|
||||
#define EXPR_OP_DIV 5 /* / */
|
||||
#define EXPR_OP_MOD 6 /* % */
|
||||
#define EXPR_OP_SUM 7 /* + */
|
||||
#define EXPR_OP_DIFF 8 /* - */
|
||||
#define EXPR_OP_GT 9 /* > */
|
||||
#define EXPR_OP_GTE 10 /* >= */
|
||||
#define EXPR_OP_LT 11 /* < */
|
||||
#define EXPR_OP_LTE 12 /* <= */
|
||||
#define EXPR_OP_EQ 13 /* == */
|
||||
#define EXPR_OP_NEQ 14 /* != */
|
||||
#define EXPR_OP_IN 15 /* in */
|
||||
#define EXPR_OP_AND 16 /* and */
|
||||
#define EXPR_OP_OR 17 /* or */
|
||||
|
||||
/* This structure represents a token in our expression. It's either
|
||||
* literals like 4, "foo", or operators like "+", "-", "and", or
|
||||
* json selectors, that start with a dot: ".age", ".properties.somearray[1]" */
|
||||
typedef struct exprtoken {
|
||||
int refcount; // Reference counting for memory reclaiming.
|
||||
int token_type; // Token type of the just parsed token.
|
||||
int offset; // Chars offset in expression.
|
||||
union {
|
||||
double num; // Value for EXPR_TOKEN_NUM.
|
||||
struct {
|
||||
char *start; // String pointer for EXPR_TOKEN_STR / SELECTOR.
|
||||
size_t len; // String len for EXPR_TOKEN_STR / SELECTOR.
|
||||
char *heapstr; // True if we have a private allocation for this
|
||||
// string. When possible, it just references to the
|
||||
// string expression we compiled, exprstate->expr.
|
||||
} str;
|
||||
int opcode; // Opcode ID for EXPR_TOKEN_OP.
|
||||
struct {
|
||||
struct exprtoken **ele;
|
||||
size_t len;
|
||||
} tuple; // Tuples are like [1, 2, 3] for "in" operator.
|
||||
};
|
||||
} exprtoken;
|
||||
|
||||
/* Simple stack of expr tokens. This is used both to represent the stack
|
||||
* of values and the stack of operands during VM execution. */
|
||||
typedef struct exprstack {
|
||||
exprtoken **items;
|
||||
int numitems;
|
||||
int allocsize;
|
||||
} exprstack;
|
||||
|
||||
typedef struct exprstate {
|
||||
char *expr; /* Expression string to compile. Note that
|
||||
* expression token strings point directly to this
|
||||
* string. */
|
||||
char *p; // Current position inside 'expr', while parsing.
|
||||
|
||||
// Virtual machine state.
|
||||
exprstack values_stack;
|
||||
exprstack ops_stack; // Operator stack used during compilation.
|
||||
exprstack tokens; // Expression processed into a sequence of tokens.
|
||||
exprstack program; // Expression compiled into opcodes and values.
|
||||
} exprstate;
|
||||
|
||||
/* Valid operators. */
|
||||
struct {
|
||||
char *opname;
|
||||
int oplen;
|
||||
int opcode;
|
||||
int precedence;
|
||||
int arity;
|
||||
} ExprOptable[] = {
|
||||
{"(", 1, EXPR_OP_OPAREN, 7, 0},
|
||||
{")", 1, EXPR_OP_CPAREN, 7, 0},
|
||||
{"!", 1, EXPR_OP_NOT, 6, 1},
|
||||
{"not", 3, EXPR_OP_NOT, 6, 1},
|
||||
{"**", 2, EXPR_OP_POW, 5, 2},
|
||||
{"*", 1, EXPR_OP_MULT, 4, 2},
|
||||
{"/", 1, EXPR_OP_DIV, 4, 2},
|
||||
{"%", 1, EXPR_OP_MOD, 4, 2},
|
||||
{"+", 1, EXPR_OP_SUM, 3, 2},
|
||||
{"-", 1, EXPR_OP_DIFF, 3, 2},
|
||||
{">", 1, EXPR_OP_GT, 2, 2},
|
||||
{">=", 2, EXPR_OP_GTE, 2, 2},
|
||||
{"<", 1, EXPR_OP_LT, 2, 2},
|
||||
{"<=", 2, EXPR_OP_LTE, 2, 2},
|
||||
{"==", 2, EXPR_OP_EQ, 2, 2},
|
||||
{"!=", 2, EXPR_OP_NEQ, 2, 2},
|
||||
{"in", 2, EXPR_OP_IN, 2, 2},
|
||||
{"and", 3, EXPR_OP_AND, 1, 2},
|
||||
{"&&", 2, EXPR_OP_AND, 1, 2},
|
||||
{"or", 2, EXPR_OP_OR, 0, 2},
|
||||
{"||", 2, EXPR_OP_OR, 0, 2},
|
||||
{NULL, 0, 0, 0, 0} // Terminator.
|
||||
};
|
||||
|
||||
#define EXPR_OP_SPECIALCHARS "+-*%/!()<>=|&"
|
||||
#define EXPR_SELECTOR_SPECIALCHARS "_-"
|
||||
|
||||
/* ================================ Expr token ============================== */
|
||||
|
||||
/* Return an heap allocated token of the specified type, setting the
|
||||
* reference count to 1. */
|
||||
exprtoken *exprNewToken(int type) {
|
||||
exprtoken *t = RedisModule_Alloc(sizeof(exprtoken));
|
||||
memset(t,0,sizeof(*t));
|
||||
t->token_type = type;
|
||||
t->refcount = 1;
|
||||
return t;
|
||||
}
|
||||
|
||||
/* Generic free token function, can be used to free stack allocated
|
||||
* objects (in this case the pointer itself will not be freed) or
|
||||
* heap allocated objects. See the wrappers below. */
|
||||
void exprTokenRelease(exprtoken *t) {
|
||||
if (t == NULL) return;
|
||||
|
||||
if (t->refcount <= 0) {
|
||||
printf("exprTokenRelease() against a token with refcount %d!\n"
|
||||
"Aborting program execution\n",
|
||||
t->refcount);
|
||||
exit(1);
|
||||
}
|
||||
t->refcount--;
|
||||
if (t->refcount > 0) return;
|
||||
|
||||
// We reached refcount 0: free the object.
|
||||
if (t->token_type == EXPR_TOKEN_STR) {
|
||||
if (t->str.heapstr != NULL) RedisModule_Free(t->str.heapstr);
|
||||
} else if (t->token_type == EXPR_TOKEN_TUPLE) {
|
||||
for (size_t j = 0; j < t->tuple.len; j++)
|
||||
exprTokenRelease(t->tuple.ele[j]);
|
||||
if (t->tuple.ele) RedisModule_Free(t->tuple.ele);
|
||||
}
|
||||
RedisModule_Free(t);
|
||||
}
|
||||
|
||||
void exprTokenRetain(exprtoken *t) {
|
||||
t->refcount++;
|
||||
}
|
||||
|
||||
/* ============================== Stack handling ============================ */
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#define EXPR_STACK_INITIAL_SIZE 16
|
||||
|
||||
/* Initialize a new expression stack. */
|
||||
void exprStackInit(exprstack *stack) {
|
||||
stack->items = RedisModule_Alloc(sizeof(exprtoken*) * EXPR_STACK_INITIAL_SIZE);
|
||||
stack->numitems = 0;
|
||||
stack->allocsize = EXPR_STACK_INITIAL_SIZE;
|
||||
}
|
||||
|
||||
/* Push a token pointer onto the stack. Does not increment the refcount
|
||||
* of the token: it is up to the caller doing this. */
|
||||
void exprStackPush(exprstack *stack, exprtoken *token) {
|
||||
/* Check if we need to grow the stack. */
|
||||
if (stack->numitems == stack->allocsize) {
|
||||
size_t newsize = stack->allocsize * 2;
|
||||
exprtoken **newitems =
|
||||
RedisModule_Realloc(stack->items, sizeof(exprtoken*) * newsize);
|
||||
stack->items = newitems;
|
||||
stack->allocsize = newsize;
|
||||
}
|
||||
stack->items[stack->numitems] = token;
|
||||
stack->numitems++;
|
||||
}
|
||||
|
||||
/* Pop a token pointer from the stack. Return NULL if the stack is
|
||||
* empty. Does NOT recrement the refcount of the token, it's up to the
|
||||
* caller to do so, as the new owner of the reference. */
|
||||
exprtoken *exprStackPop(exprstack *stack) {
|
||||
if (stack->numitems == 0) return NULL;
|
||||
stack->numitems--;
|
||||
return stack->items[stack->numitems];
|
||||
}
|
||||
|
||||
/* Just return the last element pushed, without consuming it nor altering
|
||||
* the reference count. */
|
||||
exprtoken *exprStackPeek(exprstack *stack) {
|
||||
if (stack->numitems == 0) return NULL;
|
||||
return stack->items[stack->numitems-1];
|
||||
}
|
||||
|
||||
/* Free the stack structure state, including the items it contains, that are
|
||||
* assumed to be heap allocated. The passed pointer itself is not freed. */
|
||||
void exprStackFree(exprstack *stack) {
|
||||
for (int j = 0; j < stack->numitems; j++)
|
||||
exprTokenRelease(stack->items[j]);
|
||||
RedisModule_Free(stack->items);
|
||||
}
|
||||
|
||||
/* Just reset the stack removing all the items, but leaving it in a state
|
||||
* that makes it still usable for new elements. */
|
||||
void exprStackReset(exprstack *stack) {
|
||||
for (int j = 0; j < stack->numitems; j++)
|
||||
exprTokenRelease(stack->items[j]);
|
||||
stack->numitems = 0;
|
||||
}
|
||||
|
||||
/* =========================== Expression compilation ======================= */
|
||||
|
||||
void exprConsumeSpaces(exprstate *es) {
|
||||
while(es->p[0] && isspace(es->p[0])) es->p++;
|
||||
}
|
||||
|
||||
/* Parse an operator, trying to match the longer match in the
|
||||
* operators table. */
|
||||
exprtoken *exprParseOperator(exprstate *es) {
|
||||
exprtoken *t = exprNewToken(EXPR_TOKEN_OP);
|
||||
char *start = es->p;
|
||||
|
||||
while(es->p[0] &&
|
||||
(isalpha(es->p[0]) ||
|
||||
strchr(EXPR_OP_SPECIALCHARS,es->p[0]) != NULL))
|
||||
{
|
||||
es->p++;
|
||||
}
|
||||
|
||||
int matchlen = es->p - start;
|
||||
int bestlen = 0;
|
||||
int j;
|
||||
|
||||
// Find the longest matching operator.
|
||||
for (j = 0; ExprOptable[j].opname != NULL; j++) {
|
||||
if (ExprOptable[j].oplen > matchlen) continue;
|
||||
if (memcmp(ExprOptable[j].opname, start, ExprOptable[j].oplen) != 0)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if (ExprOptable[j].oplen > bestlen) {
|
||||
t->opcode = ExprOptable[j].opcode;
|
||||
bestlen = ExprOptable[j].oplen;
|
||||
}
|
||||
}
|
||||
if (bestlen == 0) {
|
||||
exprTokenRelease(t);
|
||||
return NULL;
|
||||
} else {
|
||||
es->p = start + bestlen;
|
||||
}
|
||||
return t;
|
||||
}
|
||||
|
||||
// Valid selector charset.
|
||||
static int is_selector_char(int c) {
|
||||
return (isalpha(c) ||
|
||||
isdigit(c) ||
|
||||
strchr(EXPR_SELECTOR_SPECIALCHARS,c) != NULL);
|
||||
}
|
||||
|
||||
/* Parse selectors, they start with a dot and can have alphanumerical
|
||||
* or few special chars. */
|
||||
exprtoken *exprParseSelector(exprstate *es) {
|
||||
exprtoken *t = exprNewToken(EXPR_TOKEN_SELECTOR);
|
||||
es->p++; // Skip dot.
|
||||
char *start = es->p;
|
||||
|
||||
while(es->p[0] && is_selector_char(es->p[0])) es->p++;
|
||||
int matchlen = es->p - start;
|
||||
t->str.start = start;
|
||||
t->str.len = matchlen;
|
||||
return t;
|
||||
}
|
||||
|
||||
exprtoken *exprParseNumber(exprstate *es) {
|
||||
exprtoken *t = exprNewToken(EXPR_TOKEN_NUM);
|
||||
char num[64];
|
||||
int idx = 0;
|
||||
while(isdigit(es->p[0]) || es->p[0] == '.' || es->p[0] == 'e' ||
|
||||
es->p[0] == 'E' || (idx == 0 && es->p[0] == '-'))
|
||||
{
|
||||
if (idx >= (int)sizeof(num)-1) {
|
||||
exprTokenRelease(t);
|
||||
return NULL;
|
||||
}
|
||||
num[idx++] = es->p[0];
|
||||
es->p++;
|
||||
}
|
||||
num[idx] = 0;
|
||||
|
||||
char *endptr;
|
||||
t->num = strtod(num, &endptr);
|
||||
if (*endptr != '\0') {
|
||||
exprTokenRelease(t);
|
||||
return NULL;
|
||||
}
|
||||
return t;
|
||||
}
|
||||
|
||||
exprtoken *exprParseString(exprstate *es) {
|
||||
char quote = es->p[0]; /* Store the quote type (' or "). */
|
||||
es->p++; /* Skip opening quote. */
|
||||
|
||||
exprtoken *t = exprNewToken(EXPR_TOKEN_STR);
|
||||
t->str.start = es->p;
|
||||
|
||||
while(es->p[0] != '\0') {
|
||||
if (es->p[0] == '\\' && es->p[1] != '\0') {
|
||||
es->p += 2; // Skip escaped char.
|
||||
continue;
|
||||
}
|
||||
if (es->p[0] == quote) {
|
||||
t->str.len = es->p - t->str.start;
|
||||
es->p++; // Skip closing quote.
|
||||
return t;
|
||||
}
|
||||
es->p++;
|
||||
}
|
||||
/* If we reach here, string was not terminated. */
|
||||
exprTokenRelease(t);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Parse a tuple of the form [1, "foo", 42]. No nested tuples are
|
||||
* supported. This type is useful mostly to be used with the "IN"
|
||||
* operator. */
|
||||
exprtoken *exprParseTuple(exprstate *es) {
|
||||
exprtoken *t = exprNewToken(EXPR_TOKEN_TUPLE);
|
||||
t->tuple.ele = NULL;
|
||||
t->tuple.len = 0;
|
||||
es->p++; /* Skip opening '['. */
|
||||
|
||||
size_t allocated = 0;
|
||||
while(1) {
|
||||
exprConsumeSpaces(es);
|
||||
|
||||
/* Check for empty tuple or end. */
|
||||
if (es->p[0] == ']') {
|
||||
es->p++;
|
||||
break;
|
||||
}
|
||||
|
||||
/* Grow tuple array if needed. */
|
||||
if (t->tuple.len == allocated) {
|
||||
size_t newsize = allocated == 0 ? 4 : allocated * 2;
|
||||
exprtoken **newele = RedisModule_Realloc(t->tuple.ele,
|
||||
sizeof(exprtoken*) * newsize);
|
||||
t->tuple.ele = newele;
|
||||
allocated = newsize;
|
||||
}
|
||||
|
||||
/* Parse tuple element. */
|
||||
exprtoken *ele = NULL;
|
||||
if (isdigit(es->p[0]) || es->p[0] == '-') {
|
||||
ele = exprParseNumber(es);
|
||||
} else if (es->p[0] == '"' || es->p[0] == '\'') {
|
||||
ele = exprParseString(es);
|
||||
} else {
|
||||
exprTokenRelease(t);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Error parsing number/string? */
|
||||
if (ele == NULL) {
|
||||
exprTokenRelease(t);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Store element if no error was detected. */
|
||||
t->tuple.ele[t->tuple.len] = ele;
|
||||
t->tuple.len++;
|
||||
|
||||
/* Check for next element. */
|
||||
exprConsumeSpaces(es);
|
||||
if (es->p[0] == ']') {
|
||||
es->p++;
|
||||
break;
|
||||
}
|
||||
if (es->p[0] != ',') {
|
||||
exprTokenRelease(t);
|
||||
return NULL;
|
||||
}
|
||||
es->p++; /* Skip comma. */
|
||||
}
|
||||
return t;
|
||||
}
|
||||
|
||||
/* Deallocate the object returned by exprCompile(). */
|
||||
void exprFree(exprstate *es) {
|
||||
if (es == NULL) return;
|
||||
|
||||
/* Free the original expression string. */
|
||||
if (es->expr) RedisModule_Free(es->expr);
|
||||
|
||||
/* Free all stacks. */
|
||||
exprStackFree(&es->values_stack);
|
||||
exprStackFree(&es->ops_stack);
|
||||
exprStackFree(&es->tokens);
|
||||
exprStackFree(&es->program);
|
||||
|
||||
/* Free the state object itself. */
|
||||
RedisModule_Free(es);
|
||||
}
|
||||
|
||||
/* Split the provided expression into a stack of tokens. Returns
|
||||
* 0 on success, 1 on error. */
|
||||
int exprTokenize(exprstate *es, int *errpos) {
|
||||
/* Main parsing loop. */
|
||||
while(1) {
|
||||
exprConsumeSpaces(es);
|
||||
|
||||
/* Set a flag to see if we can consider the - part of the
|
||||
* number, or an operator. */
|
||||
int minus_is_number = 0; // By default is an operator.
|
||||
|
||||
exprtoken *last = exprStackPeek(&es->tokens);
|
||||
if (last == NULL) {
|
||||
/* If we are at the start of an expression, the minus is
|
||||
* considered a number. */
|
||||
minus_is_number = 1;
|
||||
} else if (last->token_type == EXPR_TOKEN_OP &&
|
||||
last->opcode != EXPR_OP_CPAREN)
|
||||
{
|
||||
/* Also, if the previous token was an operator, the minus
|
||||
* is considered a number, unless the previous operator is
|
||||
* a closing parens. In such case it's like (...) -5, or alike
|
||||
* and we want to emit an operator. */
|
||||
minus_is_number = 1;
|
||||
}
|
||||
|
||||
/* Parse based on the current character. */
|
||||
exprtoken *current = NULL;
|
||||
if (*es->p == '\0') {
|
||||
current = exprNewToken(EXPR_TOKEN_EOF);
|
||||
} else if (isdigit(*es->p) ||
|
||||
(minus_is_number && *es->p == '-' && isdigit(es->p[1])))
|
||||
{
|
||||
current = exprParseNumber(es);
|
||||
} else if (*es->p == '"' || *es->p == '\'') {
|
||||
current = exprParseString(es);
|
||||
} else if (*es->p == '.' && is_selector_char(es->p[1])) {
|
||||
current = exprParseSelector(es);
|
||||
} else if (isalpha(*es->p) || strchr(EXPR_OP_SPECIALCHARS, *es->p)) {
|
||||
current = exprParseOperator(es);
|
||||
} else if (*es->p == '[') {
|
||||
current = exprParseTuple(es);
|
||||
}
|
||||
|
||||
if (current == NULL) {
|
||||
if (errpos) *errpos = es->p - es->expr;
|
||||
return 1; // Syntax Error.
|
||||
}
|
||||
|
||||
/* Push the current token to tokens stack. */
|
||||
exprStackPush(&es->tokens, current);
|
||||
if (current->token_type == EXPR_TOKEN_EOF) break;
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Helper function to get operator precedence from the operator table. */
|
||||
int exprGetOpPrecedence(int opcode) {
|
||||
for (int i = 0; ExprOptable[i].opname != NULL; i++) {
|
||||
if (ExprOptable[i].opcode == opcode)
|
||||
return ExprOptable[i].precedence;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* Helper function to get operator arity from the operator table. */
|
||||
int exprGetOpArity(int opcode) {
|
||||
for (int i = 0; ExprOptable[i].opname != NULL; i++) {
|
||||
if (ExprOptable[i].opcode == opcode)
|
||||
return ExprOptable[i].arity;
|
||||
}
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* Process an operator during compilation. Returns 0 on success, 1 on error.
|
||||
* This function will retain a reference of the operator 'op' in case it
|
||||
* is pushed on the operators stack. */
|
||||
int exprProcessOperator(exprstate *es, exprtoken *op, int *stack_items, int *errpos) {
|
||||
if (op->opcode == EXPR_OP_OPAREN) {
|
||||
// This is just a marker for us. Do nothing.
|
||||
exprStackPush(&es->ops_stack, op);
|
||||
exprTokenRetain(op);
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (op->opcode == EXPR_OP_CPAREN) {
|
||||
/* Process operators until we find the matching opening parenthesis. */
|
||||
while (1) {
|
||||
exprtoken *top_op = exprStackPop(&es->ops_stack);
|
||||
if (top_op == NULL) {
|
||||
if (errpos) *errpos = op->offset;
|
||||
return 1;
|
||||
}
|
||||
|
||||
if (top_op->opcode == EXPR_OP_OPAREN) {
|
||||
/* Open parethesis found. Our work finished. */
|
||||
exprTokenRelease(top_op);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int arity = exprGetOpArity(top_op->opcode);
|
||||
if (*stack_items < arity) {
|
||||
exprTokenRelease(top_op);
|
||||
if (errpos) *errpos = top_op->offset;
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* Move the operator on the program stack. */
|
||||
exprStackPush(&es->program, top_op);
|
||||
*stack_items = *stack_items - arity + 1;
|
||||
}
|
||||
}
|
||||
|
||||
int curr_prec = exprGetOpPrecedence(op->opcode);
|
||||
|
||||
/* Process operators with higher or equal precedence. */
|
||||
while (1) {
|
||||
exprtoken *top_op = exprStackPeek(&es->ops_stack);
|
||||
if (top_op == NULL || top_op->opcode == EXPR_OP_OPAREN) break;
|
||||
|
||||
int top_prec = exprGetOpPrecedence(top_op->opcode);
|
||||
if (top_prec < curr_prec) break;
|
||||
/* Special case for **: only pop if precedence is strictly higher
|
||||
* so that the operator is right associative, that is:
|
||||
* 2 ** 3 ** 2 is evaluated as 2 ** (3 ** 2) == 512 instead
|
||||
* of (2 ** 3) ** 2 == 64. */
|
||||
if (op->opcode == EXPR_OP_POW && top_prec <= curr_prec) break;
|
||||
|
||||
/* Pop and add to program. */
|
||||
top_op = exprStackPop(&es->ops_stack);
|
||||
int arity = exprGetOpArity(top_op->opcode);
|
||||
if (*stack_items < arity) {
|
||||
exprTokenRelease(top_op);
|
||||
if (errpos) *errpos = top_op->offset;
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* Move to the program stack. */
|
||||
exprStackPush(&es->program, top_op);
|
||||
*stack_items = *stack_items - arity + 1;
|
||||
}
|
||||
|
||||
/* Push current operator. */
|
||||
exprStackPush(&es->ops_stack, op);
|
||||
exprTokenRetain(op);
|
||||
return 0;
|
||||
}
|
||||
|
||||
/* Compile the expression into a set of push-value and exec-operator
|
||||
* that exprRun() can execute. The function returns an expstate object
|
||||
* that can be used for execution of the program. On error, NULL
|
||||
* is returned, and optionally the position of the error into the
|
||||
* expression is returned by reference. */
|
||||
exprstate *exprCompile(char *expr, int *errpos) {
|
||||
/* Initialize expression state. */
|
||||
exprstate *es = RedisModule_Alloc(sizeof(exprstate));
|
||||
es->expr = RedisModule_Strdup(expr);
|
||||
es->p = es->expr;
|
||||
|
||||
/* Initialize all stacks. */
|
||||
exprStackInit(&es->values_stack);
|
||||
exprStackInit(&es->ops_stack);
|
||||
exprStackInit(&es->tokens);
|
||||
exprStackInit(&es->program);
|
||||
|
||||
/* Tokenization. */
|
||||
if (exprTokenize(es, errpos)) {
|
||||
exprFree(es);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
/* Compile the expression into a sequence of operations. */
|
||||
int stack_items = 0; // Track # of items that would be on the stack
|
||||
// during execution. This way we can detect arity
|
||||
// issues at compile time.
|
||||
|
||||
/* Process each token. */
|
||||
for (int i = 0; i < es->tokens.numitems; i++) {
|
||||
exprtoken *token = es->tokens.items[i];
|
||||
|
||||
if (token->token_type == EXPR_TOKEN_EOF) break;
|
||||
|
||||
/* Handle values (numbers, strings, selectors). */
|
||||
if (token->token_type == EXPR_TOKEN_NUM ||
|
||||
token->token_type == EXPR_TOKEN_STR ||
|
||||
token->token_type == EXPR_TOKEN_TUPLE ||
|
||||
token->token_type == EXPR_TOKEN_SELECTOR)
|
||||
{
|
||||
exprStackPush(&es->program, token);
|
||||
exprTokenRetain(token);
|
||||
stack_items++;
|
||||
continue;
|
||||
}
|
||||
|
||||
/* Handle operators. */
|
||||
if (token->token_type == EXPR_TOKEN_OP) {
|
||||
if (exprProcessOperator(es, token, &stack_items, errpos)) {
|
||||
exprFree(es);
|
||||
return NULL;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
/* Process remaining operators on the stack. */
|
||||
while (es->ops_stack.numitems > 0) {
|
||||
exprtoken *op = exprStackPop(&es->ops_stack);
|
||||
if (op->opcode == EXPR_OP_OPAREN) {
|
||||
if (errpos) *errpos = op->offset;
|
||||
exprTokenRelease(op);
|
||||
exprFree(es);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
int arity = exprGetOpArity(op->opcode);
|
||||
if (stack_items < arity) {
|
||||
if (errpos) *errpos = op->offset;
|
||||
exprTokenRelease(op);
|
||||
exprFree(es);
|
||||
return NULL;
|
||||
}
|
||||
|
||||
exprStackPush(&es->program, op);
|
||||
stack_items = stack_items - arity + 1;
|
||||
}
|
||||
|
||||
/* Verify that exactly one value would remain on the stack after
|
||||
* execution. We could also check that such value is a number, but this
|
||||
* would make the code more complex without much gains. */
|
||||
if (stack_items != 1) {
|
||||
if (errpos) {
|
||||
/* Point to the last token's offset for error reporting. */
|
||||
exprtoken *last = es->tokens.items[es->tokens.numitems - 1];
|
||||
*errpos = last->offset;
|
||||
}
|
||||
exprFree(es);
|
||||
return NULL;
|
||||
}
|
||||
return es;
|
||||
}
|
||||
|
||||
/* ============================ Expression execution ======================== */
|
||||
|
||||
/* Convert a token to its numeric value. For strings we attempt to parse them
|
||||
* as numbers, returning 0 if conversion fails. */
|
||||
double exprTokenToNum(exprtoken *t) {
|
||||
char buf[128];
|
||||
if (t->token_type == EXPR_TOKEN_NUM) {
|
||||
return t->num;
|
||||
} else if (t->token_type == EXPR_TOKEN_STR && t->str.len < sizeof(buf)) {
|
||||
memcpy(buf, t->str.start, t->str.len);
|
||||
buf[t->str.len] = '\0';
|
||||
char *endptr;
|
||||
double val = strtod(buf, &endptr);
|
||||
return *endptr == '\0' ? val : 0;
|
||||
} else {
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
/* Convert object to true/false (0 or 1) */
|
||||
double exprTokenToBool(exprtoken *t) {
|
||||
if (t->token_type == EXPR_TOKEN_NUM) {
|
||||
return t->num != 0;
|
||||
} else if (t->token_type == EXPR_TOKEN_STR && t->str.len == 0) {
|
||||
return 0; // Empty string are false, like in Javascript.
|
||||
} else {
|
||||
return 1; // Every non numerical type is true.
|
||||
}
|
||||
}
|
||||
|
||||
/* Compare two tokens. Returns true if they are equal. */
|
||||
int exprTokensEqual(exprtoken *a, exprtoken *b) {
|
||||
// If both are strings, do string comparison.
|
||||
if (a->token_type == EXPR_TOKEN_STR && b->token_type == EXPR_TOKEN_STR) {
|
||||
return a->str.len == b->str.len &&
|
||||
memcmp(a->str.start, b->str.start, a->str.len) == 0;
|
||||
}
|
||||
|
||||
// If both are numbers, do numeric comparison.
|
||||
if (a->token_type == EXPR_TOKEN_NUM && b->token_type == EXPR_TOKEN_NUM) {
|
||||
return a->num == b->num;
|
||||
}
|
||||
|
||||
// Mixed types - convert to numbers and compare.
|
||||
return exprTokenToNum(a) == exprTokenToNum(b);
|
||||
}
|
||||
|
||||
/* Convert a json object to an expression token. There is only
|
||||
* limited support for JSON arrays: they must be composed of
|
||||
* just numbers and strings. Returns NULL if the JSON object
|
||||
* cannot be converted. */
|
||||
exprtoken *exprJsonToToken(cJSON *js) {
|
||||
if (cJSON_IsNumber(js)) {
|
||||
exprtoken *obj = exprNewToken(EXPR_TOKEN_NUM);
|
||||
obj->num = cJSON_GetNumberValue(js);
|
||||
return obj;
|
||||
} else if (cJSON_IsString(js)) {
|
||||
exprtoken *obj = exprNewToken(EXPR_TOKEN_STR);
|
||||
char *strval = cJSON_GetStringValue(js);
|
||||
obj->str.heapstr = RedisModule_Strdup(strval);
|
||||
obj->str.start = obj->str.heapstr;
|
||||
obj->str.len = strlen(obj->str.heapstr);
|
||||
return obj;
|
||||
} else if (cJSON_IsBool(js)) {
|
||||
exprtoken *obj = exprNewToken(EXPR_TOKEN_NUM);
|
||||
obj->num = cJSON_IsTrue(js);
|
||||
return obj;
|
||||
} else if (cJSON_IsArray(js)) {
|
||||
// First, scan the array to ensure it only
|
||||
// contains strings and numbers. Otherwise the
|
||||
// expression will evaluate to false.
|
||||
int array_size = cJSON_GetArraySize(js);
|
||||
|
||||
for (int j = 0; j < array_size; j++) {
|
||||
cJSON *item = cJSON_GetArrayItem(js, j);
|
||||
if (!cJSON_IsNumber(item) && !cJSON_IsString(item)) return NULL;
|
||||
}
|
||||
|
||||
// Create a tuple token for the array.
|
||||
exprtoken *obj = exprNewToken(EXPR_TOKEN_TUPLE);
|
||||
obj->tuple.len = array_size;
|
||||
obj->tuple.ele = NULL;
|
||||
if (obj->tuple.len == 0) return obj; // No elements, already ok.
|
||||
|
||||
obj->tuple.ele =
|
||||
RedisModule_Alloc(sizeof(exprtoken*) * obj->tuple.len);
|
||||
|
||||
// Convert each array element to a token.
|
||||
for (size_t j = 0; j < obj->tuple.len; j++) {
|
||||
cJSON *item = cJSON_GetArrayItem(js, j);
|
||||
if (cJSON_IsNumber(item)) {
|
||||
exprtoken *eleToken = exprNewToken(EXPR_TOKEN_NUM);
|
||||
eleToken->num = cJSON_GetNumberValue(item);
|
||||
obj->tuple.ele[j] = eleToken;
|
||||
} else if (cJSON_IsString(item)) {
|
||||
exprtoken *eleToken = exprNewToken(EXPR_TOKEN_STR);
|
||||
char *strval = cJSON_GetStringValue(item);
|
||||
eleToken->str.heapstr = RedisModule_Strdup(strval);
|
||||
eleToken->str.start = eleToken->str.heapstr;
|
||||
eleToken->str.len = strlen(eleToken->str.heapstr);
|
||||
obj->tuple.ele[j] = eleToken;
|
||||
}
|
||||
}
|
||||
return obj;
|
||||
}
|
||||
return NULL; // No conversion possible for this type.
|
||||
}
|
||||
|
||||
/* Execute the compiled expression program. Returns 1 if the final stack value
|
||||
* evaluates to true, 0 otherwise. Also returns 0 if any selector callback
|
||||
* fails. */
|
||||
int exprRun(exprstate *es, char *json, size_t json_len) {
|
||||
exprStackReset(&es->values_stack);
|
||||
cJSON *parsed_json = NULL;
|
||||
|
||||
// Execute each instruction in the program.
|
||||
for (int i = 0; i < es->program.numitems; i++) {
|
||||
exprtoken *t = es->program.items[i];
|
||||
|
||||
// Handle selectors by calling the callback.
|
||||
if (t->token_type == EXPR_TOKEN_SELECTOR) {
|
||||
if (json != NULL) {
|
||||
cJSON *attrib = NULL;
|
||||
if (parsed_json == NULL) {
|
||||
parsed_json = cJSON_ParseWithLength(json,json_len);
|
||||
// Will be left to NULL if the above fails.
|
||||
}
|
||||
if (parsed_json) {
|
||||
char item_name[128];
|
||||
if (t->str.len > 0 && t->str.len < sizeof(item_name)) {
|
||||
memcpy(item_name,t->str.start,t->str.len);
|
||||
item_name[t->str.len] = 0;
|
||||
attrib = cJSON_GetObjectItem(parsed_json,item_name);
|
||||
}
|
||||
/* Fill the token according to the JSON type stored
|
||||
* at the attribute. */
|
||||
if (attrib) {
|
||||
exprtoken *obj = exprJsonToToken(attrib);
|
||||
if (obj) {
|
||||
exprStackPush(&es->values_stack, obj);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Selector not found or JSON object not convertible to
|
||||
// expression tokens. Evaluate the expression to false.
|
||||
if (parsed_json) cJSON_Delete(parsed_json);
|
||||
return 0;
|
||||
}
|
||||
|
||||
// Push non-operator values directly onto the stack.
|
||||
if (t->token_type != EXPR_TOKEN_OP) {
|
||||
exprStackPush(&es->values_stack, t);
|
||||
exprTokenRetain(t);
|
||||
continue;
|
||||
}
|
||||
|
||||
// Handle operators.
|
||||
exprtoken *result = exprNewToken(EXPR_TOKEN_NUM);
|
||||
|
||||
// Pop operands - we know we have enough from compile-time checks.
|
||||
exprtoken *b = exprStackPop(&es->values_stack);
|
||||
exprtoken *a = NULL;
|
||||
if (exprGetOpArity(t->opcode) == 2) {
|
||||
a = exprStackPop(&es->values_stack);
|
||||
}
|
||||
|
||||
switch(t->opcode) {
|
||||
case EXPR_OP_NOT:
|
||||
result->num = exprTokenToBool(b) == 0 ? 1 : 0;
|
||||
break;
|
||||
case EXPR_OP_POW: {
|
||||
double base = exprTokenToNum(a);
|
||||
double exp = exprTokenToNum(b);
|
||||
result->num = pow(base, exp);
|
||||
break;
|
||||
}
|
||||
case EXPR_OP_MULT:
|
||||
result->num = exprTokenToNum(a) * exprTokenToNum(b);
|
||||
break;
|
||||
case EXPR_OP_DIV:
|
||||
result->num = exprTokenToNum(a) / exprTokenToNum(b);
|
||||
break;
|
||||
case EXPR_OP_MOD: {
|
||||
double va = exprTokenToNum(a);
|
||||
double vb = exprTokenToNum(b);
|
||||
result->num = fmod(va, vb);
|
||||
break;
|
||||
}
|
||||
case EXPR_OP_SUM:
|
||||
result->num = exprTokenToNum(a) + exprTokenToNum(b);
|
||||
break;
|
||||
case EXPR_OP_DIFF:
|
||||
result->num = exprTokenToNum(a) - exprTokenToNum(b);
|
||||
break;
|
||||
case EXPR_OP_GT:
|
||||
result->num = exprTokenToNum(a) > exprTokenToNum(b) ? 1 : 0;
|
||||
break;
|
||||
case EXPR_OP_GTE:
|
||||
result->num = exprTokenToNum(a) >= exprTokenToNum(b) ? 1 : 0;
|
||||
break;
|
||||
case EXPR_OP_LT:
|
||||
result->num = exprTokenToNum(a) < exprTokenToNum(b) ? 1 : 0;
|
||||
break;
|
||||
case EXPR_OP_LTE:
|
||||
result->num = exprTokenToNum(a) <= exprTokenToNum(b) ? 1 : 0;
|
||||
break;
|
||||
case EXPR_OP_EQ:
|
||||
result->num = exprTokensEqual(a, b) ? 1 : 0;
|
||||
break;
|
||||
case EXPR_OP_NEQ:
|
||||
result->num = !exprTokensEqual(a, b) ? 1 : 0;
|
||||
break;
|
||||
case EXPR_OP_IN: {
|
||||
// For 'in' operator, b must be a tuple.
|
||||
result->num = 0; // Default to false.
|
||||
if (b->token_type == EXPR_TOKEN_TUPLE) {
|
||||
for (size_t j = 0; j < b->tuple.len; j++) {
|
||||
if (exprTokensEqual(a, b->tuple.ele[j])) {
|
||||
result->num = 1; // Found a match.
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
case EXPR_OP_AND:
|
||||
result->num =
|
||||
exprTokenToBool(a) != 0 && exprTokenToBool(b) != 0 ? 1 : 0;
|
||||
break;
|
||||
case EXPR_OP_OR:
|
||||
result->num =
|
||||
exprTokenToBool(a) != 0 || exprTokenToBool(b) != 0 ? 1 : 0;
|
||||
break;
|
||||
default:
|
||||
// Do nothing: we don't want runtime errors.
|
||||
break;
|
||||
}
|
||||
|
||||
// Free operands and push result.
|
||||
if (a) exprTokenRelease(a);
|
||||
exprTokenRelease(b);
|
||||
exprStackPush(&es->values_stack, result);
|
||||
}
|
||||
|
||||
if (parsed_json) cJSON_Delete(parsed_json);
|
||||
|
||||
// Get final result from stack.
|
||||
exprtoken *final = exprStackPop(&es->values_stack);
|
||||
if (final == NULL) return 0;
|
||||
|
||||
// Convert result to boolean.
|
||||
int retval = exprTokenToBool(final);
|
||||
exprTokenRelease(final);
|
||||
return retval;
|
||||
}
|
||||
|
||||
/* ============================ Simple test main ============================ */
|
||||
|
||||
#ifdef TEST_MAIN
|
||||
void exprPrintToken(exprtoken *t) {
|
||||
switch(t->token_type) {
|
||||
case EXPR_TOKEN_EOF:
|
||||
printf("EOF");
|
||||
break;
|
||||
case EXPR_TOKEN_NUM:
|
||||
printf("NUM:%g", t->num);
|
||||
break;
|
||||
case EXPR_TOKEN_STR:
|
||||
printf("STR:\"%.*s\"", (int)t->str.len, t->str.start);
|
||||
break;
|
||||
case EXPR_TOKEN_SELECTOR:
|
||||
printf("SEL:%.*s", (int)t->str.len, t->str.start);
|
||||
break;
|
||||
case EXPR_TOKEN_OP:
|
||||
printf("OP:");
|
||||
for (int i = 0; ExprOptable[i].opname != NULL; i++) {
|
||||
if (ExprOptable[i].opcode == t->opcode) {
|
||||
printf("%s", ExprOptable[i].opname);
|
||||
break;
|
||||
}
|
||||
}
|
||||
break;
|
||||
default:
|
||||
printf("UNKNOWN");
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
void exprPrintStack(exprstack *stack, const char *name) {
|
||||
printf("%s (%d items):", name, stack->numitems);
|
||||
for (int j = 0; j < stack->numitems; j++) {
|
||||
printf(" ");
|
||||
exprPrintToken(stack->items[j]);
|
||||
}
|
||||
printf("\n");
|
||||
}
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
char *testexpr = "(5+2)*3 and .year > 1980 and 'foo' == 'foo'";
|
||||
char *testjson = "{\"year\": 1984, \"name\": \"The Matrix\"}";
|
||||
if (argc >= 2) testexpr = argv[1];
|
||||
if (argc >= 3) testjson = argv[2];
|
||||
|
||||
printf("Compiling expression: %s\n", testexpr);
|
||||
|
||||
int errpos = 0;
|
||||
exprstate *es = exprCompile(testexpr,&errpos);
|
||||
if (es == NULL) {
|
||||
printf("Compilation failed near \"...%s\"\n", testexpr+errpos);
|
||||
return 1;
|
||||
}
|
||||
|
||||
exprPrintStack(&es->tokens, "Tokens");
|
||||
exprPrintStack(&es->program, "Program");
|
||||
printf("Running against object: %s\n", testjson);
|
||||
int result = exprRun(es,testjson,strlen(testjson));
|
||||
printf("Result1: %s\n", result ? "True" : "False");
|
||||
result = exprRun(es,testjson,strlen(testjson));
|
||||
printf("Result2: %s\n", result ? "True" : "False");
|
||||
|
||||
exprFree(es);
|
||||
return 0;
|
||||
}
|
||||
#endif
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,188 @@
|
||||
/*
|
||||
* HNSW (Hierarchical Navigable Small World) Implementation
|
||||
* Based on the paper by Yu. A. Malkov, D. A. Yashunin
|
||||
*
|
||||
* Copyright (c) 2009-Present, Redis Ltd.
|
||||
* All rights reserved.
|
||||
*
|
||||
* Licensed under your choice of the Redis Source Available License 2.0
|
||||
* (RSALv2) or the Server Side Public License v1 (SSPLv1).
|
||||
* Originally authored by: Salvatore Sanfilippo.
|
||||
*/
|
||||
|
||||
#ifndef HNSW_H
|
||||
#define HNSW_H
|
||||
|
||||
#include <pthread.h>
|
||||
#include <stdatomic.h>
|
||||
|
||||
#define HNSW_DEFAULT_M 16 /* Used when 0 is given at creation time. */
|
||||
#define HNSW_MIN_M 4 /* Probably even too low already. */
|
||||
#define HNSW_MAX_M 4096 /* Safeguard sanity limit. */
|
||||
#define HNSW_MAX_THREADS 32 /* Maximum number of concurrent threads */
|
||||
|
||||
/* Quantization types you can enable at creation time in hnsw_new() */
|
||||
#define HNSW_QUANT_NONE 0 // No quantization.
|
||||
#define HNSW_QUANT_Q8 1 // Q8 quantization.
|
||||
#define HNSW_QUANT_BIN 2 // Binary quantization.
|
||||
|
||||
/* Layer structure for HNSW nodes. Each node will have from one to a few
|
||||
* of this depending on its level. */
|
||||
typedef struct {
|
||||
struct hnswNode **links; /* Array of neighbors for this layer */
|
||||
uint32_t num_links; /* Number of used links */
|
||||
uint32_t max_links; /* Maximum links for this layer. We may
|
||||
* reallocate the node in very particular
|
||||
* conditions in order to allow linking of
|
||||
* new inserted nodes, so this may change
|
||||
* dynamically and be > M*2 for a small set of
|
||||
* nodes. */
|
||||
float worst_distance; /* Distance to the worst neighbor */
|
||||
uint32_t worst_idx; /* Index of the worst neighbor */
|
||||
} hnswNodeLayer;
|
||||
|
||||
/* Node structure for HNSW graph */
|
||||
typedef struct hnswNode {
|
||||
uint32_t level; /* Node's maximum level */
|
||||
uint64_t id; /* Unique identifier, may be useful in order to
|
||||
* have a bitmap of visited notes to use as
|
||||
* alternative to epoch / visited_epoch.
|
||||
* Also used in serialization in order to retain
|
||||
* links specifying IDs. */
|
||||
void *vector; /* The vector, quantized or not. */
|
||||
float quants_range; /* Quantization range for this vector:
|
||||
* min/max values will be in the range
|
||||
* -quants_range, +quants_range */
|
||||
float l2; /* L2 before normalization. */
|
||||
|
||||
/* Last time (epoch) this node was visited. We need one per thread.
|
||||
* This avoids having a different data structure where we track
|
||||
* visited nodes, but costs memory per node. */
|
||||
uint64_t visited_epoch[HNSW_MAX_THREADS];
|
||||
|
||||
void *value; /* Associated value */
|
||||
struct hnswNode *prev, *next; /* Prev/Next node in the list starting at
|
||||
* HNSW->head. */
|
||||
|
||||
/* Links (and links info) per each layer. Note that this is part
|
||||
* of the node allocation to be more cache friendly: reliable 3% speedup
|
||||
* on Apple silicon, and does not make anything more complex. */
|
||||
hnswNodeLayer layers[];
|
||||
} hnswNode;
|
||||
|
||||
struct HNSW;
|
||||
|
||||
/* It is possible to navigate an HNSW with a cursor that guarantees
|
||||
* visiting all the elements that remain in the HNSW from the start to the
|
||||
* end of the process (but not the new ones, so that the process will
|
||||
* eventually finish). Check hnsw_cursor_init(), hnsw_cursor_next() and
|
||||
* hnsw_cursor_free(). */
|
||||
typedef struct hnswCursor {
|
||||
struct HNSW *index; // Reference to the index of this cursor.
|
||||
hnswNode *current; // Element to report when hnsw_cursor_next() is called.
|
||||
struct hnswCursor *next; // Next cursor active.
|
||||
} hnswCursor;
|
||||
|
||||
/* Main HNSW index structure */
|
||||
typedef struct HNSW {
|
||||
hnswNode *enter_point; /* Entry point for the graph */
|
||||
uint32_t M; /* M as in the paper: layer 0 has M*2 max
|
||||
neighbors (M populated at insertion time)
|
||||
while all the other layers have M neighbors. */
|
||||
uint32_t max_level; /* Current maximum level in the graph */
|
||||
uint32_t vector_dim; /* Dimensionality of stored vectors */
|
||||
uint64_t node_count; /* Total number of nodes */
|
||||
_Atomic uint64_t last_id; /* Last node ID used */
|
||||
uint64_t current_epoch[HNSW_MAX_THREADS]; /* Current epoch for visit tracking */
|
||||
hnswNode *head; /* Linked list of nodes. Last first */
|
||||
|
||||
/* We have two locks here:
|
||||
* 1. A global_lock that is used to perform write operations blocking all
|
||||
* the readers.
|
||||
* 2. One mutex per epoch slot, in order for read operations to acquire
|
||||
* a lock on a specific slot to use epochs tracking of visited nodes. */
|
||||
pthread_rwlock_t global_lock; /* Global read-write lock */
|
||||
pthread_mutex_t slot_locks[HNSW_MAX_THREADS]; /* Per-slot locks */
|
||||
|
||||
_Atomic uint32_t next_slot; /* Next thread slot to try */
|
||||
_Atomic uint64_t version; /* Version for optimistic concurrency, this is
|
||||
* incremented on deletions and entry point
|
||||
* updates. */
|
||||
uint32_t quant_type; /* Quantization used. HNSW_QUANT_... */
|
||||
hnswCursor *cursors;
|
||||
} HNSW;
|
||||
|
||||
/* Serialized node. This structure is used as return value of
|
||||
* hnsw_serialize_node(). */
|
||||
typedef struct hnswSerNode {
|
||||
void *vector;
|
||||
uint32_t vector_size;
|
||||
uint64_t *params;
|
||||
uint32_t params_count;
|
||||
} hnswSerNode;
|
||||
|
||||
/* Insert preparation context */
|
||||
typedef struct InsertContext InsertContext;
|
||||
|
||||
/* Core HNSW functions */
|
||||
HNSW *hnsw_new(uint32_t vector_dim, uint32_t quant_type, uint32_t m);
|
||||
void hnsw_free(HNSW *index,void(*free_value)(void*value));
|
||||
void hnsw_node_free(hnswNode *node);
|
||||
void hnsw_print_stats(HNSW *index);
|
||||
hnswNode *hnsw_insert(HNSW *index, const float *vector, const int8_t *qvector,
|
||||
float qrange, uint64_t id, void *value, int ef);
|
||||
int hnsw_search(HNSW *index, const float *query, uint32_t k,
|
||||
hnswNode **neighbors, float *distances, uint32_t slot,
|
||||
int query_vector_is_normalized);
|
||||
int hnsw_search_with_filter
|
||||
(HNSW *index, const float *query_vector, uint32_t k,
|
||||
hnswNode **neighbors, float *distances, uint32_t slot,
|
||||
int query_vector_is_normalized,
|
||||
int (*filter_callback)(void *value, void *privdata),
|
||||
void *filter_privdata, uint32_t max_candidates);
|
||||
void hnsw_get_node_vector(HNSW *index, hnswNode *node, float *vec);
|
||||
int hnsw_delete_node(HNSW *index, hnswNode *node, void(*free_value)(void*value));
|
||||
hnswNode *hnsw_random_node(HNSW *index, int slot);
|
||||
|
||||
/* Thread safety functions. */
|
||||
int hnsw_acquire_read_slot(HNSW *index);
|
||||
void hnsw_release_read_slot(HNSW *index, int slot);
|
||||
|
||||
/* Optimistic insertion API. */
|
||||
InsertContext *hnsw_prepare_insert(HNSW *index, const float *vector, const int8_t *qvector, float qrange, uint64_t id, int ef);
|
||||
hnswNode *hnsw_try_commit_insert(HNSW *index, InsertContext *ctx, void *value);
|
||||
void hnsw_free_insert_context(InsertContext *ctx);
|
||||
|
||||
/* Serialization. */
|
||||
hnswSerNode *hnsw_serialize_node(HNSW *index, hnswNode *node);
|
||||
void hnsw_free_serialized_node(hnswSerNode *sn);
|
||||
hnswNode *hnsw_insert_serialized(HNSW *index, void *vector, uint64_t *params, uint32_t params_len, void *value);
|
||||
int hnsw_deserialize_index(HNSW *index);
|
||||
|
||||
// Helper function in case the user wants to directly copy
|
||||
// the vector bytes.
|
||||
uint32_t hnsw_quants_bytes(HNSW *index);
|
||||
|
||||
/* Cursors. */
|
||||
hnswCursor *hnsw_cursor_init(HNSW *index);
|
||||
void hnsw_cursor_free(hnswCursor *cursor);
|
||||
hnswNode *hnsw_cursor_next(hnswCursor *cursor);
|
||||
int hnsw_cursor_acquire_lock(hnswCursor *cursor);
|
||||
void hnsw_cursor_release_lock(hnswCursor *cursor);
|
||||
|
||||
/* Allocator selection. */
|
||||
void hnsw_set_allocator(void (*free_ptr)(void*), void *(*malloc_ptr)(size_t),
|
||||
void *(*realloc_ptr)(void*, size_t));
|
||||
|
||||
/* Testing. */
|
||||
int hnsw_validate_graph(HNSW *index, uint64_t *connected_nodes, int *reciprocal_links);
|
||||
void hnsw_test_graph_recall(HNSW *index, int test_ef, int verbose);
|
||||
float hnsw_distance(HNSW *index, hnswNode *a, hnswNode *b);
|
||||
int hnsw_ground_truth_with_filter
|
||||
(HNSW *index, const float *query_vector, uint32_t k,
|
||||
hnswNode **neighbors, float *distances, uint32_t slot,
|
||||
int query_vector_is_normalized,
|
||||
int (*filter_callback)(void *value, void *privdata),
|
||||
void *filter_privdata);
|
||||
|
||||
#endif /* HNSW_H */
|
||||
Executable
+230
@@ -0,0 +1,230 @@
|
||||
#!/usr/bin/env python3
|
||||
#
|
||||
# Vector set tests.
|
||||
# A Redis instance should be running in the default port.
|
||||
#
|
||||
# Copyright (c) 2009-Present, Redis Ltd.
|
||||
# All rights reserved.
|
||||
#
|
||||
# Licensed under your choice of the Redis Source Available License 2.0
|
||||
# (RSALv2) or the Server Side Public License v1 (SSPLv1).
|
||||
#
|
||||
|
||||
#!/usr/bin/env python3
|
||||
import redis
|
||||
import random
|
||||
import struct
|
||||
import math
|
||||
import time
|
||||
import sys
|
||||
import os
|
||||
import importlib
|
||||
import inspect
|
||||
from typing import List, Tuple, Optional
|
||||
from dataclasses import dataclass
|
||||
|
||||
def colored(text: str, color: str) -> str:
|
||||
colors = {
|
||||
'red': '\033[91m',
|
||||
'green': '\033[92m'
|
||||
}
|
||||
reset = '\033[0m'
|
||||
return f"{colors.get(color, '')}{text}{reset}"
|
||||
|
||||
@dataclass
|
||||
class VectorData:
|
||||
vectors: List[List[float]]
|
||||
names: List[str]
|
||||
|
||||
def find_k_nearest(self, query_vector: List[float], k: int) -> List[Tuple[str, float]]:
|
||||
"""Find k-nearest neighbors using the same scoring as Redis VSIM WITHSCORES."""
|
||||
similarities = []
|
||||
query_norm = math.sqrt(sum(x*x for x in query_vector))
|
||||
if query_norm == 0:
|
||||
return []
|
||||
|
||||
for i, vec in enumerate(self.vectors):
|
||||
vec_norm = math.sqrt(sum(x*x for x in vec))
|
||||
if vec_norm == 0:
|
||||
continue
|
||||
|
||||
dot_product = sum(a*b for a,b in zip(query_vector, vec))
|
||||
cosine_sim = dot_product / (query_norm * vec_norm)
|
||||
distance = 1.0 - cosine_sim
|
||||
redis_similarity = 1.0 - (distance/2.0)
|
||||
similarities.append((self.names[i], redis_similarity))
|
||||
|
||||
similarities.sort(key=lambda x: x[1], reverse=True)
|
||||
return similarities[:k]
|
||||
|
||||
def generate_random_vector(dim: int) -> List[float]:
|
||||
"""Generate a random normalized vector."""
|
||||
vec = [random.gauss(0, 1) for _ in range(dim)]
|
||||
norm = math.sqrt(sum(x*x for x in vec))
|
||||
return [x/norm for x in vec]
|
||||
|
||||
def fill_redis_with_vectors(r: redis.Redis, key: str, count: int, dim: int,
|
||||
with_reduce: Optional[int] = None) -> VectorData:
|
||||
"""Fill Redis with random vectors and return a VectorData object for verification."""
|
||||
vectors = []
|
||||
names = []
|
||||
|
||||
r.delete(key)
|
||||
for i in range(count):
|
||||
vec = generate_random_vector(dim)
|
||||
name = f"{key}:item:{i}"
|
||||
vectors.append(vec)
|
||||
names.append(name)
|
||||
|
||||
vec_bytes = struct.pack(f'{dim}f', *vec)
|
||||
args = [key]
|
||||
if with_reduce:
|
||||
args.extend(['REDUCE', with_reduce])
|
||||
args.extend(['FP32', vec_bytes, name])
|
||||
r.execute_command('VADD', *args)
|
||||
|
||||
return VectorData(vectors=vectors, names=names)
|
||||
|
||||
class TestCase:
|
||||
def __init__(self):
|
||||
self.error_msg = None
|
||||
self.error_details = None
|
||||
self.test_key = f"test:{self.__class__.__name__.lower()}"
|
||||
# Primary Redis instance (default port)
|
||||
self.redis = redis.Redis()
|
||||
# Replica Redis instance (port 6380)
|
||||
self.replica = redis.Redis(port=6380)
|
||||
# Replication status
|
||||
self.replication_setup = False
|
||||
|
||||
def setup(self):
|
||||
self.redis.delete(self.test_key)
|
||||
|
||||
def teardown(self):
|
||||
self.redis.delete(self.test_key)
|
||||
|
||||
def setup_replication(self) -> bool:
|
||||
"""
|
||||
Setup replication between primary and replica Redis instances.
|
||||
Returns True if replication is successfully established, False otherwise.
|
||||
"""
|
||||
# Configure replica to replicate from primary
|
||||
self.replica.execute_command('REPLICAOF', '127.0.0.1', 6379)
|
||||
|
||||
# Wait for replication to be established
|
||||
max_attempts = 10
|
||||
for attempt in range(max_attempts):
|
||||
# Check replication info
|
||||
repl_info = self.replica.info('replication')
|
||||
|
||||
# Check if replication is established
|
||||
if (repl_info.get('role') == 'slave' and
|
||||
repl_info.get('master_host') == '127.0.0.1' and
|
||||
repl_info.get('master_port') == 6379 and
|
||||
repl_info.get('master_link_status') == 'up'):
|
||||
|
||||
self.replication_setup = True
|
||||
return True
|
||||
|
||||
# Wait before next attempt
|
||||
time.sleep(0.5)
|
||||
|
||||
# If we get here, replication wasn't established
|
||||
self.error_msg = "Failed to establish replication between primary and replica"
|
||||
return False
|
||||
|
||||
def test(self):
|
||||
raise NotImplementedError("Subclasses must implement test method")
|
||||
|
||||
def run(self):
|
||||
try:
|
||||
self.setup()
|
||||
self.test()
|
||||
return True
|
||||
except AssertionError as e:
|
||||
self.error_msg = str(e)
|
||||
import traceback
|
||||
self.error_details = traceback.format_exc()
|
||||
return False
|
||||
except Exception as e:
|
||||
self.error_msg = f"Unexpected error: {str(e)}"
|
||||
import traceback
|
||||
self.error_details = traceback.format_exc()
|
||||
return False
|
||||
finally:
|
||||
self.teardown()
|
||||
|
||||
def getname(self):
|
||||
"""Each test class should override this to provide its name"""
|
||||
return self.__class__.__name__
|
||||
|
||||
def estimated_runtime(self):
|
||||
""""Each test class should override this if it takes a significant amount of time to run. Default is 100ms"""
|
||||
return 0.1
|
||||
|
||||
def find_test_classes():
|
||||
test_classes = []
|
||||
tests_dir = 'tests'
|
||||
|
||||
if not os.path.exists(tests_dir):
|
||||
return []
|
||||
|
||||
for file in os.listdir(tests_dir):
|
||||
if file.endswith('.py'):
|
||||
module_name = f"tests.{file[:-3]}"
|
||||
try:
|
||||
module = importlib.import_module(module_name)
|
||||
for name, obj in inspect.getmembers(module):
|
||||
if inspect.isclass(obj) and obj.__name__ != 'TestCase' and hasattr(obj, 'test'):
|
||||
test_classes.append(obj())
|
||||
except Exception as e:
|
||||
print(f"Error loading {file}: {e}")
|
||||
|
||||
return test_classes
|
||||
|
||||
def run_tests():
|
||||
print("================================================\n"+
|
||||
"Make sure to have Redis running in the localhost\n"+
|
||||
"with --enable-debug-command yes\n"+
|
||||
"Both primary (6379) and replica (6380) instances\n"+
|
||||
"================================================\n")
|
||||
|
||||
tests = find_test_classes()
|
||||
if not tests:
|
||||
print("No tests found!")
|
||||
return
|
||||
|
||||
# Sort tests by estimated runtime
|
||||
tests.sort(key=lambda t: t.estimated_runtime())
|
||||
|
||||
passed = 0
|
||||
total = len(tests)
|
||||
|
||||
for test in tests:
|
||||
print(f"{test.getname()}: ", end="")
|
||||
sys.stdout.flush()
|
||||
|
||||
start_time = time.time()
|
||||
success = test.run()
|
||||
duration = time.time() - start_time
|
||||
|
||||
if success:
|
||||
print(colored("OK", "green"), f"({duration:.2f}s)")
|
||||
passed += 1
|
||||
else:
|
||||
print(colored("ERR", "red"), f"({duration:.2f}s)")
|
||||
print(f"Error: {test.error_msg}")
|
||||
if test.error_details:
|
||||
print("\nTraceback:")
|
||||
print(test.error_details)
|
||||
|
||||
print("\n" + "="*50)
|
||||
print(f"\nTest Summary: {passed}/{total} tests passed")
|
||||
|
||||
if passed == total:
|
||||
print(colored("\nALL TESTS PASSED!", "green"))
|
||||
else:
|
||||
print(colored(f"\n{total-passed} TESTS FAILED!", "red"))
|
||||
|
||||
if __name__ == "__main__":
|
||||
run_tests()
|
||||
@@ -0,0 +1,21 @@
|
||||
from test import TestCase, generate_random_vector
|
||||
import struct
|
||||
|
||||
class BasicCommands(TestCase):
|
||||
def getname(self):
|
||||
return "VADD, VDIM, VCARD basic usage"
|
||||
|
||||
def test(self):
|
||||
# Test VADD
|
||||
vec = generate_random_vector(4)
|
||||
vec_bytes = struct.pack('4f', *vec)
|
||||
result = self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes, f'{self.test_key}:item:1')
|
||||
assert result == 1, "VADD should return 1 for first item"
|
||||
|
||||
# Test VDIM
|
||||
dim = self.redis.execute_command('VDIM', self.test_key)
|
||||
assert dim == 4, f"VDIM should return 4, got {dim}"
|
||||
|
||||
# Test VCARD
|
||||
card = self.redis.execute_command('VCARD', self.test_key)
|
||||
assert card == 1, f"VCARD should return 1, got {card}"
|
||||
@@ -0,0 +1,35 @@
|
||||
from test import TestCase
|
||||
|
||||
class BasicSimilarity(TestCase):
|
||||
def getname(self):
|
||||
return "VSIM reported distance makes sense with 4D vectors"
|
||||
|
||||
def test(self):
|
||||
# Add two very similar vectors, one different
|
||||
vec1 = [1, 0, 0, 0]
|
||||
vec2 = [0.99, 0.01, 0, 0]
|
||||
vec3 = [0.1, 1, -1, 0.5]
|
||||
|
||||
# Add vectors using VALUES format
|
||||
self.redis.execute_command('VADD', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1], f'{self.test_key}:item:1')
|
||||
self.redis.execute_command('VADD', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec2], f'{self.test_key}:item:2')
|
||||
self.redis.execute_command('VADD', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec3], f'{self.test_key}:item:3')
|
||||
|
||||
# Query similarity with vec1
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1], 'WITHSCORES')
|
||||
|
||||
# Convert results to dictionary
|
||||
results_dict = {}
|
||||
for i in range(0, len(result), 2):
|
||||
key = result[i].decode()
|
||||
score = float(result[i+1])
|
||||
results_dict[key] = score
|
||||
|
||||
# Verify results
|
||||
assert results_dict[f'{self.test_key}:item:1'] > 0.99, "Self-similarity should be very high"
|
||||
assert results_dict[f'{self.test_key}:item:2'] > 0.99, "Similar vector should have high similarity"
|
||||
assert results_dict[f'{self.test_key}:item:3'] < 0.8, "Not very similar vector should have low similarity"
|
||||
@@ -0,0 +1,156 @@
|
||||
from test import TestCase, generate_random_vector
|
||||
import threading
|
||||
import time
|
||||
import struct
|
||||
|
||||
class ThreadingStressTest(TestCase):
|
||||
def getname(self):
|
||||
return "Concurrent VADD/DEL/VSIM operations stress test"
|
||||
|
||||
def estimated_runtime(self):
|
||||
return 10 # Test runs for 10 seconds
|
||||
|
||||
def test(self):
|
||||
# Constants - easy to modify if needed
|
||||
NUM_VADD_THREADS = 10
|
||||
NUM_VSIM_THREADS = 1
|
||||
NUM_DEL_THREADS = 1
|
||||
TEST_DURATION = 10 # seconds
|
||||
VECTOR_DIM = 100
|
||||
DEL_INTERVAL = 1 # seconds
|
||||
|
||||
# Shared flags and state
|
||||
stop_event = threading.Event()
|
||||
error_list = []
|
||||
error_lock = threading.Lock()
|
||||
|
||||
def log_error(thread_name, error):
|
||||
with error_lock:
|
||||
error_list.append(f"{thread_name}: {error}")
|
||||
|
||||
def vadd_worker(thread_id):
|
||||
"""Thread function to perform VADD operations"""
|
||||
thread_name = f"VADD-{thread_id}"
|
||||
try:
|
||||
vector_count = 0
|
||||
while not stop_event.is_set():
|
||||
try:
|
||||
# Generate random vector
|
||||
vec = generate_random_vector(VECTOR_DIM)
|
||||
vec_bytes = struct.pack(f'{VECTOR_DIM}f', *vec)
|
||||
|
||||
# Add vector with CAS option
|
||||
self.redis.execute_command(
|
||||
'VADD',
|
||||
self.test_key,
|
||||
'FP32',
|
||||
vec_bytes,
|
||||
f'{self.test_key}:item:{thread_id}:{vector_count}',
|
||||
'CAS'
|
||||
)
|
||||
|
||||
vector_count += 1
|
||||
|
||||
# Small sleep to reduce CPU pressure
|
||||
if vector_count % 10 == 0:
|
||||
time.sleep(0.001)
|
||||
except Exception as e:
|
||||
log_error(thread_name, f"Error: {str(e)}")
|
||||
time.sleep(0.1) # Slight backoff on error
|
||||
except Exception as e:
|
||||
log_error(thread_name, f"Thread error: {str(e)}")
|
||||
|
||||
def del_worker():
|
||||
"""Thread function that deletes the key periodically"""
|
||||
thread_name = "DEL"
|
||||
try:
|
||||
del_count = 0
|
||||
while not stop_event.is_set():
|
||||
try:
|
||||
# Sleep first, then delete
|
||||
time.sleep(DEL_INTERVAL)
|
||||
if stop_event.is_set():
|
||||
break
|
||||
|
||||
self.redis.delete(self.test_key)
|
||||
del_count += 1
|
||||
except Exception as e:
|
||||
log_error(thread_name, f"Error: {str(e)}")
|
||||
except Exception as e:
|
||||
log_error(thread_name, f"Thread error: {str(e)}")
|
||||
|
||||
def vsim_worker(thread_id):
|
||||
"""Thread function to perform VSIM operations"""
|
||||
thread_name = f"VSIM-{thread_id}"
|
||||
try:
|
||||
search_count = 0
|
||||
while not stop_event.is_set():
|
||||
try:
|
||||
# Generate query vector
|
||||
query_vec = generate_random_vector(VECTOR_DIM)
|
||||
query_str = [str(x) for x in query_vec]
|
||||
|
||||
# Perform similarity search
|
||||
args = ['VSIM', self.test_key, 'VALUES', VECTOR_DIM]
|
||||
args.extend(query_str)
|
||||
args.extend(['COUNT', 10])
|
||||
self.redis.execute_command(*args)
|
||||
|
||||
search_count += 1
|
||||
|
||||
# Small sleep to reduce CPU pressure
|
||||
if search_count % 10 == 0:
|
||||
time.sleep(0.005)
|
||||
except Exception as e:
|
||||
# Don't log empty array errors, as they're expected when key doesn't exist
|
||||
if "empty array" not in str(e).lower():
|
||||
log_error(thread_name, f"Error: {str(e)}")
|
||||
time.sleep(0.1) # Slight backoff on error
|
||||
except Exception as e:
|
||||
log_error(thread_name, f"Thread error: {str(e)}")
|
||||
|
||||
# Start all threads
|
||||
threads = []
|
||||
|
||||
# VADD threads
|
||||
for i in range(NUM_VADD_THREADS):
|
||||
thread = threading.Thread(target=vadd_worker, args=(i,))
|
||||
thread.start()
|
||||
threads.append(thread)
|
||||
|
||||
# DEL threads
|
||||
for _ in range(NUM_DEL_THREADS):
|
||||
thread = threading.Thread(target=del_worker)
|
||||
thread.start()
|
||||
threads.append(thread)
|
||||
|
||||
# VSIM threads
|
||||
for i in range(NUM_VSIM_THREADS):
|
||||
thread = threading.Thread(target=vsim_worker, args=(i,))
|
||||
thread.start()
|
||||
threads.append(thread)
|
||||
|
||||
# Let the test run for the specified duration
|
||||
time.sleep(TEST_DURATION)
|
||||
|
||||
# Signal all threads to stop
|
||||
stop_event.set()
|
||||
|
||||
# Wait for threads to finish
|
||||
for thread in threads:
|
||||
thread.join(timeout=2.0)
|
||||
|
||||
# Check if Redis is still responsive
|
||||
try:
|
||||
ping_result = self.redis.ping()
|
||||
assert ping_result, "Redis did not respond to PING after stress test"
|
||||
except Exception as e:
|
||||
assert False, f"Redis connection failed after stress test: {str(e)}"
|
||||
|
||||
# Report any errors for diagnosis, but don't fail the test unless PING fails
|
||||
if error_list:
|
||||
error_count = len(error_list)
|
||||
print(f"\nEncountered {error_count} errors during stress test.")
|
||||
print("First 5 errors:")
|
||||
for error in error_list[:5]:
|
||||
print(f"- {error}")
|
||||
@@ -0,0 +1,48 @@
|
||||
from test import TestCase, fill_redis_with_vectors, generate_random_vector
|
||||
import threading, time
|
||||
|
||||
class ConcurrentVSIMAndDEL(TestCase):
|
||||
def getname(self):
|
||||
return "Concurrent VSIM and DEL operations"
|
||||
|
||||
def estimated_runtime(self):
|
||||
return 2
|
||||
|
||||
def test(self):
|
||||
# Fill the key with 5000 random vectors
|
||||
dim = 128
|
||||
count = 5000
|
||||
fill_redis_with_vectors(self.redis, self.test_key, count, dim)
|
||||
|
||||
# List to store results from threads
|
||||
thread_results = []
|
||||
|
||||
def vsim_thread():
|
||||
"""Thread function to perform VSIM operations until the key is deleted"""
|
||||
while True:
|
||||
query_vec = generate_random_vector(dim)
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', dim,
|
||||
*[str(x) for x in query_vec], 'COUNT', 10)
|
||||
if not result:
|
||||
# Empty array detected, key is deleted
|
||||
thread_results.append(True)
|
||||
break
|
||||
|
||||
# Start multiple threads to perform VSIM operations
|
||||
threads = []
|
||||
for _ in range(4): # Start 4 threads
|
||||
t = threading.Thread(target=vsim_thread)
|
||||
t.start()
|
||||
threads.append(t)
|
||||
|
||||
# Delete the key while threads are still running
|
||||
time.sleep(1)
|
||||
self.redis.delete(self.test_key)
|
||||
|
||||
# Wait for all threads to finish (they will exit once they detect the key is deleted)
|
||||
for t in threads:
|
||||
t.join()
|
||||
|
||||
# Verify that all threads detected an empty array or error
|
||||
assert len(thread_results) == len(threads), "Not all threads detected the key deletion"
|
||||
assert all(thread_results), "Some threads did not detect an empty array or error after DEL"
|
||||
@@ -0,0 +1,39 @@
|
||||
from test import TestCase, generate_random_vector
|
||||
import struct
|
||||
|
||||
class DebugDigestTest(TestCase):
|
||||
def getname(self):
|
||||
return "[regression] DEBUG DIGEST-VALUE with attributes"
|
||||
|
||||
def test(self):
|
||||
# Generate random vectors
|
||||
vec1 = generate_random_vector(4)
|
||||
vec2 = generate_random_vector(4)
|
||||
vec_bytes1 = struct.pack('4f', *vec1)
|
||||
vec_bytes2 = struct.pack('4f', *vec2)
|
||||
|
||||
# Add vectors to the key, one with attribute, one without
|
||||
self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes1, f'{self.test_key}:item:1')
|
||||
self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes2, f'{self.test_key}:item:2', 'SETATTR', '{"color":"red"}')
|
||||
|
||||
# Call DEBUG DIGEST-VALUE on the key
|
||||
try:
|
||||
digest1 = self.redis.execute_command('DEBUG', 'DIGEST-VALUE', self.test_key)
|
||||
assert digest1 is not None, "DEBUG DIGEST-VALUE should return a value"
|
||||
|
||||
# Change attribute and verify digest changes
|
||||
self.redis.execute_command('VSETATTR', self.test_key, f'{self.test_key}:item:2', '{"color":"blue"}')
|
||||
|
||||
digest2 = self.redis.execute_command('DEBUG', 'DIGEST-VALUE', self.test_key)
|
||||
assert digest2 is not None, "DEBUG DIGEST-VALUE should return a value after attribute change"
|
||||
assert digest1 != digest2, "Digest should change when an attribute is modified"
|
||||
|
||||
# Remove attribute and verify digest changes again
|
||||
self.redis.execute_command('VSETATTR', self.test_key, f'{self.test_key}:item:2', '')
|
||||
|
||||
digest3 = self.redis.execute_command('DEBUG', 'DIGEST-VALUE', self.test_key)
|
||||
assert digest3 is not None, "DEBUG DIGEST-VALUE should return a value after attribute removal"
|
||||
assert digest2 != digest3, "Digest should change when an attribute is removed"
|
||||
|
||||
except Exception as e:
|
||||
raise AssertionError(f"DEBUG DIGEST-VALUE command failed: {str(e)}")
|
||||
@@ -0,0 +1,173 @@
|
||||
from test import TestCase, fill_redis_with_vectors, generate_random_vector
|
||||
import random
|
||||
|
||||
"""
|
||||
A note about this test:
|
||||
It was experimentally tried to modify hnsw.c in order to
|
||||
avoid calling hnsw_reconnect_nodes(). In this case, the test
|
||||
fails very often with EF set to 250, while it hardly
|
||||
fails at all with the same parameters if hnsw_reconnect_nodes()
|
||||
is called.
|
||||
|
||||
Note that for the nature of the test (it is very strict) it can
|
||||
still fail from time to time, without this signaling any
|
||||
actual bug.
|
||||
"""
|
||||
|
||||
class VREM(TestCase):
|
||||
def getname(self):
|
||||
return "Deletion and graph state after deletion"
|
||||
|
||||
def estimated_runtime(self):
|
||||
return 2.0
|
||||
|
||||
def format_neighbors_with_scores(self, links_result, old_links=None, items_to_remove=None):
|
||||
"""Format neighbors with their similarity scores and status indicators"""
|
||||
if not links_result:
|
||||
return "No neighbors"
|
||||
|
||||
output = []
|
||||
for level, neighbors in enumerate(links_result):
|
||||
level_num = len(links_result) - level - 1
|
||||
output.append(f"Level {level_num}:")
|
||||
|
||||
# Get neighbors and scores
|
||||
neighbors_with_scores = []
|
||||
for i in range(0, len(neighbors), 2):
|
||||
neighbor = neighbors[i].decode() if isinstance(neighbors[i], bytes) else neighbors[i]
|
||||
score = float(neighbors[i+1]) if i+1 < len(neighbors) else None
|
||||
status = ""
|
||||
|
||||
# For old links, mark deleted ones
|
||||
if items_to_remove and neighbor in items_to_remove:
|
||||
status = " [lost]"
|
||||
# For new links, mark newly added ones
|
||||
elif old_links is not None:
|
||||
# Check if this neighbor was in the old links at this level
|
||||
was_present = False
|
||||
if old_links and level < len(old_links):
|
||||
old_neighbors = [n.decode() if isinstance(n, bytes) else n
|
||||
for n in old_links[level]]
|
||||
was_present = neighbor in old_neighbors
|
||||
if not was_present:
|
||||
status = " [gained]"
|
||||
|
||||
if score is not None:
|
||||
neighbors_with_scores.append(f"{len(neighbors_with_scores)+1}. {neighbor} ({score:.6f}){status}")
|
||||
else:
|
||||
neighbors_with_scores.append(f"{len(neighbors_with_scores)+1}. {neighbor}{status}")
|
||||
|
||||
output.extend([" " + n for n in neighbors_with_scores])
|
||||
return "\n".join(output)
|
||||
|
||||
def test(self):
|
||||
# 1. Fill server with random elements
|
||||
dim = 128
|
||||
count = 5000
|
||||
data = fill_redis_with_vectors(self.redis, self.test_key, count, dim)
|
||||
|
||||
# 2. Do VSIM to get 200 items
|
||||
query_vec = generate_random_vector(dim)
|
||||
results = self.redis.execute_command('VSIM', self.test_key, 'VALUES', dim,
|
||||
*[str(x) for x in query_vec],
|
||||
'COUNT', 200, 'WITHSCORES')
|
||||
|
||||
# Convert results to list of (item, score) pairs, sorted by score
|
||||
items = []
|
||||
for i in range(0, len(results), 2):
|
||||
item = results[i].decode()
|
||||
score = float(results[i+1])
|
||||
items.append((item, score))
|
||||
items.sort(key=lambda x: x[1], reverse=True) # Sort by similarity
|
||||
|
||||
# Store the graph structure for all items before deletion
|
||||
neighbors_before = {}
|
||||
for item, _ in items:
|
||||
links = self.redis.execute_command('VLINKS', self.test_key, item, 'WITHSCORES')
|
||||
if links: # Some items might not have links
|
||||
neighbors_before[item] = links
|
||||
|
||||
# 3. Remove 100 random items
|
||||
items_to_remove = set(item for item, _ in random.sample(items, 100))
|
||||
# Keep track of top 10 non-removed items
|
||||
top_remaining = []
|
||||
for item, score in items:
|
||||
if item not in items_to_remove:
|
||||
top_remaining.append((item, score))
|
||||
if len(top_remaining) == 10:
|
||||
break
|
||||
|
||||
# Remove the items
|
||||
for item in items_to_remove:
|
||||
result = self.redis.execute_command('VREM', self.test_key, item)
|
||||
assert result == 1, f"VREM failed to remove {item}"
|
||||
|
||||
# 4. Do VSIM again with same vector
|
||||
new_results = self.redis.execute_command('VSIM', self.test_key, 'VALUES', dim,
|
||||
*[str(x) for x in query_vec],
|
||||
'COUNT', 200, 'WITHSCORES',
|
||||
'EF', 500)
|
||||
|
||||
# Convert new results to dict of item -> score
|
||||
new_scores = {}
|
||||
for i in range(0, len(new_results), 2):
|
||||
item = new_results[i].decode()
|
||||
score = float(new_results[i+1])
|
||||
new_scores[item] = score
|
||||
|
||||
failure = False
|
||||
failed_item = None
|
||||
failed_reason = None
|
||||
# 5. Verify all top 10 non-removed items are still found with similar scores
|
||||
for item, old_score in top_remaining:
|
||||
if item not in new_scores:
|
||||
failure = True
|
||||
failed_item = item
|
||||
failed_reason = "missing"
|
||||
break
|
||||
new_score = new_scores[item]
|
||||
if abs(new_score - old_score) >= 0.01:
|
||||
failure = True
|
||||
failed_item = item
|
||||
failed_reason = f"score changed: {old_score:.6f} -> {new_score:.6f}"
|
||||
break
|
||||
|
||||
if failure:
|
||||
print("\nTest failed!")
|
||||
print(f"Problem with item: {failed_item} ({failed_reason})")
|
||||
|
||||
print("\nOriginal neighbors (with similarity scores):")
|
||||
if failed_item in neighbors_before:
|
||||
print(self.format_neighbors_with_scores(
|
||||
neighbors_before[failed_item],
|
||||
items_to_remove=items_to_remove))
|
||||
else:
|
||||
print("No neighbors found in original graph")
|
||||
|
||||
print("\nCurrent neighbors (with similarity scores):")
|
||||
current_links = self.redis.execute_command('VLINKS', self.test_key,
|
||||
failed_item, 'WITHSCORES')
|
||||
if current_links:
|
||||
print(self.format_neighbors_with_scores(
|
||||
current_links,
|
||||
old_links=neighbors_before.get(failed_item)))
|
||||
else:
|
||||
print("No neighbors in current graph")
|
||||
|
||||
print("\nOriginal results (top 20):")
|
||||
for item, score in items[:20]:
|
||||
deleted = "[deleted]" if item in items_to_remove else ""
|
||||
print(f"{item}: {score:.6f} {deleted}")
|
||||
|
||||
print("\nNew results after removal (top 20):")
|
||||
new_items = []
|
||||
for i in range(0, len(new_results), 2):
|
||||
item = new_results[i].decode()
|
||||
score = float(new_results[i+1])
|
||||
new_items.append((item, score))
|
||||
new_items.sort(key=lambda x: x[1], reverse=True)
|
||||
for item, score in new_items[:20]:
|
||||
print(f"{item}: {score:.6f}")
|
||||
|
||||
raise AssertionError(f"Test failed: Problem with item {failed_item} ({failed_reason}). *** IMPORTANT *** This test may fail from time to time without indicating that there is a bug. However normally it should pass. The fact is that it's a quite extreme test where we destroy 50% of nodes of top results and still expect perfect recall, with vectors that are very hostile because of the distribution used.")
|
||||
|
||||
@@ -0,0 +1,67 @@
|
||||
from test import TestCase, generate_random_vector
|
||||
import struct
|
||||
import redis.exceptions
|
||||
|
||||
class DimensionValidation(TestCase):
|
||||
def getname(self):
|
||||
return "[regression] Dimension Validation with Projection"
|
||||
|
||||
def estimated_runtime(self):
|
||||
return 0.5
|
||||
|
||||
def test(self):
|
||||
# Test scenario 1: Create a set with projection
|
||||
original_dim = 100
|
||||
reduced_dim = 50
|
||||
|
||||
# Create the initial vector and set with projection
|
||||
vec1 = generate_random_vector(original_dim)
|
||||
vec1_bytes = struct.pack(f'{original_dim}f', *vec1)
|
||||
|
||||
# Add first vector with projection
|
||||
result = self.redis.execute_command('VADD', self.test_key,
|
||||
'REDUCE', reduced_dim,
|
||||
'FP32', vec1_bytes, f'{self.test_key}:item:1')
|
||||
assert result == 1, "First VADD with REDUCE should return 1"
|
||||
|
||||
# Check VINFO returns the correct projection information
|
||||
info = self.redis.execute_command('VINFO', self.test_key)
|
||||
info_map = {k.decode('utf-8'): v for k, v in zip(info[::2], info[1::2])}
|
||||
assert 'vector-dim' in info_map, "VINFO should contain vector-dim"
|
||||
assert info_map['vector-dim'] == reduced_dim, f"Expected reduced dimension {reduced_dim}, got {info['vector-dim']}"
|
||||
assert 'projection-input-dim' in info_map, "VINFO should contain projection-input-dim"
|
||||
assert info_map['projection-input-dim'] == original_dim, f"Expected original dimension {original_dim}, got {info['projection-input-dim']}"
|
||||
|
||||
# Test scenario 2: Try adding a mismatched vector - should fail
|
||||
wrong_dim = 80
|
||||
wrong_vec = generate_random_vector(wrong_dim)
|
||||
wrong_vec_bytes = struct.pack(f'{wrong_dim}f', *wrong_vec)
|
||||
|
||||
# This should fail with dimension mismatch error
|
||||
try:
|
||||
self.redis.execute_command('VADD', self.test_key,
|
||||
'REDUCE', reduced_dim,
|
||||
'FP32', wrong_vec_bytes, f'{self.test_key}:item:2')
|
||||
assert False, "VADD with wrong dimension should fail"
|
||||
except redis.exceptions.ResponseError as e:
|
||||
assert "Input dimension mismatch for projection" in str(e), f"Expected dimension mismatch error, got: {e}"
|
||||
|
||||
# Test scenario 3: Add a correctly-sized vector
|
||||
vec2 = generate_random_vector(original_dim)
|
||||
vec2_bytes = struct.pack(f'{original_dim}f', *vec2)
|
||||
|
||||
# This should succeed
|
||||
result = self.redis.execute_command('VADD', self.test_key,
|
||||
'REDUCE', reduced_dim,
|
||||
'FP32', vec2_bytes, f'{self.test_key}:item:3')
|
||||
assert result == 1, "VADD with correct dimensions should succeed"
|
||||
|
||||
# Check VSIM also validates input dimensions
|
||||
wrong_query = generate_random_vector(wrong_dim)
|
||||
try:
|
||||
self.redis.execute_command('VSIM', self.test_key,
|
||||
'VALUES', wrong_dim, *[str(x) for x in wrong_query],
|
||||
'COUNT', 10)
|
||||
assert False, "VSIM with wrong dimension should fail"
|
||||
except redis.exceptions.ResponseError as e:
|
||||
assert "Input dimension mismatch for projection" in str(e), f"Expected dimension mismatch error in VSIM, got: {e}"
|
||||
@@ -0,0 +1,27 @@
|
||||
from test import TestCase, generate_random_vector
|
||||
import struct
|
||||
|
||||
class VREM_LastItemDeletesKey(TestCase):
|
||||
def getname(self):
|
||||
return "VREM last item deletes key"
|
||||
|
||||
def test(self):
|
||||
# Generate a random vector
|
||||
vec = generate_random_vector(4)
|
||||
vec_bytes = struct.pack('4f', *vec)
|
||||
|
||||
# Add the vector to the key
|
||||
result = self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes, f'{self.test_key}:item:1')
|
||||
assert result == 1, "VADD should return 1 for first item"
|
||||
|
||||
# Verify the key exists
|
||||
exists = self.redis.exists(self.test_key)
|
||||
assert exists == 1, "Key should exist after VADD"
|
||||
|
||||
# Remove the item
|
||||
result = self.redis.execute_command('VREM', self.test_key, f'{self.test_key}:item:1')
|
||||
assert result == 1, "VREM should return 1 for successful removal"
|
||||
|
||||
# Verify the key no longer exists
|
||||
exists = self.redis.exists(self.test_key)
|
||||
assert exists == 0, "Key should no longer exist after VREM of last item"
|
||||
@@ -0,0 +1,177 @@
|
||||
from test import TestCase
|
||||
|
||||
class VSIMFilterExpressions(TestCase):
|
||||
def getname(self):
|
||||
return "VSIM FILTER expressions basic functionality"
|
||||
|
||||
def test(self):
|
||||
# Create a small set of vectors with different attributes
|
||||
|
||||
# Basic vectors for testing - all orthogonal for clear results
|
||||
vec1 = [1, 0, 0, 0]
|
||||
vec2 = [0, 1, 0, 0]
|
||||
vec3 = [0, 0, 1, 0]
|
||||
vec4 = [0, 0, 0, 1]
|
||||
vec5 = [0.5, 0.5, 0, 0]
|
||||
|
||||
# Add vectors with various attributes
|
||||
self.redis.execute_command('VADD', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1], f'{self.test_key}:item:1')
|
||||
self.redis.execute_command('VSETATTR', self.test_key, f'{self.test_key}:item:1',
|
||||
'{"age": 25, "name": "Alice", "active": true, "scores": [85, 90, 95], "city": "New York"}')
|
||||
|
||||
self.redis.execute_command('VADD', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec2], f'{self.test_key}:item:2')
|
||||
self.redis.execute_command('VSETATTR', self.test_key, f'{self.test_key}:item:2',
|
||||
'{"age": 30, "name": "Bob", "active": false, "scores": [70, 75, 80], "city": "Boston"}')
|
||||
|
||||
self.redis.execute_command('VADD', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec3], f'{self.test_key}:item:3')
|
||||
self.redis.execute_command('VSETATTR', self.test_key, f'{self.test_key}:item:3',
|
||||
'{"age": 35, "name": "Charlie", "scores": [60, 65, 70], "city": "Seattle"}')
|
||||
|
||||
self.redis.execute_command('VADD', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec4], f'{self.test_key}:item:4')
|
||||
# Item 4 has no attribute at all
|
||||
|
||||
self.redis.execute_command('VADD', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec5], f'{self.test_key}:item:5')
|
||||
self.redis.execute_command('VSETATTR', self.test_key, f'{self.test_key}:item:5',
|
||||
'invalid json') # Intentionally malformed JSON
|
||||
|
||||
# Test 1: Basic equality with numbers
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.age == 25')
|
||||
assert len(result) == 1, "Expected 1 result for age == 25"
|
||||
assert result[0].decode() == f'{self.test_key}:item:1', "Expected item:1 for age == 25"
|
||||
|
||||
# Test 2: Greater than
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.age > 25')
|
||||
assert len(result) == 2, "Expected 2 results for age > 25"
|
||||
|
||||
# Test 3: Less than or equal
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.age <= 30')
|
||||
assert len(result) == 2, "Expected 2 results for age <= 30"
|
||||
|
||||
# Test 4: String equality
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.name == "Alice"')
|
||||
assert len(result) == 1, "Expected 1 result for name == Alice"
|
||||
assert result[0].decode() == f'{self.test_key}:item:1', "Expected item:1 for name == Alice"
|
||||
|
||||
# Test 5: String inequality
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.name != "Alice"')
|
||||
assert len(result) == 2, "Expected 2 results for name != Alice"
|
||||
|
||||
# Test 6: Boolean value
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.active')
|
||||
assert len(result) == 1, "Expected 1 result for .active being true"
|
||||
|
||||
# Test 7: Logical AND
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.age > 20 and .age < 30')
|
||||
assert len(result) == 1, "Expected 1 result for 20 < age < 30"
|
||||
assert result[0].decode() == f'{self.test_key}:item:1', "Expected item:1 for 20 < age < 30"
|
||||
|
||||
# Test 8: Logical OR
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.age < 30 or .age > 35')
|
||||
assert len(result) == 1, "Expected 1 result for age < 30 or age > 35"
|
||||
|
||||
# Test 9: Logical NOT
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '!(.age == 25)')
|
||||
assert len(result) == 2, "Expected 2 results for NOT(age == 25)"
|
||||
|
||||
# Test 10: The "in" operator with array
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.age in [25, 35]')
|
||||
assert len(result) == 2, "Expected 2 results for age in [25, 35]"
|
||||
|
||||
# Test 11: The "in" operator with strings in array
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.name in ["Alice", "David"]')
|
||||
assert len(result) == 1, "Expected 1 result for name in [Alice, David]"
|
||||
|
||||
# Test 12: Arithmetic operations - addition
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.age + 10 > 40')
|
||||
assert len(result) == 1, "Expected 1 result for age + 10 > 40"
|
||||
|
||||
# Test 13: Arithmetic operations - multiplication
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.age * 2 > 60')
|
||||
assert len(result) == 1, "Expected 1 result for age * 2 > 60"
|
||||
|
||||
# Test 14: Arithmetic operations - division
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.age / 5 == 5')
|
||||
assert len(result) == 1, "Expected 1 result for age / 5 == 5"
|
||||
|
||||
# Test 15: Arithmetic operations - modulo
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.age % 2 == 0')
|
||||
assert len(result) == 1, "Expected 1 result for age % 2 == 0"
|
||||
|
||||
# Test 16: Power operator
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.age ** 2 > 900')
|
||||
assert len(result) == 1, "Expected 1 result for age^2 > 900"
|
||||
|
||||
# Test 17: Missing attribute (should exclude items missing that attribute)
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.missing_field == "value"')
|
||||
assert len(result) == 0, "Expected 0 results for missing_field == value"
|
||||
|
||||
# Test 18: No attribute set at all
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.any_field')
|
||||
assert f'{self.test_key}:item:4' not in [item.decode() for item in result], "Item with no attribute should be excluded"
|
||||
|
||||
# Test 19: Malformed JSON
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.any_field')
|
||||
assert f'{self.test_key}:item:5' not in [item.decode() for item in result], "Item with malformed JSON should be excluded"
|
||||
|
||||
# Test 20: Complex expression combining multiple operators
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '(.age > 20 and .age < 40) and (.city == "Boston" or .city == "New York")')
|
||||
assert len(result) == 2, "Expected 2 results for the complex expression"
|
||||
expected_items = [f'{self.test_key}:item:1', f'{self.test_key}:item:2']
|
||||
assert set([item.decode() for item in result]) == set(expected_items), "Expected item:1 and item:2 for the complex expression"
|
||||
|
||||
# Test 21: Parentheses to control operator precedence
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.age > (20 + 10)')
|
||||
assert len(result) == 1, "Expected 1 result for age > (20 + 10)"
|
||||
|
||||
# Test 22: Array access (arrays evaluate to true)
|
||||
result = self.redis.execute_command('VSIM', self.test_key, 'VALUES', 4,
|
||||
*[str(x) for x in vec1],
|
||||
'FILTER', '.scores')
|
||||
assert len(result) == 3, "Expected 3 results for .scores (arrays evaluate to true)"
|
||||
@@ -0,0 +1,668 @@
|
||||
from test import TestCase, generate_random_vector
|
||||
import struct
|
||||
import random
|
||||
import math
|
||||
import json
|
||||
import time
|
||||
|
||||
class VSIMFilterAdvanced(TestCase):
|
||||
def getname(self):
|
||||
return "VSIM FILTER comprehensive functionality testing"
|
||||
|
||||
def estimated_runtime(self):
|
||||
return 15 # This test might take up to 15 seconds for the large dataset
|
||||
|
||||
def setup(self):
|
||||
super().setup()
|
||||
self.dim = 32 # Vector dimension
|
||||
self.count = 5000 # Number of vectors for large tests
|
||||
self.small_count = 50 # Number of vectors for small/quick tests
|
||||
|
||||
# Categories for attributes
|
||||
self.categories = ["electronics", "furniture", "clothing", "books", "food"]
|
||||
self.cities = ["New York", "London", "Tokyo", "Paris", "Berlin", "Sydney", "Toronto", "Singapore"]
|
||||
self.price_ranges = [(10, 50), (50, 200), (200, 1000), (1000, 5000)]
|
||||
self.years = list(range(2000, 2025))
|
||||
|
||||
def create_attributes(self, index):
|
||||
"""Create realistic attributes for a vector"""
|
||||
category = random.choice(self.categories)
|
||||
city = random.choice(self.cities)
|
||||
min_price, max_price = random.choice(self.price_ranges)
|
||||
price = round(random.uniform(min_price, max_price), 2)
|
||||
year = random.choice(self.years)
|
||||
in_stock = random.random() > 0.3 # 70% chance of being in stock
|
||||
rating = round(random.uniform(1, 5), 1)
|
||||
views = int(random.expovariate(1/1000)) # Exponential distribution for page views
|
||||
tags = random.sample(["popular", "sale", "new", "limited", "exclusive", "clearance"],
|
||||
k=random.randint(0, 3))
|
||||
|
||||
# Add some specific patterns for testing
|
||||
# Every 10th item has a specific property combination for testing
|
||||
is_premium = (index % 10 == 0)
|
||||
|
||||
# Create attributes dictionary
|
||||
attrs = {
|
||||
"id": index,
|
||||
"category": category,
|
||||
"location": city,
|
||||
"price": price,
|
||||
"year": year,
|
||||
"in_stock": in_stock,
|
||||
"rating": rating,
|
||||
"views": views,
|
||||
"tags": tags
|
||||
}
|
||||
|
||||
if is_premium:
|
||||
attrs["is_premium"] = True
|
||||
attrs["special_features"] = ["premium", "warranty", "support"]
|
||||
|
||||
# Add sub-categories for more complex filters
|
||||
if category == "electronics":
|
||||
attrs["subcategory"] = random.choice(["phones", "computers", "cameras", "audio"])
|
||||
elif category == "furniture":
|
||||
attrs["subcategory"] = random.choice(["chairs", "tables", "sofas", "beds"])
|
||||
elif category == "clothing":
|
||||
attrs["subcategory"] = random.choice(["shirts", "pants", "dresses", "shoes"])
|
||||
|
||||
# Add some intentionally missing fields for testing
|
||||
if random.random() > 0.9: # 10% chance of missing price
|
||||
del attrs["price"]
|
||||
|
||||
# Some items have promotion field
|
||||
if random.random() > 0.7: # 30% chance of having a promotion
|
||||
attrs["promotion"] = random.choice(["discount", "bundle", "gift"])
|
||||
|
||||
# Create invalid JSON for a small percentage of vectors
|
||||
if random.random() > 0.98: # 2% chance of having invalid JSON
|
||||
return "{{invalid json}}"
|
||||
|
||||
return json.dumps(attrs)
|
||||
|
||||
def create_vectors_with_attributes(self, key, count):
|
||||
"""Create vectors and add attributes to them"""
|
||||
vectors = []
|
||||
names = []
|
||||
attribute_map = {} # To store attributes for verification
|
||||
|
||||
# Create vectors
|
||||
for i in range(count):
|
||||
vec = generate_random_vector(self.dim)
|
||||
vectors.append(vec)
|
||||
name = f"{key}:item:{i}"
|
||||
names.append(name)
|
||||
|
||||
# Add to Redis
|
||||
vec_bytes = struct.pack(f'{self.dim}f', *vec)
|
||||
self.redis.execute_command('VADD', key, 'FP32', vec_bytes, name)
|
||||
|
||||
# Create and add attributes
|
||||
attrs = self.create_attributes(i)
|
||||
self.redis.execute_command('VSETATTR', key, name, attrs)
|
||||
|
||||
# Store attributes for later verification
|
||||
try:
|
||||
attribute_map[name] = json.loads(attrs) if '{' in attrs else None
|
||||
except json.JSONDecodeError:
|
||||
attribute_map[name] = None
|
||||
|
||||
return vectors, names, attribute_map
|
||||
|
||||
def filter_linear_search(self, vectors, names, query_vector, filter_expr, attribute_map, k=10):
|
||||
"""Perform a linear search with filtering for verification"""
|
||||
similarities = []
|
||||
query_norm = math.sqrt(sum(x*x for x in query_vector))
|
||||
|
||||
if query_norm == 0:
|
||||
return []
|
||||
|
||||
for i, vec in enumerate(vectors):
|
||||
name = names[i]
|
||||
attributes = attribute_map.get(name)
|
||||
|
||||
# Skip if doesn't match filter
|
||||
if not self.matches_filter(attributes, filter_expr):
|
||||
continue
|
||||
|
||||
vec_norm = math.sqrt(sum(x*x for x in vec))
|
||||
if vec_norm == 0:
|
||||
continue
|
||||
|
||||
dot_product = sum(a*b for a,b in zip(query_vector, vec))
|
||||
cosine_sim = dot_product / (query_norm * vec_norm)
|
||||
distance = 1.0 - cosine_sim
|
||||
redis_similarity = 1.0 - (distance/2.0)
|
||||
similarities.append((name, redis_similarity))
|
||||
|
||||
similarities.sort(key=lambda x: x[1], reverse=True)
|
||||
return similarities[:k]
|
||||
|
||||
def matches_filter(self, attributes, filter_expr):
|
||||
"""Filter matching for verification - uses Python eval to handle complex expressions"""
|
||||
if attributes is None:
|
||||
return False # No attributes or invalid JSON
|
||||
|
||||
# Replace JSON path selectors with Python dictionary access
|
||||
py_expr = filter_expr
|
||||
|
||||
# Handle `.field` notation (replace with attributes['field'])
|
||||
i = 0
|
||||
while i < len(py_expr):
|
||||
if py_expr[i] == '.' and (i == 0 or not py_expr[i-1].isalnum()):
|
||||
# Find the end of the selector (stops at operators or whitespace)
|
||||
j = i + 1
|
||||
while j < len(py_expr) and (py_expr[j].isalnum() or py_expr[j] == '_'):
|
||||
j += 1
|
||||
|
||||
if j > i + 1: # Found a valid selector
|
||||
field = py_expr[i+1:j]
|
||||
# Use a safe access pattern that returns a default value based on context
|
||||
py_expr = py_expr[:i] + f"attributes.get('{field}')" + py_expr[j:]
|
||||
i = i + len(f"attributes.get('{field}')")
|
||||
else:
|
||||
i += 1
|
||||
else:
|
||||
i += 1
|
||||
|
||||
# Convert not operator if needed
|
||||
py_expr = py_expr.replace('!', ' not ')
|
||||
|
||||
try:
|
||||
# Custom evaluation that handles exceptions for missing fields
|
||||
# by returning False for the entire expression
|
||||
|
||||
# Split the expression on logical operators
|
||||
parts = []
|
||||
for op in [' and ', ' or ']:
|
||||
if op in py_expr:
|
||||
parts = py_expr.split(op)
|
||||
break
|
||||
|
||||
if not parts: # No logical operators found
|
||||
parts = [py_expr]
|
||||
|
||||
# Try to evaluate each part - if any part fails,
|
||||
# the whole expression should fail
|
||||
try:
|
||||
result = eval(py_expr, {"attributes": attributes})
|
||||
return bool(result)
|
||||
except (TypeError, AttributeError):
|
||||
# This typically happens when trying to compare None with
|
||||
# numbers or other types, or when an attribute doesn't exist
|
||||
return False
|
||||
except Exception as e:
|
||||
print(f"Error evaluating filter expression '{filter_expr}' as '{py_expr}': {e}")
|
||||
return False
|
||||
|
||||
except Exception as e:
|
||||
print(f"Error evaluating filter expression '{filter_expr}' as '{py_expr}': {e}")
|
||||
return False
|
||||
|
||||
def safe_decode(self,item):
|
||||
return item.decode() if isinstance(item, bytes) else item
|
||||
|
||||
def calculate_recall(self, redis_results, linear_results, k=10):
|
||||
"""Calculate recall (percentage of correct results retrieved)"""
|
||||
redis_set = set(self.safe_decode(item) for item in redis_results)
|
||||
linear_set = set(item[0] for item in linear_results[:k])
|
||||
|
||||
if not linear_set:
|
||||
return 1.0 # If no linear results, consider it perfect recall
|
||||
|
||||
intersection = redis_set.intersection(linear_set)
|
||||
return len(intersection) / len(linear_set)
|
||||
|
||||
def test_recall_with_filter(self, filter_expr, ef=500, filter_ef=None):
|
||||
"""Test recall for a given filter expression"""
|
||||
# Create query vector
|
||||
query_vec = generate_random_vector(self.dim)
|
||||
|
||||
# First, get ground truth using linear scan
|
||||
linear_results = self.filter_linear_search(
|
||||
self.vectors, self.names, query_vec, filter_expr, self.attribute_map, k=50)
|
||||
|
||||
# Calculate true selectivity from ground truth
|
||||
true_selectivity = len(linear_results) / len(self.names) if self.names else 0
|
||||
|
||||
# Perform Redis search with filter
|
||||
cmd_args = ['VSIM', self.test_key, 'VALUES', self.dim]
|
||||
cmd_args.extend([str(x) for x in query_vec])
|
||||
cmd_args.extend(['COUNT', 50, 'WITHSCORES', 'EF', ef, 'FILTER', filter_expr])
|
||||
if filter_ef:
|
||||
cmd_args.extend(['FILTER-EF', filter_ef])
|
||||
|
||||
start_time = time.time()
|
||||
redis_results = self.redis.execute_command(*cmd_args)
|
||||
query_time = time.time() - start_time
|
||||
|
||||
# Convert Redis results to dict
|
||||
redis_items = {}
|
||||
for i in range(0, len(redis_results), 2):
|
||||
key = redis_results[i].decode() if isinstance(redis_results[i], bytes) else redis_results[i]
|
||||
score = float(redis_results[i+1])
|
||||
redis_items[key] = score
|
||||
|
||||
# Calculate metrics
|
||||
recall = self.calculate_recall(redis_items.keys(), linear_results)
|
||||
selectivity = len(redis_items) / len(self.names) if redis_items else 0
|
||||
|
||||
# Compare against the true selectivity from linear scan
|
||||
assert abs(selectivity - true_selectivity) < 0.1, \
|
||||
f"Redis selectivity {selectivity:.3f} differs significantly from ground truth {true_selectivity:.3f}"
|
||||
|
||||
# We expect high recall for standard parameters
|
||||
if ef >= 500 and (filter_ef is None or filter_ef >= 1000):
|
||||
try:
|
||||
assert recall >= 0.7, \
|
||||
f"Low recall {recall:.2f} for filter '{filter_expr}'"
|
||||
except AssertionError as e:
|
||||
# Get items found in each set
|
||||
redis_items_set = set(redis_items.keys())
|
||||
linear_items_set = set(item[0] for item in linear_results)
|
||||
|
||||
# Find items in each set
|
||||
only_in_redis = redis_items_set - linear_items_set
|
||||
only_in_linear = linear_items_set - redis_items_set
|
||||
in_both = redis_items_set & linear_items_set
|
||||
|
||||
# Build comprehensive debug message
|
||||
debug = f"\nGround Truth: {len(linear_results)} matching items (total vectors: {len(self.vectors)})"
|
||||
debug += f"\nRedis Found: {len(redis_items)} items with FILTER-EF: {filter_ef or 'default'}"
|
||||
debug += f"\nItems in both sets: {len(in_both)} (recall: {recall:.4f})"
|
||||
debug += f"\nItems only in Redis: {len(only_in_redis)}"
|
||||
debug += f"\nItems only in Ground Truth: {len(only_in_linear)}"
|
||||
|
||||
# Show some example items from each set with their scores
|
||||
if only_in_redis:
|
||||
debug += "\n\nTOP 5 ITEMS ONLY IN REDIS:"
|
||||
sorted_redis = sorted([(k, v) for k, v in redis_items.items()], key=lambda x: x[1], reverse=True)
|
||||
for i, (item, score) in enumerate(sorted_redis[:5]):
|
||||
if item in only_in_redis:
|
||||
debug += f"\n {i+1}. {item} (Score: {score:.4f})"
|
||||
|
||||
# Show attribute that should match filter
|
||||
attr = self.attribute_map.get(item)
|
||||
if attr:
|
||||
debug += f" - Attrs: {attr.get('category', 'N/A')}, Price: {attr.get('price', 'N/A')}"
|
||||
|
||||
if only_in_linear:
|
||||
debug += "\n\nTOP 5 ITEMS ONLY IN GROUND TRUTH:"
|
||||
for i, (item, score) in enumerate(linear_results[:5]):
|
||||
if item in only_in_linear:
|
||||
debug += f"\n {i+1}. {item} (Score: {score:.4f})"
|
||||
|
||||
# Show attribute that should match filter
|
||||
attr = self.attribute_map.get(item)
|
||||
if attr:
|
||||
debug += f" - Attrs: {attr.get('category', 'N/A')}, Price: {attr.get('price', 'N/A')}"
|
||||
|
||||
# Help identify parsing issues
|
||||
debug += "\n\nPARSING CHECK:"
|
||||
debug += f"\nRedis command: VSIM {self.test_key} VALUES {self.dim} [...] FILTER '{filter_expr}'"
|
||||
|
||||
# Check for WITHSCORES handling issues
|
||||
if len(redis_results) > 0 and len(redis_results) % 2 == 0:
|
||||
debug += f"\nRedis returned {len(redis_results)} items (looks like item,score pairs)"
|
||||
debug += f"\nFirst few results: {redis_results[:4]}"
|
||||
|
||||
# Check the filter implementation
|
||||
debug += "\n\nFILTER IMPLEMENTATION CHECK:"
|
||||
debug += f"\nFilter expression: '{filter_expr}'"
|
||||
debug += "\nSample attribute matches from attribute_map:"
|
||||
count_matching = 0
|
||||
for i, (name, attrs) in enumerate(self.attribute_map.items()):
|
||||
if attrs and self.matches_filter(attrs, filter_expr):
|
||||
count_matching += 1
|
||||
if i < 3: # Show first 3 matches
|
||||
debug += f"\n - {name}: {attrs}"
|
||||
debug += f"\nTotal items matching filter in attribute_map: {count_matching}"
|
||||
|
||||
# Check if results array handling could be wrong
|
||||
debug += "\n\nRESULT ARRAYS CHECK:"
|
||||
if len(linear_results) >= 1:
|
||||
debug += f"\nlinear_results[0]: {linear_results[0]}"
|
||||
if isinstance(linear_results[0], tuple) and len(linear_results[0]) == 2:
|
||||
debug += " (correct tuple format: (name, score))"
|
||||
else:
|
||||
debug += " (UNEXPECTED FORMAT!)"
|
||||
|
||||
# Debug sort order
|
||||
debug += "\n\nSORTING CHECK:"
|
||||
if len(linear_results) >= 2:
|
||||
debug += f"\nGround truth first item score: {linear_results[0][1]}"
|
||||
debug += f"\nGround truth second item score: {linear_results[1][1]}"
|
||||
debug += f"\nCorrectly sorted by similarity? {linear_results[0][1] >= linear_results[1][1]}"
|
||||
|
||||
# Re-raise with detailed information
|
||||
raise AssertionError(str(e) + debug)
|
||||
|
||||
return recall, selectivity, query_time, len(redis_items)
|
||||
|
||||
def test(self):
|
||||
print(f"\nRunning comprehensive VSIM FILTER tests...")
|
||||
|
||||
# Create a larger dataset for testing
|
||||
print(f"Creating dataset with {self.count} vectors and attributes...")
|
||||
self.vectors, self.names, self.attribute_map = self.create_vectors_with_attributes(
|
||||
self.test_key, self.count)
|
||||
|
||||
# ==== 1. Recall and Precision Testing ====
|
||||
print("Testing recall for various filters...")
|
||||
|
||||
# Test basic filters with different selectivity
|
||||
results = {}
|
||||
results["category"] = self.test_recall_with_filter('.category == "electronics"')
|
||||
results["price_high"] = self.test_recall_with_filter('.price > 1000')
|
||||
results["in_stock"] = self.test_recall_with_filter('.in_stock')
|
||||
results["rating"] = self.test_recall_with_filter('.rating >= 4')
|
||||
results["complex1"] = self.test_recall_with_filter('.category == "electronics" and .price < 500')
|
||||
|
||||
print("Filter | Recall | Selectivity | Time (ms) | Results")
|
||||
print("----------------------------------------------------")
|
||||
for name, (recall, selectivity, time_ms, count) in results.items():
|
||||
print(f"{name:7} | {recall:.3f} | {selectivity:.3f} | {time_ms*1000:.1f} | {count}")
|
||||
|
||||
# ==== 2. Filter Selectivity Performance ====
|
||||
print("\nTesting filter selectivity performance...")
|
||||
|
||||
# High selectivity (very few matches)
|
||||
high_sel_recall, _, high_sel_time, _ = self.test_recall_with_filter('.is_premium')
|
||||
|
||||
# Medium selectivity
|
||||
med_sel_recall, _, med_sel_time, _ = self.test_recall_with_filter('.price > 100 and .price < 1000')
|
||||
|
||||
# Low selectivity (many matches)
|
||||
low_sel_recall, _, low_sel_time, _ = self.test_recall_with_filter('.year > 2000')
|
||||
|
||||
print(f"High selectivity recall: {high_sel_recall:.3f}, time: {high_sel_time*1000:.1f}ms")
|
||||
print(f"Med selectivity recall: {med_sel_recall:.3f}, time: {med_sel_time*1000:.1f}ms")
|
||||
print(f"Low selectivity recall: {low_sel_recall:.3f}, time: {low_sel_time*1000:.1f}ms")
|
||||
|
||||
# ==== 3. FILTER-EF Parameter Testing ====
|
||||
print("\nTesting FILTER-EF parameter...")
|
||||
|
||||
# Test with different FILTER-EF values
|
||||
filter_expr = '.category == "electronics" and .price > 200'
|
||||
ef_values = [100, 500, 2000, 5000]
|
||||
|
||||
print("FILTER-EF | Recall | Time (ms)")
|
||||
print("-----------------------------")
|
||||
for filter_ef in ef_values:
|
||||
recall, _, query_time, _ = self.test_recall_with_filter(
|
||||
filter_expr, ef=500, filter_ef=filter_ef)
|
||||
print(f"{filter_ef:9} | {recall:.3f} | {query_time*1000:.1f}")
|
||||
|
||||
# Assert that higher FILTER-EF generally gives better recall
|
||||
low_ef_recall, _, _, _ = self.test_recall_with_filter(filter_expr, filter_ef=100)
|
||||
high_ef_recall, _, _, _ = self.test_recall_with_filter(filter_expr, filter_ef=5000)
|
||||
|
||||
# This might not always be true due to randomness, but generally holds
|
||||
# We use a softer assertion to avoid flaky tests
|
||||
assert high_ef_recall >= low_ef_recall * 0.8, \
|
||||
f"Higher FILTER-EF should generally give better recall: {high_ef_recall:.3f} vs {low_ef_recall:.3f}"
|
||||
|
||||
# ==== 4. Complex Filter Expressions ====
|
||||
print("\nTesting complex filter expressions...")
|
||||
|
||||
# Test a variety of complex expressions
|
||||
complex_filters = [
|
||||
'.price > 100 and (.category == "electronics" or .category == "furniture")',
|
||||
'(.rating > 4 and .in_stock) or (.price < 50 and .views > 1000)',
|
||||
'.category in ["electronics", "clothing"] and .price > 200 and .rating >= 3',
|
||||
'(.category == "electronics" and .subcategory == "phones") or (.category == "furniture" and .price > 1000)',
|
||||
'.year > 2010 and !(.price < 100) and .in_stock'
|
||||
]
|
||||
|
||||
print("Expression | Results | Time (ms)")
|
||||
print("-----------------------------")
|
||||
for i, expr in enumerate(complex_filters):
|
||||
try:
|
||||
_, _, query_time, result_count = self.test_recall_with_filter(expr)
|
||||
print(f"Complex {i+1} | {result_count:7} | {query_time*1000:.1f}")
|
||||
except Exception as e:
|
||||
print(f"Complex {i+1} | Error: {str(e)}")
|
||||
|
||||
# ==== 5. Attribute Type Testing ====
|
||||
print("\nTesting different attribute types...")
|
||||
|
||||
type_filters = [
|
||||
('.price > 500', "Numeric"),
|
||||
('.category == "books"', "String equality"),
|
||||
('.in_stock', "Boolean"),
|
||||
('.tags in ["sale", "new"]', "Array membership"),
|
||||
('.rating * 2 > 8', "Arithmetic")
|
||||
]
|
||||
|
||||
for expr, type_name in type_filters:
|
||||
try:
|
||||
_, _, query_time, result_count = self.test_recall_with_filter(expr)
|
||||
print(f"{type_name:16} | {expr:30} | {result_count:5} results | {query_time*1000:.1f}ms")
|
||||
except Exception as e:
|
||||
print(f"{type_name:16} | {expr:30} | Error: {str(e)}")
|
||||
|
||||
# ==== 6. Filter + Count Interaction ====
|
||||
print("\nTesting COUNT parameter with filters...")
|
||||
|
||||
filter_expr = '.category == "electronics"'
|
||||
counts = [5, 20, 100]
|
||||
|
||||
for count in counts:
|
||||
query_vec = generate_random_vector(self.dim)
|
||||
cmd_args = ['VSIM', self.test_key, 'VALUES', self.dim]
|
||||
cmd_args.extend([str(x) for x in query_vec])
|
||||
cmd_args.extend(['COUNT', count, 'WITHSCORES', 'FILTER', filter_expr])
|
||||
|
||||
results = self.redis.execute_command(*cmd_args)
|
||||
result_count = len(results) // 2 # Divide by 2 because WITHSCORES returns pairs
|
||||
|
||||
# We expect result count to be at most the requested count
|
||||
assert result_count <= count, f"Got {result_count} results with COUNT {count}"
|
||||
print(f"COUNT {count:3} | Got {result_count:3} results")
|
||||
|
||||
# ==== 7. Edge Cases ====
|
||||
print("\nTesting edge cases...")
|
||||
|
||||
# Test with no matching items
|
||||
no_match_expr = '.category == "nonexistent_category"'
|
||||
results = self.redis.execute_command('VSIM', self.test_key, 'VALUES', self.dim,
|
||||
*[str(x) for x in generate_random_vector(self.dim)],
|
||||
'FILTER', no_match_expr)
|
||||
assert len(results) == 0, f"Expected 0 results for non-matching filter, got {len(results)}"
|
||||
print(f"No matching items: {len(results)} results (expected 0)")
|
||||
|
||||
# Test with invalid filter syntax
|
||||
try:
|
||||
self.redis.execute_command('VSIM', self.test_key, 'VALUES', self.dim,
|
||||
*[str(x) for x in generate_random_vector(self.dim)],
|
||||
'FILTER', '.category === "books"') # Triple equals is invalid
|
||||
assert False, "Expected error for invalid filter syntax"
|
||||
except:
|
||||
print("Invalid filter syntax correctly raised an error")
|
||||
|
||||
# Test with extremely long complex expression
|
||||
long_expr = ' and '.join([f'.rating > {i/10}' for i in range(10)])
|
||||
try:
|
||||
results = self.redis.execute_command('VSIM', self.test_key, 'VALUES', self.dim,
|
||||
*[str(x) for x in generate_random_vector(self.dim)],
|
||||
'FILTER', long_expr)
|
||||
print(f"Long expression: {len(results)} results")
|
||||
except Exception as e:
|
||||
print(f"Long expression error: {str(e)}")
|
||||
|
||||
print("\nComprehensive VSIM FILTER tests completed successfully")
|
||||
|
||||
|
||||
class VSIMFilterSelectivityTest(TestCase):
|
||||
def getname(self):
|
||||
return "VSIM FILTER selectivity performance benchmark"
|
||||
|
||||
def estimated_runtime(self):
|
||||
return 8 # This test might take up to 8 seconds
|
||||
|
||||
def setup(self):
|
||||
super().setup()
|
||||
self.dim = 32
|
||||
self.count = 10000
|
||||
self.test_key = f"{self.test_key}:selectivity" # Use a different key
|
||||
|
||||
def create_vector_with_age_attribute(self, name, age):
|
||||
"""Create a vector with a specific age attribute"""
|
||||
vec = generate_random_vector(self.dim)
|
||||
vec_bytes = struct.pack(f'{self.dim}f', *vec)
|
||||
self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes, name)
|
||||
self.redis.execute_command('VSETATTR', self.test_key, name, json.dumps({"age": age}))
|
||||
|
||||
def test(self):
|
||||
print("\nRunning VSIM FILTER selectivity benchmark...")
|
||||
|
||||
# Create a dataset where we control the exact selectivity
|
||||
print(f"Creating controlled dataset with {self.count} vectors...")
|
||||
|
||||
# Create vectors with age attributes from 1 to 100
|
||||
for i in range(self.count):
|
||||
age = (i % 100) + 1 # Ages from 1 to 100
|
||||
name = f"{self.test_key}:item:{i}"
|
||||
self.create_vector_with_age_attribute(name, age)
|
||||
|
||||
# Create a query vector
|
||||
query_vec = generate_random_vector(self.dim)
|
||||
|
||||
# Test filters with different selectivities
|
||||
selectivities = [0.01, 0.05, 0.10, 0.25, 0.50, 0.75, 0.99]
|
||||
results = []
|
||||
|
||||
print("\nSelectivity | Filter | Results | Time (ms)")
|
||||
print("--------------------------------------------------")
|
||||
|
||||
for target_selectivity in selectivities:
|
||||
# Calculate age threshold for desired selectivity
|
||||
# For example, age <= 10 gives 10% selectivity
|
||||
age_threshold = int(target_selectivity * 100)
|
||||
filter_expr = f'.age <= {age_threshold}'
|
||||
|
||||
# Run query and measure time
|
||||
start_time = time.time()
|
||||
cmd_args = ['VSIM', self.test_key, 'VALUES', self.dim]
|
||||
cmd_args.extend([str(x) for x in query_vec])
|
||||
cmd_args.extend(['COUNT', 100, 'FILTER', filter_expr])
|
||||
|
||||
results = self.redis.execute_command(*cmd_args)
|
||||
query_time = time.time() - start_time
|
||||
|
||||
actual_selectivity = len(results) / min(100, int(target_selectivity * self.count))
|
||||
print(f"{target_selectivity:.2f} | {filter_expr:15} | {len(results):7} | {query_time*1000:.1f}")
|
||||
|
||||
# Add assertion to ensure reasonable performance for different selectivities
|
||||
# For very selective queries (1%), we might need more exploration
|
||||
if target_selectivity <= 0.05:
|
||||
# For very selective queries, ensure we can find some results
|
||||
assert len(results) > 0, f"No results found for {filter_expr}"
|
||||
else:
|
||||
# For less selective queries, performance should be reasonable
|
||||
assert query_time < 1.0, f"Query too slow: {query_time:.3f}s for {filter_expr}"
|
||||
|
||||
print("\nSelectivity benchmark completed successfully")
|
||||
|
||||
|
||||
class VSIMFilterComparisonTest(TestCase):
|
||||
def getname(self):
|
||||
return "VSIM FILTER EF parameter comparison"
|
||||
|
||||
def estimated_runtime(self):
|
||||
return 8 # This test might take up to 8 seconds
|
||||
|
||||
def setup(self):
|
||||
super().setup()
|
||||
self.dim = 32
|
||||
self.count = 5000
|
||||
self.test_key = f"{self.test_key}:efparams" # Use a different key
|
||||
|
||||
def create_dataset(self):
|
||||
"""Create a dataset with specific attribute patterns for testing FILTER-EF"""
|
||||
vectors = []
|
||||
names = []
|
||||
|
||||
# Create vectors with category and quality score attributes
|
||||
for i in range(self.count):
|
||||
vec = generate_random_vector(self.dim)
|
||||
name = f"{self.test_key}:item:{i}"
|
||||
|
||||
# Add vector to Redis
|
||||
vec_bytes = struct.pack(f'{self.dim}f', *vec)
|
||||
self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes, name)
|
||||
|
||||
# Create attributes - we want a very selective filter
|
||||
# Only 2% of items have category=premium AND quality>90
|
||||
category = "premium" if random.random() < 0.1 else random.choice(["standard", "economy", "basic"])
|
||||
quality = random.randint(1, 100)
|
||||
|
||||
attrs = {
|
||||
"id": i,
|
||||
"category": category,
|
||||
"quality": quality
|
||||
}
|
||||
|
||||
self.redis.execute_command('VSETATTR', self.test_key, name, json.dumps(attrs))
|
||||
vectors.append(vec)
|
||||
names.append(name)
|
||||
|
||||
return vectors, names
|
||||
|
||||
def test(self):
|
||||
print("\nRunning VSIM FILTER-EF parameter comparison...")
|
||||
|
||||
# Create dataset
|
||||
vectors, names = self.create_dataset()
|
||||
|
||||
# Create a selective filter that matches ~2% of items
|
||||
filter_expr = '.category == "premium" and .quality > 90'
|
||||
|
||||
# Create query vector
|
||||
query_vec = generate_random_vector(self.dim)
|
||||
|
||||
# Test different FILTER-EF values
|
||||
ef_values = [50, 100, 500, 1000, 5000]
|
||||
results = []
|
||||
|
||||
print("\nFILTER-EF | Results | Time (ms) | Notes")
|
||||
print("---------------------------------------")
|
||||
|
||||
baseline_count = None
|
||||
|
||||
for ef in ef_values:
|
||||
# Run query and measure time
|
||||
start_time = time.time()
|
||||
cmd_args = ['VSIM', self.test_key, 'VALUES', self.dim]
|
||||
cmd_args.extend([str(x) for x in query_vec])
|
||||
cmd_args.extend(['COUNT', 100, 'FILTER', filter_expr, 'FILTER-EF', ef])
|
||||
|
||||
query_results = self.redis.execute_command(*cmd_args)
|
||||
query_time = time.time() - start_time
|
||||
|
||||
# Set baseline for comparison
|
||||
if baseline_count is None:
|
||||
baseline_count = len(query_results)
|
||||
|
||||
recall_rate = len(query_results) / max(1, baseline_count) if baseline_count > 0 else 1.0
|
||||
|
||||
notes = ""
|
||||
if ef == 5000:
|
||||
notes = "Baseline"
|
||||
elif recall_rate < 0.5:
|
||||
notes = "Low recall!"
|
||||
|
||||
print(f"{ef:9} | {len(query_results):7} | {query_time*1000:.1f} | {notes}")
|
||||
results.append((ef, len(query_results), query_time))
|
||||
|
||||
# If we have enough results at highest EF, check that recall improves with higher EF
|
||||
if results[-1][1] >= 5: # At least 5 results for highest EF
|
||||
# Extract result counts
|
||||
result_counts = [r[1] for r in results]
|
||||
|
||||
# The last result (highest EF) should typically find more results than the first (lowest EF)
|
||||
# but we use a soft assertion to avoid flaky tests
|
||||
assert result_counts[-1] >= result_counts[0], \
|
||||
f"Higher FILTER-EF should find at least as many results: {result_counts[-1]} vs {result_counts[0]}"
|
||||
|
||||
print("\nFILTER-EF parameter comparison completed successfully")
|
||||
@@ -0,0 +1,56 @@
|
||||
from test import TestCase, fill_redis_with_vectors, generate_random_vector
|
||||
import random
|
||||
|
||||
class LargeScale(TestCase):
|
||||
def getname(self):
|
||||
return "Large Scale Comparison"
|
||||
|
||||
def estimated_runtime(self):
|
||||
return 10
|
||||
|
||||
def test(self):
|
||||
dim = 300
|
||||
count = 20000
|
||||
k = 50
|
||||
|
||||
# Fill Redis and get reference data for comparison
|
||||
random.seed(42) # Make test deterministic
|
||||
data = fill_redis_with_vectors(self.redis, self.test_key, count, dim)
|
||||
|
||||
# Generate query vector
|
||||
query_vec = generate_random_vector(dim)
|
||||
|
||||
# Get results from Redis with good exploration factor
|
||||
redis_raw = self.redis.execute_command('VSIM', self.test_key, 'VALUES', dim,
|
||||
*[str(x) for x in query_vec],
|
||||
'COUNT', k, 'WITHSCORES', 'EF', 500)
|
||||
|
||||
# Convert Redis results to dict
|
||||
redis_results = {}
|
||||
for i in range(0, len(redis_raw), 2):
|
||||
key = redis_raw[i].decode()
|
||||
score = float(redis_raw[i+1])
|
||||
redis_results[key] = score
|
||||
|
||||
# Get results from linear scan
|
||||
linear_results = data.find_k_nearest(query_vec, k)
|
||||
linear_items = {name: score for name, score in linear_results}
|
||||
|
||||
# Compare overlap
|
||||
redis_set = set(redis_results.keys())
|
||||
linear_set = set(linear_items.keys())
|
||||
overlap = len(redis_set & linear_set)
|
||||
|
||||
# If test fails, print comparison for debugging
|
||||
if overlap < k * 0.7:
|
||||
data.print_comparison({'items': redis_results, 'query_vector': query_vec}, k)
|
||||
|
||||
assert overlap >= k * 0.7, \
|
||||
f"Expected at least 70% overlap in top {k} results, got {overlap/k*100:.1f}%"
|
||||
|
||||
# Verify scores for common items
|
||||
for item in redis_set & linear_set:
|
||||
redis_score = redis_results[item]
|
||||
linear_score = linear_items[item]
|
||||
assert abs(redis_score - linear_score) < 0.01, \
|
||||
f"Score mismatch for {item}: Redis={redis_score:.3f} Linear={linear_score:.3f}"
|
||||
@@ -0,0 +1,36 @@
|
||||
from test import TestCase, generate_random_vector
|
||||
import struct
|
||||
|
||||
class MemoryUsageTest(TestCase):
|
||||
def getname(self):
|
||||
return "[regression] MEMORY USAGE with attributes"
|
||||
|
||||
def test(self):
|
||||
# Generate random vectors
|
||||
vec1 = generate_random_vector(4)
|
||||
vec2 = generate_random_vector(4)
|
||||
vec_bytes1 = struct.pack('4f', *vec1)
|
||||
vec_bytes2 = struct.pack('4f', *vec2)
|
||||
|
||||
# Add vectors to the key, one with attribute, one without
|
||||
self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes1, f'{self.test_key}:item:1')
|
||||
self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes2, f'{self.test_key}:item:2', 'SETATTR', '{"color":"red"}')
|
||||
|
||||
# Get memory usage for the key
|
||||
try:
|
||||
memory_usage = self.redis.execute_command('MEMORY', 'USAGE', self.test_key)
|
||||
# If we got here without exception, the command worked
|
||||
assert memory_usage > 0, "MEMORY USAGE should return a positive value"
|
||||
|
||||
# Add more attributes to increase complexity
|
||||
self.redis.execute_command('VSETATTR', self.test_key, f'{self.test_key}:item:1', '{"color":"blue","size":10}')
|
||||
|
||||
# Check memory usage again
|
||||
new_memory_usage = self.redis.execute_command('MEMORY', 'USAGE', self.test_key)
|
||||
assert new_memory_usage > 0, "MEMORY USAGE should still return a positive value after setting attributes"
|
||||
|
||||
# Memory usage should be higher after adding attributes
|
||||
assert new_memory_usage > memory_usage, "Memory usage increase after adding attributes"
|
||||
|
||||
except Exception as e:
|
||||
raise AssertionError(f"MEMORY USAGE command failed: {str(e)}")
|
||||
@@ -0,0 +1,85 @@
|
||||
from test import TestCase, generate_random_vector
|
||||
import struct
|
||||
import math
|
||||
import random
|
||||
|
||||
class VectorUpdateAndClusters(TestCase):
|
||||
def getname(self):
|
||||
return "VADD vector update with cluster relocation"
|
||||
|
||||
def estimated_runtime(self):
|
||||
return 2.0 # Should take around 2 seconds
|
||||
|
||||
def generate_cluster_vector(self, base_vec, noise=0.1):
|
||||
"""Generate a vector that's similar to base_vec with some noise."""
|
||||
vec = [x + random.gauss(0, noise) for x in base_vec]
|
||||
# Normalize
|
||||
norm = math.sqrt(sum(x*x for x in vec))
|
||||
return [x/norm for x in vec]
|
||||
|
||||
def test(self):
|
||||
dim = 128
|
||||
vectors_per_cluster = 5000
|
||||
|
||||
# Create two very different base vectors for our clusters
|
||||
cluster1_base = generate_random_vector(dim)
|
||||
cluster2_base = [-x for x in cluster1_base] # Opposite direction
|
||||
|
||||
# Add vectors from first cluster
|
||||
for i in range(vectors_per_cluster):
|
||||
vec = self.generate_cluster_vector(cluster1_base)
|
||||
vec_bytes = struct.pack(f'{dim}f', *vec)
|
||||
self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes,
|
||||
f'{self.test_key}:cluster1:{i}')
|
||||
|
||||
# Add vectors from second cluster
|
||||
for i in range(vectors_per_cluster):
|
||||
vec = self.generate_cluster_vector(cluster2_base)
|
||||
vec_bytes = struct.pack(f'{dim}f', *vec)
|
||||
self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes,
|
||||
f'{self.test_key}:cluster2:{i}')
|
||||
|
||||
# Pick a test vector from cluster1
|
||||
test_key = f'{self.test_key}:cluster1:0'
|
||||
|
||||
# Verify it's in cluster1 using VSIM
|
||||
initial_vec = self.generate_cluster_vector(cluster1_base)
|
||||
results = self.redis.execute_command('VSIM', self.test_key, 'VALUES', dim,
|
||||
*[str(x) for x in initial_vec],
|
||||
'COUNT', 100, 'WITHSCORES')
|
||||
|
||||
# Count how many cluster1 items are in top results
|
||||
cluster1_count = sum(1 for i in range(0, len(results), 2)
|
||||
if b'cluster1' in results[i])
|
||||
assert cluster1_count > 80, "Initial clustering check failed"
|
||||
|
||||
# Now update the test vector to be in cluster2
|
||||
new_vec = self.generate_cluster_vector(cluster2_base, noise=0.05)
|
||||
vec_bytes = struct.pack(f'{dim}f', *new_vec)
|
||||
self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes, test_key)
|
||||
|
||||
# Verify the embedding was actually updated using VEMB
|
||||
emb_result = self.redis.execute_command('VEMB', self.test_key, test_key)
|
||||
updated_vec = [float(x) for x in emb_result]
|
||||
|
||||
# Verify updated vector matches what we inserted
|
||||
dot_product = sum(a*b for a,b in zip(updated_vec, new_vec))
|
||||
similarity = dot_product / (math.sqrt(sum(x*x for x in updated_vec)) *
|
||||
math.sqrt(sum(x*x for x in new_vec)))
|
||||
assert similarity > 0.9, "Vector was not properly updated"
|
||||
|
||||
# Verify it's now in cluster2 using VSIM
|
||||
results = self.redis.execute_command('VSIM', self.test_key, 'VALUES', dim,
|
||||
*[str(x) for x in cluster2_base],
|
||||
'COUNT', 100, 'WITHSCORES')
|
||||
|
||||
# Verify our updated vector is among top results
|
||||
found = False
|
||||
for i in range(0, len(results), 2):
|
||||
if results[i].decode() == test_key:
|
||||
found = True
|
||||
similarity = float(results[i+1])
|
||||
assert similarity > 0.80, f"Updated vector has low similarity: {similarity}"
|
||||
break
|
||||
|
||||
assert found, "Updated vector not found in cluster2 proximity"
|
||||
@@ -0,0 +1,83 @@
|
||||
from test import TestCase, fill_redis_with_vectors, generate_random_vector
|
||||
import random
|
||||
|
||||
class HNSWPersistence(TestCase):
|
||||
def getname(self):
|
||||
return "HNSW Persistence"
|
||||
|
||||
def estimated_runtime(self):
|
||||
return 30
|
||||
|
||||
def _verify_results(self, key, dim, query_vec, reduced_dim=None):
|
||||
"""Run a query and return results dict"""
|
||||
k = 10
|
||||
args = ['VSIM', key]
|
||||
|
||||
if reduced_dim:
|
||||
args.extend(['VALUES', dim])
|
||||
args.extend([str(x) for x in query_vec])
|
||||
else:
|
||||
args.extend(['VALUES', dim])
|
||||
args.extend([str(x) for x in query_vec])
|
||||
|
||||
args.extend(['COUNT', k, 'WITHSCORES'])
|
||||
results = self.redis.execute_command(*args)
|
||||
|
||||
results_dict = {}
|
||||
for i in range(0, len(results), 2):
|
||||
key = results[i].decode()
|
||||
score = float(results[i+1])
|
||||
results_dict[key] = score
|
||||
return results_dict
|
||||
|
||||
def test(self):
|
||||
# Setup dimensions
|
||||
dim = 128
|
||||
reduced_dim = 32
|
||||
count = 5000
|
||||
random.seed(42)
|
||||
|
||||
# Create two datasets - one normal and one with dimension reduction
|
||||
normal_data = fill_redis_with_vectors(self.redis, f"{self.test_key}:normal", count, dim)
|
||||
projected_data = fill_redis_with_vectors(self.redis, f"{self.test_key}:projected",
|
||||
count, dim, reduced_dim)
|
||||
|
||||
# Generate query vectors we'll use before and after reload
|
||||
query_vec_normal = generate_random_vector(dim)
|
||||
query_vec_projected = generate_random_vector(dim)
|
||||
|
||||
# Get initial results for both sets
|
||||
initial_normal = self._verify_results(f"{self.test_key}:normal",
|
||||
dim, query_vec_normal)
|
||||
initial_projected = self._verify_results(f"{self.test_key}:projected",
|
||||
dim, query_vec_projected, reduced_dim)
|
||||
|
||||
# Force Redis to save and reload the dataset
|
||||
self.redis.execute_command('DEBUG', 'RELOAD')
|
||||
|
||||
# Verify results after reload
|
||||
reloaded_normal = self._verify_results(f"{self.test_key}:normal",
|
||||
dim, query_vec_normal)
|
||||
reloaded_projected = self._verify_results(f"{self.test_key}:projected",
|
||||
dim, query_vec_projected, reduced_dim)
|
||||
|
||||
# Verify normal vectors results
|
||||
assert len(initial_normal) == len(reloaded_normal), \
|
||||
"Normal vectors: Result count mismatch before/after reload"
|
||||
|
||||
for key in initial_normal:
|
||||
assert key in reloaded_normal, f"Normal vectors: Missing item after reload: {key}"
|
||||
assert abs(initial_normal[key] - reloaded_normal[key]) < 0.0001, \
|
||||
f"Normal vectors: Score mismatch for {key}: " + \
|
||||
f"before={initial_normal[key]:.6f}, after={reloaded_normal[key]:.6f}"
|
||||
|
||||
# Verify projected vectors results
|
||||
assert len(initial_projected) == len(reloaded_projected), \
|
||||
"Projected vectors: Result count mismatch before/after reload"
|
||||
|
||||
for key in initial_projected:
|
||||
assert key in reloaded_projected, \
|
||||
f"Projected vectors: Missing item after reload: {key}"
|
||||
assert abs(initial_projected[key] - reloaded_projected[key]) < 0.0001, \
|
||||
f"Projected vectors: Score mismatch for {key}: " + \
|
||||
f"before={initial_projected[key]:.6f}, after={reloaded_projected[key]:.6f}"
|
||||
@@ -0,0 +1,71 @@
|
||||
from test import TestCase, fill_redis_with_vectors, generate_random_vector
|
||||
|
||||
class Reduce(TestCase):
|
||||
def getname(self):
|
||||
return "Dimension Reduction"
|
||||
|
||||
def estimated_runtime(self):
|
||||
return 0.2
|
||||
|
||||
def test(self):
|
||||
original_dim = 100
|
||||
reduced_dim = 80
|
||||
count = 1000
|
||||
k = 50 # Number of nearest neighbors to check
|
||||
|
||||
# Fill Redis with vectors using REDUCE and get reference data
|
||||
data = fill_redis_with_vectors(self.redis, self.test_key, count, original_dim, reduced_dim)
|
||||
|
||||
# Verify dimension is reduced
|
||||
dim = self.redis.execute_command('VDIM', self.test_key)
|
||||
assert dim == reduced_dim, f"Expected dimension {reduced_dim}, got {dim}"
|
||||
|
||||
# Generate query vector and get nearest neighbors using Redis
|
||||
query_vec = generate_random_vector(original_dim)
|
||||
redis_raw = self.redis.execute_command('VSIM', self.test_key, 'VALUES',
|
||||
original_dim, *[str(x) for x in query_vec],
|
||||
'COUNT', k, 'WITHSCORES')
|
||||
|
||||
# Convert Redis results to dict
|
||||
redis_results = {}
|
||||
for i in range(0, len(redis_raw), 2):
|
||||
key = redis_raw[i].decode()
|
||||
score = float(redis_raw[i+1])
|
||||
redis_results[key] = score
|
||||
|
||||
# Get results from linear scan with original vectors
|
||||
linear_results = data.find_k_nearest(query_vec, k)
|
||||
linear_items = {name: score for name, score in linear_results}
|
||||
|
||||
# Compare overlap between reduced and non-reduced results
|
||||
redis_set = set(redis_results.keys())
|
||||
linear_set = set(linear_items.keys())
|
||||
overlap = len(redis_set & linear_set)
|
||||
overlap_ratio = overlap / k
|
||||
|
||||
# With random projection, we expect some loss of accuracy but should
|
||||
# maintain at least some similarity structure.
|
||||
# Note that gaussian distribution is the worse with this test, so
|
||||
# in real world practice, things will be better.
|
||||
min_expected_overlap = 0.1 # At least 10% overlap in top-k
|
||||
assert overlap_ratio >= min_expected_overlap, \
|
||||
f"Dimension reduction lost too much structure. Only {overlap_ratio*100:.1f}% overlap in top {k}"
|
||||
|
||||
# For items that appear in both results, scores should be reasonably correlated
|
||||
common_items = redis_set & linear_set
|
||||
for item in common_items:
|
||||
redis_score = redis_results[item]
|
||||
linear_score = linear_items[item]
|
||||
# Allow for some deviation due to dimensionality reduction
|
||||
assert abs(redis_score - linear_score) < 0.2, \
|
||||
f"Score mismatch too high for {item}: Redis={redis_score:.3f} Linear={linear_score:.3f}"
|
||||
|
||||
# If test fails, print comparison for debugging
|
||||
if overlap_ratio < min_expected_overlap:
|
||||
print("\nLow overlap in results. Details:")
|
||||
print("\nTop results from linear scan (original vectors):")
|
||||
for name, score in linear_results:
|
||||
print(f"{name}: {score:.3f}")
|
||||
print("\nTop results from Redis (reduced vectors):")
|
||||
for item, score in sorted(redis_results.items(), key=lambda x: x[1], reverse=True):
|
||||
print(f"{item}: {score:.3f}")
|
||||
@@ -0,0 +1,92 @@
|
||||
from test import TestCase, generate_random_vector
|
||||
import struct
|
||||
import random
|
||||
import time
|
||||
|
||||
class ComprehensiveReplicationTest(TestCase):
|
||||
def getname(self):
|
||||
return "Comprehensive Replication Test with mixed operations"
|
||||
|
||||
def estimated_runtime(self):
|
||||
# This test will take longer than the default 100ms
|
||||
return 20.0 # 20 seconds estimate
|
||||
|
||||
def test(self):
|
||||
# Setup replication between primary and replica
|
||||
assert self.setup_replication(), "Failed to setup replication"
|
||||
|
||||
# Test parameters
|
||||
num_vectors = 5000
|
||||
vector_dim = 8
|
||||
delete_probability = 0.1
|
||||
cas_probability = 0.3
|
||||
|
||||
# Keep track of added items for potential deletion
|
||||
added_items = []
|
||||
|
||||
# Add vectors and occasionally delete
|
||||
for i in range(num_vectors):
|
||||
# Generate a random vector
|
||||
vec = generate_random_vector(vector_dim)
|
||||
vec_bytes = struct.pack(f'{vector_dim}f', *vec)
|
||||
item_name = f"{self.test_key}:item:{i}"
|
||||
|
||||
# Decide whether to use CAS or not
|
||||
use_cas = random.random() < cas_probability
|
||||
|
||||
if use_cas and added_items:
|
||||
# Get an existing item for CAS reference (if available)
|
||||
cas_item = random.choice(added_items)
|
||||
try:
|
||||
# Add with CAS
|
||||
result = self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes,
|
||||
item_name, 'CAS')
|
||||
# Only add to our list if actually added (CAS might fail)
|
||||
if result == 1:
|
||||
added_items.append(item_name)
|
||||
except Exception as e:
|
||||
print(f" CAS VADD failed: {e}")
|
||||
else:
|
||||
try:
|
||||
# Add without CAS
|
||||
result = self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes, item_name)
|
||||
# Only add to our list if actually added
|
||||
if result == 1:
|
||||
added_items.append(item_name)
|
||||
except Exception as e:
|
||||
print(f" VADD failed: {e}")
|
||||
|
||||
# Randomly delete items (with 10% probability)
|
||||
if random.random() < delete_probability and added_items:
|
||||
try:
|
||||
# Select a random item to delete
|
||||
item_to_delete = random.choice(added_items)
|
||||
# Delete the item using VREM (not VDEL)
|
||||
self.redis.execute_command('VREM', self.test_key, item_to_delete)
|
||||
# Remove from our list
|
||||
added_items.remove(item_to_delete)
|
||||
except Exception as e:
|
||||
print(f" VREM failed: {e}")
|
||||
|
||||
# Allow time for replication to complete
|
||||
time.sleep(2.0)
|
||||
|
||||
# Verify final VCARD matches
|
||||
primary_card = self.redis.execute_command('VCARD', self.test_key)
|
||||
replica_card = self.replica.execute_command('VCARD', self.test_key)
|
||||
assert primary_card == replica_card, f"Final VCARD mismatch: primary={primary_card}, replica={replica_card}"
|
||||
|
||||
# Verify VDIM matches
|
||||
primary_dim = self.redis.execute_command('VDIM', self.test_key)
|
||||
replica_dim = self.replica.execute_command('VDIM', self.test_key)
|
||||
assert primary_dim == replica_dim, f"VDIM mismatch: primary={primary_dim}, replica={replica_dim}"
|
||||
|
||||
# Verify digests match using DEBUG DIGEST
|
||||
primary_digest = self.redis.execute_command('DEBUG', 'DIGEST-VALUE', self.test_key)
|
||||
replica_digest = self.replica.execute_command('DEBUG', 'DIGEST-VALUE', self.test_key)
|
||||
assert primary_digest == replica_digest, f"Digest mismatch: primary={primary_digest}, replica={replica_digest}"
|
||||
|
||||
# Print summary
|
||||
print(f"\n Added and maintained {len(added_items)} vectors with dimension {vector_dim}")
|
||||
print(f" Final vector count: {primary_card}")
|
||||
print(f" Final digest: {primary_digest[0].decode()}")
|
||||
@@ -0,0 +1,98 @@
|
||||
from test import TestCase, generate_random_vector
|
||||
import threading
|
||||
import struct
|
||||
import math
|
||||
import time
|
||||
import random
|
||||
from typing import List, Dict
|
||||
|
||||
class ConcurrentCASTest(TestCase):
|
||||
def getname(self):
|
||||
return "Concurrent VADD with CAS"
|
||||
|
||||
def estimated_runtime(self):
|
||||
return 1.5
|
||||
|
||||
def worker(self, vectors: List[List[float]], start_idx: int, end_idx: int,
|
||||
dim: int, results: Dict[str, bool]):
|
||||
"""Worker thread that adds a subset of vectors using VADD CAS"""
|
||||
for i in range(start_idx, end_idx):
|
||||
vec = vectors[i]
|
||||
name = f"{self.test_key}:item:{i}"
|
||||
vec_bytes = struct.pack(f'{dim}f', *vec)
|
||||
|
||||
# Try to add the vector with CAS
|
||||
try:
|
||||
result = self.redis.execute_command('VADD', self.test_key, 'FP32',
|
||||
vec_bytes, name, 'CAS')
|
||||
results[name] = (result == 1) # Store if it was actually added
|
||||
except Exception as e:
|
||||
results[name] = False
|
||||
print(f"Error adding {name}: {e}")
|
||||
|
||||
def verify_vector_similarity(self, vec1: List[float], vec2: List[float]) -> float:
|
||||
"""Calculate cosine similarity between two vectors"""
|
||||
dot_product = sum(a*b for a,b in zip(vec1, vec2))
|
||||
norm1 = math.sqrt(sum(x*x for x in vec1))
|
||||
norm2 = math.sqrt(sum(x*x for x in vec2))
|
||||
return dot_product / (norm1 * norm2) if norm1 > 0 and norm2 > 0 else 0
|
||||
|
||||
def test(self):
|
||||
# Test parameters
|
||||
dim = 128
|
||||
total_vectors = 5000
|
||||
num_threads = 8
|
||||
vectors_per_thread = total_vectors // num_threads
|
||||
|
||||
# Generate all vectors upfront
|
||||
random.seed(42) # For reproducibility
|
||||
vectors = [generate_random_vector(dim) for _ in range(total_vectors)]
|
||||
|
||||
# Prepare threads and results dictionary
|
||||
threads = []
|
||||
results = {} # Will store success/failure for each vector
|
||||
|
||||
# Launch threads
|
||||
for i in range(num_threads):
|
||||
start_idx = i * vectors_per_thread
|
||||
end_idx = start_idx + vectors_per_thread if i < num_threads-1 else total_vectors
|
||||
thread = threading.Thread(target=self.worker,
|
||||
args=(vectors, start_idx, end_idx, dim, results))
|
||||
threads.append(thread)
|
||||
thread.start()
|
||||
|
||||
# Wait for all threads to complete
|
||||
for thread in threads:
|
||||
thread.join()
|
||||
|
||||
# Verify cardinality
|
||||
card = self.redis.execute_command('VCARD', self.test_key)
|
||||
assert card == total_vectors, \
|
||||
f"Expected {total_vectors} elements, but found {card}"
|
||||
|
||||
# Verify each vector
|
||||
num_verified = 0
|
||||
for i in range(total_vectors):
|
||||
name = f"{self.test_key}:item:{i}"
|
||||
|
||||
# Verify the item was successfully added
|
||||
assert results[name], f"Vector {name} was not successfully added"
|
||||
|
||||
# Get the stored vector
|
||||
stored_vec_raw = self.redis.execute_command('VEMB', self.test_key, name)
|
||||
stored_vec = [float(x) for x in stored_vec_raw]
|
||||
|
||||
# Verify vector dimensions
|
||||
assert len(stored_vec) == dim, \
|
||||
f"Stored vector dimension mismatch for {name}: {len(stored_vec)} != {dim}"
|
||||
|
||||
# Calculate similarity with original vector
|
||||
similarity = self.verify_vector_similarity(vectors[i], stored_vec)
|
||||
assert similarity > 0.99, \
|
||||
f"Low similarity ({similarity}) for {name}"
|
||||
|
||||
num_verified += 1
|
||||
|
||||
# Final verification
|
||||
assert num_verified == total_vectors, \
|
||||
f"Only verified {num_verified} out of {total_vectors} vectors"
|
||||
@@ -0,0 +1,41 @@
|
||||
from test import TestCase
|
||||
import struct
|
||||
import math
|
||||
|
||||
class VEMB(TestCase):
|
||||
def getname(self):
|
||||
return "VEMB Command"
|
||||
|
||||
def test(self):
|
||||
dim = 4
|
||||
|
||||
# Add same vector in both formats
|
||||
vec = [1, 0, 0, 0]
|
||||
norm = math.sqrt(sum(x*x for x in vec))
|
||||
vec = [x/norm for x in vec] # Normalize the vector
|
||||
|
||||
# Add using FP32
|
||||
vec_bytes = struct.pack(f'{dim}f', *vec)
|
||||
self.redis.execute_command('VADD', self.test_key, 'FP32', vec_bytes, f'{self.test_key}:item:1')
|
||||
|
||||
# Add using VALUES
|
||||
self.redis.execute_command('VADD', self.test_key, 'VALUES', dim,
|
||||
*[str(x) for x in vec], f'{self.test_key}:item:2')
|
||||
|
||||
# Get both back with VEMB
|
||||
result1 = self.redis.execute_command('VEMB', self.test_key, f'{self.test_key}:item:1')
|
||||
result2 = self.redis.execute_command('VEMB', self.test_key, f'{self.test_key}:item:2')
|
||||
|
||||
retrieved_vec1 = [float(x) for x in result1]
|
||||
retrieved_vec2 = [float(x) for x in result2]
|
||||
|
||||
# Compare both vectors with original (allow for small quantization errors)
|
||||
for i in range(dim):
|
||||
assert abs(vec[i] - retrieved_vec1[i]) < 0.01, \
|
||||
f"FP32 vector component {i} mismatch: expected {vec[i]}, got {retrieved_vec1[i]}"
|
||||
assert abs(vec[i] - retrieved_vec2[i]) < 0.01, \
|
||||
f"VALUES vector component {i} mismatch: expected {vec[i]}, got {retrieved_vec2[i]}"
|
||||
|
||||
# Test non-existent item
|
||||
result = self.redis.execute_command('VEMB', self.test_key, 'nonexistent')
|
||||
assert result is None, "Non-existent item should return nil"
|
||||
@@ -0,0 +1,55 @@
|
||||
from test import TestCase, generate_random_vector, fill_redis_with_vectors
|
||||
import struct
|
||||
|
||||
class VRANDMEMBERTest(TestCase):
|
||||
def getname(self):
|
||||
return "VRANDMEMBER basic functionality"
|
||||
|
||||
def test(self):
|
||||
# Test with empty key
|
||||
result = self.redis.execute_command('VRANDMEMBER', self.test_key)
|
||||
assert result is None, "VRANDMEMBER on non-existent key should return NULL"
|
||||
|
||||
result = self.redis.execute_command('VRANDMEMBER', self.test_key, 5)
|
||||
assert isinstance(result, list) and len(result) == 0, "VRANDMEMBER with count on non-existent key should return empty array"
|
||||
|
||||
# Fill with vectors
|
||||
dim = 4
|
||||
count = 100
|
||||
data = fill_redis_with_vectors(self.redis, self.test_key, count, dim)
|
||||
|
||||
# Test single random member
|
||||
result = self.redis.execute_command('VRANDMEMBER', self.test_key)
|
||||
assert result is not None, "VRANDMEMBER should return a random member"
|
||||
assert result.decode() in data.names, "Random member should be in the set"
|
||||
|
||||
# Test multiple unique members (positive count)
|
||||
positive_count = 10
|
||||
result = self.redis.execute_command('VRANDMEMBER', self.test_key, positive_count)
|
||||
assert isinstance(result, list), "VRANDMEMBER with positive count should return an array"
|
||||
assert len(result) == positive_count, f"Should return {positive_count} members"
|
||||
|
||||
# Check for uniqueness
|
||||
decoded_results = [r.decode() for r in result]
|
||||
assert len(decoded_results) == len(set(decoded_results)), "Results should be unique with positive count"
|
||||
for item in decoded_results:
|
||||
assert item in data.names, "All returned items should be in the set"
|
||||
|
||||
# Test more members than in the set
|
||||
result = self.redis.execute_command('VRANDMEMBER', self.test_key, count + 10)
|
||||
assert len(result) == count, "Should return only the available members when asking for more than exist"
|
||||
|
||||
# Test with duplicates (negative count)
|
||||
negative_count = -20
|
||||
result = self.redis.execute_command('VRANDMEMBER', self.test_key, negative_count)
|
||||
assert isinstance(result, list), "VRANDMEMBER with negative count should return an array"
|
||||
assert len(result) == abs(negative_count), f"Should return {abs(negative_count)} members"
|
||||
|
||||
# Check that all returned elements are valid
|
||||
decoded_results = [r.decode() for r in result]
|
||||
for item in decoded_results:
|
||||
assert item in data.names, "All returned items should be in the set"
|
||||
|
||||
# Test with count = 0 (edge case)
|
||||
result = self.redis.execute_command('VRANDMEMBER', self.test_key, 0)
|
||||
assert isinstance(result, list) and len(result) == 0, "VRANDMEMBER with count=0 should return empty array"
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,514 @@
|
||||
/*
|
||||
* HNSW (Hierarchical Navigable Small World) Implementation
|
||||
* Based on the paper by Yu. A. Malkov, D. A. Yashunin
|
||||
*
|
||||
* Copyright (c) 2009-Present, Redis Ltd.
|
||||
* All rights reserved.
|
||||
*
|
||||
* Licensed under your choice of the Redis Source Available License 2.0
|
||||
* (RSALv2) or the Server Side Public License v1 (SSPLv1).
|
||||
* Originally authored by: Salvatore Sanfilippo
|
||||
*/
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <sys/time.h>
|
||||
#include <time.h>
|
||||
#include <stdint.h>
|
||||
#include <pthread.h>
|
||||
#include <stdatomic.h>
|
||||
#include <math.h>
|
||||
|
||||
#include "hnsw.h"
|
||||
|
||||
/* Get current time in milliseconds */
|
||||
uint64_t ms_time(void) {
|
||||
struct timeval tv;
|
||||
gettimeofday(&tv, NULL);
|
||||
return (uint64_t)tv.tv_sec * 1000 + (tv.tv_usec / 1000);
|
||||
}
|
||||
|
||||
/* Implementation of the recall test with random vectors. */
|
||||
void test_recall(HNSW *index, int ef) {
|
||||
const int num_test_vectors = 10000;
|
||||
const int k = 100; // Number of nearest neighbors to find.
|
||||
if (ef < k) ef = k;
|
||||
|
||||
// Add recall distribution counters (2% bins from 0-100%).
|
||||
int recall_bins[50] = {0};
|
||||
|
||||
// Create array to store vectors for mixing.
|
||||
int num_source_vectors = 1000; // Enough, since we mix them.
|
||||
float **source_vectors = malloc(sizeof(float*) * num_source_vectors);
|
||||
if (!source_vectors) {
|
||||
printf("Failed to allocate memory for source vectors\n");
|
||||
return;
|
||||
}
|
||||
|
||||
// Allocate memory for each source vector.
|
||||
for (int i = 0; i < num_source_vectors; i++) {
|
||||
source_vectors[i] = malloc(sizeof(float) * 300);
|
||||
if (!source_vectors[i]) {
|
||||
printf("Failed to allocate memory for source vector %d\n", i);
|
||||
// Clean up already allocated vectors.
|
||||
for (int j = 0; j < i; j++) free(source_vectors[j]);
|
||||
free(source_vectors);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
/* Populate source vectors from the index, we just scan the
|
||||
* first N items. */
|
||||
int source_count = 0;
|
||||
hnswNode *current = index->head;
|
||||
while (current && source_count < num_source_vectors) {
|
||||
hnsw_get_node_vector(index, current, source_vectors[source_count]);
|
||||
source_count++;
|
||||
current = current->next;
|
||||
}
|
||||
|
||||
if (source_count < num_source_vectors) {
|
||||
printf("Warning: Only found %d nodes for source vectors\n",
|
||||
source_count);
|
||||
num_source_vectors = source_count;
|
||||
}
|
||||
|
||||
// Allocate memory for test vector.
|
||||
float *test_vector = malloc(sizeof(float) * 300);
|
||||
if (!test_vector) {
|
||||
printf("Failed to allocate memory for test vector\n");
|
||||
for (int i = 0; i < num_source_vectors; i++) {
|
||||
free(source_vectors[i]);
|
||||
}
|
||||
free(source_vectors);
|
||||
return;
|
||||
}
|
||||
|
||||
// Allocate memory for results.
|
||||
hnswNode **hnsw_results = malloc(sizeof(hnswNode*) * ef);
|
||||
hnswNode **linear_results = malloc(sizeof(hnswNode*) * ef);
|
||||
float *hnsw_distances = malloc(sizeof(float) * ef);
|
||||
float *linear_distances = malloc(sizeof(float) * ef);
|
||||
|
||||
if (!hnsw_results || !linear_results || !hnsw_distances || !linear_distances) {
|
||||
printf("Failed to allocate memory for results\n");
|
||||
if (hnsw_results) free(hnsw_results);
|
||||
if (linear_results) free(linear_results);
|
||||
if (hnsw_distances) free(hnsw_distances);
|
||||
if (linear_distances) free(linear_distances);
|
||||
for (int i = 0; i < num_source_vectors; i++) free(source_vectors[i]);
|
||||
free(source_vectors);
|
||||
free(test_vector);
|
||||
return;
|
||||
}
|
||||
|
||||
// Initialize random seed.
|
||||
srand(time(NULL));
|
||||
|
||||
// Perform recall test.
|
||||
printf("\nPerforming recall test with EF=%d on %d random vectors...\n",
|
||||
ef, num_test_vectors);
|
||||
double total_recall = 0.0;
|
||||
|
||||
for (int t = 0; t < num_test_vectors; t++) {
|
||||
// Create a random vector by mixing 3 existing vectors.
|
||||
float weights[3] = {0.0};
|
||||
int src_indices[3] = {0};
|
||||
|
||||
// Generate random weights.
|
||||
float weight_sum = 0.0;
|
||||
for (int i = 0; i < 3; i++) {
|
||||
weights[i] = (float)rand() / RAND_MAX;
|
||||
weight_sum += weights[i];
|
||||
src_indices[i] = rand() % num_source_vectors;
|
||||
}
|
||||
|
||||
// Normalize weights.
|
||||
for (int i = 0; i < 3; i++) weights[i] /= weight_sum;
|
||||
|
||||
// Mix vectors.
|
||||
memset(test_vector, 0, sizeof(float) * 300);
|
||||
for (int i = 0; i < 3; i++) {
|
||||
for (int j = 0; j < 300; j++) {
|
||||
test_vector[j] +=
|
||||
weights[i] * source_vectors[src_indices[i]][j];
|
||||
}
|
||||
}
|
||||
|
||||
// Perform HNSW search with the specified EF parameter.
|
||||
int slot = hnsw_acquire_read_slot(index);
|
||||
int hnsw_found = hnsw_search(index, test_vector, ef, hnsw_results, hnsw_distances, slot, 0);
|
||||
|
||||
// Perform linear search (ground truth).
|
||||
int linear_found = hnsw_ground_truth_with_filter(index, test_vector, ef, linear_results, linear_distances, slot, 0, NULL, NULL);
|
||||
hnsw_release_read_slot(index, slot);
|
||||
|
||||
// Calculate recall for this query (intersection size / k).
|
||||
if (hnsw_found > k) hnsw_found = k;
|
||||
if (linear_found > k) linear_found = k;
|
||||
int intersection_count = 0;
|
||||
for (int i = 0; i < linear_found; i++) {
|
||||
for (int j = 0; j < hnsw_found; j++) {
|
||||
if (linear_results[i] == hnsw_results[j]) {
|
||||
intersection_count++;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
double recall = (double)intersection_count / linear_found;
|
||||
total_recall += recall;
|
||||
|
||||
// Add to distribution bins (2% steps)
|
||||
int bin_index = (int)(recall * 50);
|
||||
if (bin_index >= 50) bin_index = 49; // Handle 100% recall case
|
||||
recall_bins[bin_index]++;
|
||||
|
||||
// Show progress.
|
||||
if ((t+1) % 1000 == 0 || t == num_test_vectors-1) {
|
||||
printf("Processed %d/%d queries, current avg recall: %.2f%%\n",
|
||||
t+1, num_test_vectors, (total_recall / (t+1)) * 100);
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate and print final average recall.
|
||||
double avg_recall = (total_recall / num_test_vectors) * 100;
|
||||
printf("\nRecall Test Results:\n");
|
||||
printf("Average recall@%d (EF=%d): %.2f%%\n", k, ef, avg_recall);
|
||||
|
||||
// Print recall distribution histogram.
|
||||
printf("\nRecall Distribution (2%% bins):\n");
|
||||
printf("================================\n");
|
||||
|
||||
// Find the maximum bin count for scaling.
|
||||
int max_count = 0;
|
||||
for (int i = 0; i < 50; i++) {
|
||||
if (recall_bins[i] > max_count) max_count = recall_bins[i];
|
||||
}
|
||||
|
||||
// Scale factor for histogram (max 50 chars wide)
|
||||
const int max_bars = 50;
|
||||
double scale = (max_count > max_bars) ? (double)max_bars / max_count : 1.0;
|
||||
|
||||
// Print the histogram.
|
||||
for (int i = 0; i < 50; i++) {
|
||||
int bar_len = (int)(recall_bins[i] * scale);
|
||||
printf("%3d%%-%-3d%% | %-6d |", i*2, (i+1)*2, recall_bins[i]);
|
||||
for (int j = 0; j < bar_len; j++) printf("#");
|
||||
printf("\n");
|
||||
}
|
||||
|
||||
// Cleanup.
|
||||
free(hnsw_results);
|
||||
free(linear_results);
|
||||
free(hnsw_distances);
|
||||
free(linear_distances);
|
||||
free(test_vector);
|
||||
for (int i = 0; i < num_source_vectors; i++) free(source_vectors[i]);
|
||||
free(source_vectors);
|
||||
}
|
||||
|
||||
/* Example usage in main() */
|
||||
int w2v_single_thread(int m_param, int quantization, uint64_t numele, int massdel, int self_recall, int recall_ef) {
|
||||
/* Create index */
|
||||
HNSW *index = hnsw_new(300, quantization, m_param);
|
||||
float v[300];
|
||||
uint16_t wlen;
|
||||
|
||||
FILE *fp = fopen("word2vec.bin","rb");
|
||||
if (fp == NULL) {
|
||||
perror("word2vec.bin file missing");
|
||||
exit(1);
|
||||
}
|
||||
unsigned char header[8];
|
||||
fread(header,8,1,fp); // Skip header
|
||||
|
||||
uint64_t id = 0;
|
||||
uint64_t start_time = ms_time();
|
||||
char *word = NULL;
|
||||
hnswNode *search_node = NULL;
|
||||
|
||||
while(id < numele) {
|
||||
if (fread(&wlen,2,1,fp) == 0) break;
|
||||
word = malloc(wlen+1);
|
||||
fread(word,wlen,1,fp);
|
||||
word[wlen] = 0;
|
||||
fread(v,300*sizeof(float),1,fp);
|
||||
|
||||
// Plain API that acquires a write lock for the whole time.
|
||||
hnswNode *added = hnsw_insert(index, v, NULL, 0, id++, word, 200);
|
||||
|
||||
if (!strcmp(word,"banana")) search_node = added;
|
||||
if (!(id % 10000)) printf("%llu added\n", (unsigned long long)id);
|
||||
}
|
||||
uint64_t elapsed = ms_time() - start_time;
|
||||
fclose(fp);
|
||||
|
||||
printf("%llu words added (%llu words/sec), last word: %s\n",
|
||||
(unsigned long long)index->node_count,
|
||||
(unsigned long long)id*1000/elapsed, word);
|
||||
|
||||
/* Search query */
|
||||
if (search_node == NULL) search_node = index->head;
|
||||
hnsw_get_node_vector(index,search_node,v);
|
||||
hnswNode *neighbors[10];
|
||||
float distances[10];
|
||||
|
||||
int found, j;
|
||||
start_time = ms_time();
|
||||
for (j = 0; j < 20000; j++)
|
||||
found = hnsw_search(index, v, 10, neighbors, distances, 0, 0);
|
||||
elapsed = ms_time() - start_time;
|
||||
printf("%d searches performed (%llu searches/sec), nodes found: %d\n",
|
||||
j, (unsigned long long)j*1000/elapsed, found);
|
||||
|
||||
if (found > 0) {
|
||||
printf("Found %d neighbors:\n", found);
|
||||
for (int i = 0; i < found; i++) {
|
||||
printf("Node ID: %llu, distance: %f, word: %s\n",
|
||||
(unsigned long long)neighbors[i]->id,
|
||||
distances[i], (char*)neighbors[i]->value);
|
||||
}
|
||||
}
|
||||
|
||||
// Self-recall test (ability to find the node by its own vector).
|
||||
if (self_recall) {
|
||||
hnsw_print_stats(index);
|
||||
hnsw_test_graph_recall(index,200,0);
|
||||
}
|
||||
|
||||
// Recall test with random vectors.
|
||||
if (recall_ef > 0) {
|
||||
test_recall(index, recall_ef);
|
||||
}
|
||||
|
||||
uint64_t connected_nodes;
|
||||
int reciprocal_links;
|
||||
hnsw_validate_graph(index, &connected_nodes, &reciprocal_links);
|
||||
|
||||
if (massdel) {
|
||||
int remove_perc = 95;
|
||||
printf("\nRemoving %d%% of nodes...\n", remove_perc);
|
||||
uint64_t initial_nodes = index->node_count;
|
||||
|
||||
hnswNode *current = index->head;
|
||||
while (current && index->node_count > initial_nodes*(100-remove_perc)/100) {
|
||||
hnswNode *next = current->next;
|
||||
hnsw_delete_node(index,current,free);
|
||||
current = next;
|
||||
// In order to don't remove only contiguous nodes, from time
|
||||
// skip a node.
|
||||
if (current && !(random() % remove_perc)) current = current->next;
|
||||
}
|
||||
printf("%llu nodes left\n", (unsigned long long)index->node_count);
|
||||
|
||||
// Test again.
|
||||
hnsw_validate_graph(index, &connected_nodes, &reciprocal_links);
|
||||
hnsw_test_graph_recall(index,200,0);
|
||||
}
|
||||
|
||||
hnsw_free(index,free);
|
||||
return 0;
|
||||
}
|
||||
|
||||
struct threadContext {
|
||||
pthread_mutex_t FileAccessMutex;
|
||||
uint64_t numele;
|
||||
_Atomic uint64_t SearchesDone;
|
||||
_Atomic uint64_t id;
|
||||
FILE *fp;
|
||||
HNSW *index;
|
||||
float *search_vector;
|
||||
};
|
||||
|
||||
// Note that in practical terms inserting with many concurrent threads
|
||||
// may be *slower* and not faster, because there is a lot of
|
||||
// contention. So this is more a robustness test than anything else.
|
||||
//
|
||||
// The optimistic commit API goal is actually to exploit the ability to
|
||||
// add faster when there are many concurrent reads.
|
||||
void *threaded_insert(void *ctxptr) {
|
||||
struct threadContext *ctx = ctxptr;
|
||||
char *word;
|
||||
float v[300];
|
||||
uint16_t wlen;
|
||||
|
||||
while(1) {
|
||||
pthread_mutex_lock(&ctx->FileAccessMutex);
|
||||
if (fread(&wlen,2,1,ctx->fp) == 0) break;
|
||||
pthread_mutex_unlock(&ctx->FileAccessMutex);
|
||||
word = malloc(wlen+1);
|
||||
fread(word,wlen,1,ctx->fp);
|
||||
word[wlen] = 0;
|
||||
fread(v,300*sizeof(float),1,ctx->fp);
|
||||
|
||||
// Check-and-set API that performs the costly scan for similar
|
||||
// nodes concurrently with other read threads, and finally
|
||||
// applies the check if the graph wasn't modified.
|
||||
InsertContext *ic;
|
||||
uint64_t next_id = ctx->id++;
|
||||
ic = hnsw_prepare_insert(ctx->index, v, NULL, 0, next_id, 200);
|
||||
if (hnsw_try_commit_insert(ctx->index, ic, word) == NULL) {
|
||||
// This time try locking since the start.
|
||||
hnsw_insert(ctx->index, v, NULL, 0, next_id, word, 200);
|
||||
}
|
||||
|
||||
if (next_id >= ctx->numele) break;
|
||||
if (!((next_id+1) % 10000))
|
||||
printf("%llu added\n", (unsigned long long)next_id+1);
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void *threaded_search(void *ctxptr) {
|
||||
struct threadContext *ctx = ctxptr;
|
||||
|
||||
/* Search query */
|
||||
hnswNode *neighbors[10];
|
||||
float distances[10];
|
||||
int found = 0;
|
||||
uint64_t last_id = 0;
|
||||
|
||||
while(ctx->id < 1000000) {
|
||||
int slot = hnsw_acquire_read_slot(ctx->index);
|
||||
found = hnsw_search(ctx->index, ctx->search_vector, 10, neighbors, distances, slot, 0);
|
||||
hnsw_release_read_slot(ctx->index,slot);
|
||||
last_id = ++ctx->id;
|
||||
}
|
||||
|
||||
if (found > 0 && last_id == 1000000) {
|
||||
printf("Found %d neighbors:\n", found);
|
||||
for (int i = 0; i < found; i++) {
|
||||
printf("Node ID: %llu, distance: %f, word: %s\n",
|
||||
(unsigned long long)neighbors[i]->id,
|
||||
distances[i], (char*)neighbors[i]->value);
|
||||
}
|
||||
}
|
||||
return NULL;
|
||||
}
|
||||
|
||||
int w2v_multi_thread(int m_param, int numthreads, int quantization, uint64_t numele) {
|
||||
/* Create index */
|
||||
struct threadContext ctx;
|
||||
|
||||
ctx.index = hnsw_new(300, quantization, m_param);
|
||||
|
||||
ctx.fp = fopen("word2vec.bin","rb");
|
||||
if (ctx.fp == NULL) {
|
||||
perror("word2vec.bin file missing");
|
||||
exit(1);
|
||||
}
|
||||
|
||||
unsigned char header[8];
|
||||
fread(header,8,1,ctx.fp); // Skip header
|
||||
pthread_mutex_init(&ctx.FileAccessMutex,NULL);
|
||||
|
||||
uint64_t start_time = ms_time();
|
||||
ctx.id = 0;
|
||||
ctx.numele = numele;
|
||||
pthread_t threads[numthreads];
|
||||
for (int j = 0; j < numthreads; j++)
|
||||
pthread_create(&threads[j], NULL, threaded_insert, &ctx);
|
||||
|
||||
// Wait for all the threads to terminate adding items.
|
||||
for (int j = 0; j < numthreads; j++)
|
||||
pthread_join(threads[j],NULL);
|
||||
|
||||
uint64_t elapsed = ms_time() - start_time;
|
||||
fclose(ctx.fp);
|
||||
|
||||
// Obtain the last word.
|
||||
hnswNode *node = ctx.index->head;
|
||||
char *word = node->value;
|
||||
|
||||
// We will search this last inserted word in the next test.
|
||||
// Let's save its embedding.
|
||||
ctx.search_vector = malloc(sizeof(float)*300);
|
||||
hnsw_get_node_vector(ctx.index,node,ctx.search_vector);
|
||||
|
||||
printf("%llu words added (%llu words/sec), last word: %s\n",
|
||||
(unsigned long long)ctx.index->node_count,
|
||||
(unsigned long long)ctx.id*1000/elapsed, word);
|
||||
|
||||
/* Search query */
|
||||
start_time = ms_time();
|
||||
ctx.id = 0; // We will use this atomic field to stop at N queries done.
|
||||
|
||||
for (int j = 0; j < numthreads; j++)
|
||||
pthread_create(&threads[j], NULL, threaded_search, &ctx);
|
||||
|
||||
// Wait for all the threads to terminate searching.
|
||||
for (int j = 0; j < numthreads; j++)
|
||||
pthread_join(threads[j],NULL);
|
||||
|
||||
elapsed = ms_time() - start_time;
|
||||
printf("%llu searches performed (%llu searches/sec)\n",
|
||||
(unsigned long long)ctx.id,
|
||||
(unsigned long long)ctx.id*1000/elapsed);
|
||||
|
||||
hnsw_print_stats(ctx.index);
|
||||
uint64_t connected_nodes;
|
||||
int reciprocal_links;
|
||||
hnsw_validate_graph(ctx.index, &connected_nodes, &reciprocal_links);
|
||||
printf("%llu connected nodes. Links all reciprocal: %d\n",
|
||||
(unsigned long long)connected_nodes, reciprocal_links);
|
||||
hnsw_free(ctx.index,free);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
int quantization = HNSW_QUANT_NONE;
|
||||
int numthreads = 0;
|
||||
uint64_t numele = 20000;
|
||||
int m_param = 0; // Default value (0 means use HNSW_DEFAULT_M)
|
||||
|
||||
/* This you can enable in single thread mode for testing: */
|
||||
int massdel = 0; // If true, does the mass deletion test.
|
||||
int self_recall = 0; // If true, does the self-recall test.
|
||||
int recall_ef = 0; // If not 0, does the recall test with this EF value.
|
||||
|
||||
for (int j = 1; j < argc; j++) {
|
||||
int moreargs = argc-j-1;
|
||||
|
||||
if (!strcasecmp(argv[j],"--quant")) {
|
||||
quantization = HNSW_QUANT_Q8;
|
||||
} else if (!strcasecmp(argv[j],"--bin")) {
|
||||
quantization = HNSW_QUANT_BIN;
|
||||
} else if (!strcasecmp(argv[j],"--mass-del")) {
|
||||
massdel = 1;
|
||||
} else if (!strcasecmp(argv[j],"--self-recall")) {
|
||||
self_recall = 1;
|
||||
} else if (moreargs >= 1 && !strcasecmp(argv[j],"--recall")) {
|
||||
recall_ef = atoi(argv[j+1]);
|
||||
j++;
|
||||
} else if (moreargs >= 1 && !strcasecmp(argv[j],"--threads")) {
|
||||
numthreads = atoi(argv[j+1]);
|
||||
j++;
|
||||
} else if (moreargs >= 1 && !strcasecmp(argv[j],"--numele")) {
|
||||
numele = strtoll(argv[j+1],NULL,0);
|
||||
j++;
|
||||
if (numele < 1) numele = 1;
|
||||
} else if (moreargs >= 1 && !strcasecmp(argv[j],"--m")) {
|
||||
m_param = atoi(argv[j+1]);
|
||||
j++;
|
||||
} else if (!strcasecmp(argv[j],"--help")) {
|
||||
printf("%s [--quant] [--bin] [--thread <count>] [--numele <count>] [--m <count>] [--mass-del] [--self-recall] [--recall <ef>]\n", argv[0]);
|
||||
exit(0);
|
||||
} else {
|
||||
printf("Unrecognized option or wrong number of arguments: %s\n", argv[j]);
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
if (quantization == HNSW_QUANT_NONE) {
|
||||
printf("You can enable quantization with --quant\n");
|
||||
}
|
||||
|
||||
if (numthreads > 0) {
|
||||
w2v_multi_thread(m_param, numthreads, quantization, numele);
|
||||
} else {
|
||||
printf("Single thread execution. Use --threads 4 for concurrent API\n");
|
||||
w2v_single_thread(m_param, quantization, numele, massdel, self_recall, recall_ef);
|
||||
}
|
||||
}
|
||||
+376
@@ -0,0 +1,376 @@
|
||||
include redis.conf
|
||||
|
||||
loadmodule ./modules/redisbloom/redisbloom.so
|
||||
loadmodule ./modules/redisearch/redisearch.so
|
||||
loadmodule ./modules/redisjson/rejson.so
|
||||
loadmodule ./modules/redistimeseries/redistimeseries.so
|
||||
|
||||
############################## QUERY ENGINE CONFIG ############################
|
||||
|
||||
# Keep numeric ranges in numeric tree parent nodes of leafs for `x` generations.
|
||||
# numeric, valid range: [0, 2], default: 0
|
||||
#
|
||||
# search-_numeric-ranges-parents 0
|
||||
|
||||
# The number of iterations to run while performing background indexing
|
||||
# before we call usleep(1) (sleep for 1 micro-second) and make sure that we
|
||||
# allow redis to process other commands.
|
||||
# numeric, valid range: [1, UINT32_MAX], default: 100
|
||||
#
|
||||
# search-bg-index-sleep-gap 100
|
||||
|
||||
# The default dialect used in search queries.
|
||||
# numeric, valid range: [1, 4], default: 1
|
||||
#
|
||||
# search-default-dialect 1
|
||||
|
||||
# the fork gc will only start to clean when the number of not cleaned document
|
||||
# will exceed this threshold.
|
||||
# numeric, valid range: [1, LLONG_MAX], default: 100
|
||||
#
|
||||
# search-fork-gc-clean-threshold 100
|
||||
|
||||
# interval (in seconds) in which to retry running the forkgc after failure.
|
||||
# numeric, valid range: [1, LLONG_MAX], default: 5
|
||||
#
|
||||
# search-fork-gc-retry-interval 5
|
||||
|
||||
# interval (in seconds) in which to run the fork gc (relevant only when fork
|
||||
# gc is used).
|
||||
# numeric, valid range: [1, LLONG_MAX], default: 30
|
||||
#
|
||||
# search-fork-gc-run-interval 30
|
||||
|
||||
# the amount of seconds for the fork GC to sleep before exiting.
|
||||
# numeric, valid range: [0, LLONG_MAX], default: 0
|
||||
#
|
||||
# search-fork-gc-sleep-before-exit 0
|
||||
|
||||
# Scan this many documents at a time during every GC iteration.
|
||||
# numeric, valid range: [1, LLONG_MAX], default: 100
|
||||
#
|
||||
# search-gc-scan-size 100
|
||||
|
||||
# Max number of cursors for a given index that can be opened inside of a shard.
|
||||
# numeric, valid range: [0, LLONG_MAX], default: 128
|
||||
#
|
||||
# search-index-cursor-limit 128
|
||||
|
||||
# Maximum number of results from ft.aggregate command.
|
||||
# numeric, valid range: [0, (1ULL << 31)], default: 1ULL << 31
|
||||
#
|
||||
# search-max-aggregate-results 2147483648
|
||||
|
||||
# Maximum prefix expansions to be used in a query.
|
||||
# numeric, valid range: [1, LLONG_MAX], default: 200
|
||||
#
|
||||
# search-max-prefix-expansions 200
|
||||
|
||||
# Maximum runtime document table size (for this process).
|
||||
# numeric, valid range: [1, 100000000], default: 1000000
|
||||
#
|
||||
# search-max-doctablesize 1000000
|
||||
|
||||
# max idle time allowed to be set for cursor, setting it high might cause
|
||||
# high memory consumption.
|
||||
# numeric, valid range: [1, LLONG_MAX], default: 300000
|
||||
#
|
||||
# search-cursor-max-idle 300000
|
||||
|
||||
# Maximum number of results from ft.search command.
|
||||
# numeric, valid range: [0, 1ULL << 31], default: 1000000
|
||||
#
|
||||
# search-max-search-results 1000000
|
||||
|
||||
# Number of worker threads to use for background tasks when the server is
|
||||
# in an operation event.
|
||||
# numeric, valid range: [1, 16], default: 4
|
||||
#
|
||||
# search-min-operation-workers 4
|
||||
|
||||
# Minimum length of term to be considered for phonetic matching.
|
||||
# numeric, valid range: [1, LLONG_MAX], default: 3
|
||||
#
|
||||
# search-min-phonetic-term-len 3
|
||||
|
||||
# the minimum prefix for expansions (`*`).
|
||||
# numeric, valid range: [1, LLONG_MAX], default: 2
|
||||
#
|
||||
# search-min-prefix 2
|
||||
|
||||
# the minimum word length to stem.
|
||||
# numeric, valid range: [2, UINT32_MAX], default: 4
|
||||
#
|
||||
# search-min-stem-len 4
|
||||
|
||||
# Delta used to increase positional offsets between array
|
||||
# slots for multi text values.
|
||||
# Can control the level of separation between phrases in different
|
||||
# array slots (related to the SLOP parameter of ft.search command)"
|
||||
# numeric, valid range: [1, UINT32_MAX], default: 100
|
||||
#
|
||||
# search-multi-text-slop 100
|
||||
|
||||
# Used for setting the buffer limit threshold for vector similarity tiered
|
||||
# HNSW index, so that if we are using WORKERS for indexing, and the
|
||||
# number of vectors waiting in the buffer to be indexed exceeds this limit,
|
||||
# we insert new vectors directly into HNSW.
|
||||
# numeric, valid range: [0, LLONG_MAX], default: 1024
|
||||
#
|
||||
# search-tiered-hnsw-buffer-limit 1024
|
||||
|
||||
# Query timeout.
|
||||
# numeric, valid range: [1, LLONG_MAX], default: 500
|
||||
#
|
||||
# search-timeout 500
|
||||
|
||||
# minimum number of iterators in a union from which the iterator will
|
||||
# will switch to heap-based implementation.
|
||||
# numeric, valid range: [1, LLONG_MAX], default: 20
|
||||
# switch to heap based implementation.
|
||||
#
|
||||
# search-union-iterator-heap 20
|
||||
|
||||
# The maximum memory resize for vector similarity indexes (in bytes).
|
||||
# numeric, valid range: [0, UINT32_MAX], default: 0
|
||||
#
|
||||
# search-vss-max-resize 0
|
||||
|
||||
# Number of worker threads to use for query processing and background tasks.
|
||||
# numeric, valid range: [0, 16], default: 0
|
||||
# This configuration also affects the number of connections per shard.
|
||||
#
|
||||
# search-workers 0
|
||||
|
||||
# The number of high priority tasks to be executed at any given time by the
|
||||
# worker thread pool, before executing low priority tasks. After this number
|
||||
# of high priority tasks are being executed, the worker thread pool will
|
||||
# execute high and low priority tasks alternately.
|
||||
# numeric, valid range: [0, LLONG_MAX], default: 1
|
||||
#
|
||||
# search-workers-priority-bias-threshold 1
|
||||
|
||||
# Load extension scoring/expansion module. Immutable.
|
||||
# string, default: ""
|
||||
#
|
||||
# search-ext-load ""
|
||||
|
||||
# Path to Chinese dictionary configuration file (for Chinese tokenization). Immutable.
|
||||
# string, default: ""
|
||||
#
|
||||
# search-friso-ini ""
|
||||
|
||||
# Action to perform when search timeout is exceeded (choose RETURN or FAIL).
|
||||
# enum, valid values: ["return", "fail"], default: "fail"
|
||||
#
|
||||
# search-on-timeout fail
|
||||
|
||||
# Determine whether some index resources are free on a second thread.
|
||||
# bool, default: yes
|
||||
#
|
||||
# search-_free-resource-on-thread yes
|
||||
|
||||
# Enable legacy compression of double to float.
|
||||
# bool, default: no
|
||||
#
|
||||
# search-_numeric-compress no
|
||||
|
||||
# Disable print of time for ft.profile. For testing only.
|
||||
# bool, default: yes
|
||||
#
|
||||
# search-_print-profile-clock yes
|
||||
|
||||
# Intersection iterator orders the children iterators by their relative estimated
|
||||
# number of results in ascending order, so that if we see first iterators with
|
||||
# a lower count of results we will skip a larger number of results, which
|
||||
# translates into faster iteration. If this flag is set, we use this
|
||||
# optimization in a way where union iterators are being factorize by the number
|
||||
# of their own children, so that we sort by the number of children times the
|
||||
# overall estimated number of results instead.
|
||||
# bool, default: no
|
||||
#
|
||||
# search-_prioritize-intersect-union-children no
|
||||
|
||||
# Set to run without memory pools.
|
||||
# bool, default: no
|
||||
#
|
||||
# search-no-mem-pools no
|
||||
|
||||
# Disable garbage collection (for this process).
|
||||
# bool, default: no
|
||||
#
|
||||
# search-no-gc no
|
||||
|
||||
# Enable commands filter which optimize indexing on partial hash updates.
|
||||
# bool, default: no
|
||||
#
|
||||
# search-partial-indexed-docs no
|
||||
|
||||
# Disable compression for DocID inverted index. Boost CPU performance.
|
||||
# bool, default: no
|
||||
#
|
||||
# search-raw-docid-encoding no
|
||||
|
||||
# Number of search threads in the coordinator thread pool.
|
||||
# numeric, valid range: [1, LLONG_MAX], default: 20
|
||||
#
|
||||
# search-threads 20
|
||||
|
||||
# Timeout for topology validation (in milliseconds). After this timeout,
|
||||
# any pending requests will be processed, even if the topology is not fully connected.
|
||||
# numeric, valid range: [0, LLONG_MAX], default: 30000
|
||||
#
|
||||
# search-topology-validation-timeout 30000
|
||||
|
||||
|
||||
############################## TIME SERIES CONFIG #############################
|
||||
|
||||
# The maximal number of per-shard threads for cross-key queries when using cluster mode
|
||||
# (TS.MRANGE, TS.MREVRANGE, TS.MGET, and TS.QUERYINDEX).
|
||||
# Note: increasing this value may either increase or decrease the performance.
|
||||
# integer, valid range: [1..16], default: 3
|
||||
# This is a load-time configuration parameter.
|
||||
#
|
||||
# ts-num-threads 3
|
||||
|
||||
|
||||
# Default compaction rules for newly created key with TS.ADD, TS.INCRBY, and TS.DECRBY.
|
||||
# Has no effect on keys created with TS.CREATE.
|
||||
# This default value is applied to each new time series upon its creation.
|
||||
# string, see documentation for rules format, default: no compaction rules
|
||||
#
|
||||
# ts-compaction-policy ""
|
||||
|
||||
# Default chunk encoding for automatically-created compacted time series.
|
||||
# This default value is applied to each new compacted time series automatically
|
||||
# created when ts-compaction-policy is specified.
|
||||
# valid values: COMPRESSED, UNCOMPRESSED, default: COMPRESSED
|
||||
#
|
||||
# ts-encoding COMPRESSED
|
||||
|
||||
|
||||
# Default retention period, in milliseconds. 0 means no expiration.
|
||||
# This default value is applied to each new time series upon its creation.
|
||||
# If ts-compaction-policy is specified - it is overridden for created
|
||||
# compactions as specified in ts-compaction-policy.
|
||||
# integer, valid range: [0 .. LLONG_MAX], default: 0
|
||||
#
|
||||
# ts-retention-policy 0
|
||||
|
||||
# Default policy for handling insertion (TS.ADD and TS.MADD) of multiple
|
||||
# samples with identical timestamps.
|
||||
# This default value is applied to each new time series upon its creation.
|
||||
# string, valid values: BLOCK, FIRST, LAST, MIN, MAX, SUM, default: BLOCK
|
||||
#
|
||||
# ts-duplicate-policy BLOCK
|
||||
|
||||
# Default initial allocation size, in bytes, for the data part of each new chunk
|
||||
# This default value is applied to each new time series upon its creation.
|
||||
# integer, valid range: [48 .. 1048576]; must be a multiple of 8, default: 4096
|
||||
#
|
||||
# ts-chunk-size-bytes 4096
|
||||
|
||||
# Default values for newly created time series.
|
||||
# Many sensors report data periodically. Often, the difference between the measured
|
||||
# value and the previous measured value is negligible and related to random noise
|
||||
# or to measurement accuracy limitations. In such situations it may be preferable
|
||||
# not to add the new measurement to the time series.
|
||||
# A new sample is considered a duplicate and is ignored if the following conditions are met:
|
||||
# - The time series is not a compaction;
|
||||
# - The time series' DUPLICATE_POLICY IS LAST;
|
||||
# - The sample is added in-order (timestamp >= max_timestamp);
|
||||
# - The difference of the current timestamp from the previous timestamp
|
||||
# (timestamp - max_timestamp) is less than or equal to ts-ignore-max-time-diff
|
||||
# - The absolute value difference of the current value from the value at the previous maximum timestamp
|
||||
# (abs(value - value_at_max_timestamp) is less than or equal to ts-ignore-max-val-diff.
|
||||
# where max_timestamp is the timestamp of the sample with the largest timestamp in the time series,
|
||||
# and value_at_max_timestamp is the value at max_timestamp.
|
||||
# ts-ignore-max-time-diff: integer, valid range: [0 .. LLONG_MAX], default: 0
|
||||
# ts-ignore-max-val-diff: double, Valid range: [0 .. DBL_MAX], default: 0
|
||||
#
|
||||
# ts-ignore-max-time-diff 0
|
||||
# ts-ignore-max-val-diff 0
|
||||
|
||||
|
||||
########################### BLOOM FILTERS CONFIG ##############################
|
||||
|
||||
# Defaults values for new Bloom filters created with BF.ADD, BF.MADD, BF.INSERT, and BF.RESERVE
|
||||
# These defaults are applied to each new Bloom filter upon its creation.
|
||||
|
||||
# Error ratio
|
||||
# The desired probability for false positives.
|
||||
# For a false positive rate of 0.1% (1 in 1000) - the value should be 0.001.
|
||||
# double, Valid range: (0 .. 1), value greater than 0.25 is treated as 0.25, default: 0.01
|
||||
#
|
||||
# bf-error-rate 0.01
|
||||
|
||||
# Initial capacity
|
||||
# The number of entries intended to be added to the filter.
|
||||
# integer, valid range: [1 .. 1GB], default: 100
|
||||
#
|
||||
# bf-initial-size 100
|
||||
|
||||
# Expansion factor
|
||||
# When capacity is reached, an additional sub-filter is created.
|
||||
# The size of the new sub-filter is the size of the last sub-filter multiplied
|
||||
# by expansion.
|
||||
# integer, [0 .. 32768]. 0 is equivalent to NONSCALING. default: 2
|
||||
#
|
||||
# bf-expansion-factor 2
|
||||
|
||||
|
||||
########################### CUCKOO FILTERS CONFIG #############################
|
||||
|
||||
# Defaults values for new Cuckoo filters created with
|
||||
# CF.ADD, CF.ADDNX, CF.INSERT, CF.INSERTNX, and CF.RESERVE
|
||||
# These defaults are applied to each new Cuckoo filter upon its creation.
|
||||
|
||||
# Initial capacity
|
||||
# A filter will likely not fill up to 100% of its capacity.
|
||||
# Make sure to reserve extra capacity if you want to avoid expansions.
|
||||
# value is rounded to the next 2^n integer.
|
||||
# integer, valid range: [2*cf-bucket-size .. 1GB], default: 1024
|
||||
#
|
||||
# cf-initial-size 1024
|
||||
|
||||
# Number of items in each bucket
|
||||
# The minimal false positive rate is 2/255 ~ 0.78% when bucket size of 1 is used.
|
||||
# Larger buckets increase the error rate linearly, but improve the fill rate.
|
||||
# integer, valid range: [1 .. 255], default: 2
|
||||
#
|
||||
# cf-bucket-size 2
|
||||
|
||||
# Maximum iterations
|
||||
# Number of attempts to swap items between buckets before declaring filter
|
||||
# as full and creating an additional filter.
|
||||
# A lower value improves performance. A higher value improves fill rate.
|
||||
# integer, Valid range: [1 .. 65535], default: 20
|
||||
#
|
||||
# cf-max-iterations 20
|
||||
|
||||
# Expansion factor
|
||||
# When a new filter is created, its size is the size of the current filter
|
||||
# multiplied by this factor.
|
||||
# integer, Valid range: [0 .. 32768], 0 is equivalent to NONSCALING, default: 1
|
||||
#
|
||||
# cf-expansion-factor 1
|
||||
|
||||
# Maximum expansions
|
||||
# integer, Valid range: [1 .. 65536], default: 32
|
||||
#
|
||||
# cf-max-expansions 32
|
||||
|
||||
|
||||
################################## SECURITY ###################################
|
||||
#
|
||||
# The following is a list of command categories and their meanings:
|
||||
#
|
||||
# * search - Query engine related.
|
||||
# * json - Data type: JSON related.
|
||||
# * timeseries - Data type: time series related.
|
||||
# * bloom - Data type: Bloom filter related.
|
||||
# * cuckoo - Data type: cuckoo filter related.
|
||||
# * topk - Data type: top-k related.
|
||||
# * cms - Data type: count-min sketch related.
|
||||
# * tdigest - Data type: t-digest related.
|
||||
|
||||
+49
-27
@@ -668,7 +668,7 @@ repl-diskless-sync-max-replicas 0
|
||||
repl-diskless-load disabled
|
||||
|
||||
# Master send PINGs to its replicas in a predefined interval. It's possible to
|
||||
# change this interval with the repl_ping_replica_period option. The default
|
||||
# change this interval with the repl-ping-replica-period option. The default
|
||||
# value is 10 seconds.
|
||||
#
|
||||
# repl-ping-replica-period 10
|
||||
@@ -727,6 +727,24 @@ repl-disable-tcp-nodelay no
|
||||
#
|
||||
# repl-backlog-ttl 3600
|
||||
|
||||
# During a fullsync, the master may decide to send both the RDB file and the
|
||||
# replication stream to the replica in parallel. This approach shifts the
|
||||
# responsibility of buffering the replication stream to the replica during the
|
||||
# fullsync process. The replica accumulates the replication stream data until
|
||||
# the RDB file is fully loaded. Once the RDB delivery is completed and
|
||||
# successfully loaded, the replica begins processing and applying the
|
||||
# accumulated replication data to the db. The configuration below controls how
|
||||
# much replication data the replica can accumulate during a fullsync.
|
||||
#
|
||||
# When the replica reaches this limit, it will stop accumulating further data.
|
||||
# At this point, additional data accumulation may occur on the master side
|
||||
# depending on the 'client-output-buffer-limit <replica>' config of master.
|
||||
#
|
||||
# A value of 0 means replica inherits hard limit of
|
||||
# 'client-output-buffer-limit <replica>' config to limit accumulation size.
|
||||
#
|
||||
# replica-full-sync-buffer-limit 0
|
||||
|
||||
# The replica priority is an integer number published by Redis in the INFO
|
||||
# output. It is used by Redis Sentinel in order to select a replica to promote
|
||||
# into a master if the master is no longer working correctly.
|
||||
@@ -838,7 +856,7 @@ replica-priority 100
|
||||
# this is used in order to send invalidation messages to clients. Please
|
||||
# check this page to understand more about the feature:
|
||||
#
|
||||
# https://redis.io/topics/client-side-caching
|
||||
# https://redis.io/docs/latest/develop/use/client-side-caching/
|
||||
#
|
||||
# When tracking is enabled for a client, all the read only queries are assumed
|
||||
# to be cached: this will force Redis to store information in the invalidation
|
||||
@@ -1016,7 +1034,7 @@ replica-priority 100
|
||||
# * stream - Data type: streams related.
|
||||
#
|
||||
# For more information about ACL configuration please refer to
|
||||
# the Redis web site at https://redis.io/topics/acl
|
||||
# the Redis web site at https://redis.io/docs/latest/operate/oss_and_stack/management/security/acl/
|
||||
|
||||
# ACL LOG
|
||||
#
|
||||
@@ -1291,38 +1309,27 @@ lazyfree-lazy-user-flush no
|
||||
# in different I/O threads. Since especially writing is so slow, normally
|
||||
# Redis users use pipelining in order to speed up the Redis performances per
|
||||
# core, and spawn multiple instances in order to scale more. Using I/O
|
||||
# threads it is possible to easily speedup two times Redis without resorting
|
||||
# threads it is possible to easily speedup several times Redis without resorting
|
||||
# to pipelining nor sharding of the instance.
|
||||
#
|
||||
# By default threading is disabled, we suggest enabling it only in machines
|
||||
# that have at least 4 or more cores, leaving at least one spare core.
|
||||
# Using more than 8 threads is unlikely to help much. We also recommend using
|
||||
# threaded I/O only if you actually have performance problems, with Redis
|
||||
# instances being able to use a quite big percentage of CPU time, otherwise
|
||||
# there is no point in using this feature.
|
||||
# We also recommend using threaded I/O only if you actually have performance
|
||||
# problems, with Redis instances being able to use a quite big percentage of
|
||||
# CPU time, otherwise there is no point in using this feature.
|
||||
#
|
||||
# So for instance if you have a four cores boxes, try to use 2 or 3 I/O
|
||||
# threads, if you have a 8 cores, try to use 6 threads. In order to
|
||||
# So for instance if you have a four cores boxes, try to use 3 I/O
|
||||
# threads, if you have a 8 cores, try to use 7 threads. In order to
|
||||
# enable I/O threads use the following configuration directive:
|
||||
#
|
||||
# io-threads 4
|
||||
#
|
||||
# Setting io-threads to 1 will just use the main thread as usual.
|
||||
# When I/O threads are enabled, we only use threads for writes, that is
|
||||
# to thread the write(2) syscall and transfer the client buffers to the
|
||||
# socket. However it is also possible to enable threading of reads and
|
||||
# protocol parsing using the following configuration directive, by setting
|
||||
# it to yes:
|
||||
# When I/O threads are enabled, we not only use threads for writes, that
|
||||
# is to thread the write(2) syscall and transfer the client buffers to the
|
||||
# socket, but also use threads for reads and protocol parsing.
|
||||
#
|
||||
# io-threads-do-reads no
|
||||
#
|
||||
# Usually threading reads doesn't help much.
|
||||
#
|
||||
# NOTE 1: This configuration directive cannot be changed at runtime via
|
||||
# CONFIG SET. Also, this feature currently does not work when SSL is
|
||||
# enabled.
|
||||
#
|
||||
# NOTE 2: If you want to test the Redis speedup using redis-benchmark, make
|
||||
# NOTE: If you want to test the Redis speedup using redis-benchmark, make
|
||||
# sure you also run the benchmark itself in threaded mode, using the
|
||||
# --threads option to match the number of Redis threads, otherwise you'll not
|
||||
# be able to notice the improvements.
|
||||
@@ -1362,7 +1369,7 @@ oom-score-adj-values 0 200 800
|
||||
#################### KERNEL transparent hugepage CONTROL ######################
|
||||
|
||||
# Usually the kernel Transparent Huge Pages control is set to "madvise" or
|
||||
# or "never" by default (/sys/kernel/mm/transparent_hugepage/enabled), in which
|
||||
# "never" by default (/sys/kernel/mm/transparent_hugepage/enabled), in which
|
||||
# case this config has no effect. On systems in which it is set to "always",
|
||||
# redis will attempt to disable it specifically for the redis process in order
|
||||
# to avoid latency problems specifically with fork(2) and CoW.
|
||||
@@ -1393,7 +1400,7 @@ disable-thp yes
|
||||
# restarting the server can lead to data loss. A conversion needs to be done
|
||||
# by setting it via CONFIG command on a live server first.
|
||||
#
|
||||
# Please check https://redis.io/topics/persistence for more information.
|
||||
# Please check https://redis.io/docs/latest/operate/oss_and_stack/management/persistence/ for more information.
|
||||
|
||||
appendonly no
|
||||
|
||||
@@ -1776,6 +1783,21 @@ aof-timestamp-enabled no
|
||||
#
|
||||
# cluster-preferred-endpoint-type ip
|
||||
|
||||
# This configuration defines the sampling ratio (0-100) for checking command
|
||||
# compatibility in cluster mode. When a command is executed, it is sampled at
|
||||
# the specified ratio to determine if it complies with Redis cluster constraints,
|
||||
# such as cross-slot restrictions.
|
||||
#
|
||||
# - A value of 0 means no commands are sampled for compatibility checks.
|
||||
# - A value of 100 means all commands are checked.
|
||||
# - Intermediate values (e.g., 10) mean that approximately 10% of the commands
|
||||
# are randomly selected for compatibility verification.
|
||||
#
|
||||
# Higher sampling ratios may introduce additional performance overhead, especially
|
||||
# under high QPS. The default value is 0 (no sampling).
|
||||
#
|
||||
# cluster-compatibility-sample-ratio 0
|
||||
|
||||
# In order to setup your cluster make sure to read the documentation
|
||||
# available at https://redis.io web site.
|
||||
|
||||
@@ -1880,7 +1902,7 @@ latency-monitor-threshold 0
|
||||
############################# EVENT NOTIFICATION ##############################
|
||||
|
||||
# Redis can notify Pub/Sub clients about events happening in the key space.
|
||||
# This feature is documented at https://redis.io/topics/notifications
|
||||
# This feature is documented at https://redis.io/docs/latest/develop/use/keyspace-notifications/
|
||||
#
|
||||
# For instance if keyspace events notification is enabled, and a client
|
||||
# performs a DEL operation on key "foo" stored in the Database 0, two
|
||||
|
||||
@@ -56,4 +56,5 @@ $TCLSH tests/test_helper.tcl \
|
||||
--single unit/moduleapi/moduleauth \
|
||||
--single unit/moduleapi/rdbloadsave \
|
||||
--single unit/moduleapi/crash \
|
||||
--single unit/moduleapi/internalsecret \
|
||||
"${@}"
|
||||
|
||||
+3
-3
@@ -133,7 +133,7 @@ sentinel monitor mymaster 127.0.0.1 6379 2
|
||||
sentinel down-after-milliseconds mymaster 30000
|
||||
|
||||
# IMPORTANT NOTE: starting with Redis 6.2 ACL capability is supported for
|
||||
# Sentinel mode, please refer to the Redis website https://redis.io/topics/acl
|
||||
# Sentinel mode, please refer to the Redis website https://redis.io/docs/latest/operate/oss_and_stack/management/security/acl/
|
||||
# for more details.
|
||||
|
||||
# Sentinel's ACL users are defined in the following format:
|
||||
@@ -145,7 +145,7 @@ sentinel down-after-milliseconds mymaster 30000
|
||||
# user worker +@admin +@connection ~* on >ffa9203c493aa99
|
||||
#
|
||||
# For more information about ACL configuration please refer to the Redis
|
||||
# website at https://redis.io/topics/acl and redis server configuration
|
||||
# website at https://redis.io/docs/latest/operate/oss_and_stack/management/security/acl/ and redis server configuration
|
||||
# template redis.conf.
|
||||
|
||||
# ACL LOG
|
||||
@@ -174,7 +174,7 @@ acllog-max-len 128
|
||||
# so Sentinel will try to authenticate with the same password to all the
|
||||
# other Sentinels. So you need to configure all your Sentinels in a given
|
||||
# group with the same "requirepass" password. Check the following documentation
|
||||
# for more info: https://redis.io/topics/sentinel
|
||||
# for more info: https://redis.io/docs/latest/operate/oss_and_stack/management/sentinel/
|
||||
#
|
||||
# IMPORTANT NOTE: starting with Redis 6.2 "requirepass" is a compatibility
|
||||
# layer on top of the ACL system. The option effect will be just setting
|
||||
|
||||
+21
-12
@@ -34,7 +34,7 @@ endif
|
||||
ifneq ($(OPTIMIZATION),-O0)
|
||||
OPTIMIZATION+=-fno-omit-frame-pointer
|
||||
endif
|
||||
DEPENDENCY_TARGETS=hiredis linenoise lua hdr_histogram fpconv
|
||||
DEPENDENCY_TARGETS=hiredis linenoise lua hdr_histogram fpconv fast_float
|
||||
NODEPS:=clean distclean
|
||||
|
||||
# Default settings
|
||||
@@ -52,6 +52,7 @@ endif
|
||||
WARN=-Wall -W -Wno-missing-field-initializers -Werror=deprecated-declarations -Wstrict-prototypes
|
||||
OPT=$(OPTIMIZATION)
|
||||
|
||||
SKIP_VEC_SETS?=no
|
||||
# Detect if the compiler supports C11 _Atomic.
|
||||
# NUMBER_SIGN_CHAR is a workaround to support both GNU Make 4.3 and older versions.
|
||||
NUMBER_SIGN_CHAR := \#
|
||||
@@ -61,6 +62,7 @@ C11_ATOMIC := $(shell sh -c 'echo "$(NUMBER_SIGN_CHAR)include <stdatomic.h>" > f
|
||||
ifeq ($(C11_ATOMIC),yes)
|
||||
STD+=-std=gnu11
|
||||
else
|
||||
SKIP_VEC_SETS=yes
|
||||
STD+=-std=c99
|
||||
endif
|
||||
|
||||
@@ -127,7 +129,7 @@ endif
|
||||
|
||||
FINAL_CFLAGS=$(STD) $(WARN) $(OPT) $(DEBUG) $(CFLAGS) $(REDIS_CFLAGS)
|
||||
FINAL_LDFLAGS=$(LDFLAGS) $(OPT) $(REDIS_LDFLAGS) $(DEBUG)
|
||||
FINAL_LIBS=-lm
|
||||
FINAL_LIBS=-lm -lstdc++
|
||||
DEBUG=-g -ggdb
|
||||
|
||||
# Linux ARM32 needs -latomic at linking time
|
||||
@@ -235,7 +237,7 @@ ifdef OPENSSL_PREFIX
|
||||
endif
|
||||
|
||||
# Include paths to dependencies
|
||||
FINAL_CFLAGS+= -I../deps/hiredis -I../deps/linenoise -I../deps/lua/src -I../deps/hdr_histogram -I../deps/fpconv
|
||||
FINAL_CFLAGS+= -I../deps/hiredis -I../deps/linenoise -I../deps/lua/src -I../deps/hdr_histogram -I../deps/fpconv -I../deps/fast_float
|
||||
|
||||
# Determine systemd support and/or build preference (defaulting to auto-detection)
|
||||
BUILD_WITH_SYSTEMD=no
|
||||
@@ -315,6 +317,12 @@ ifeq ($(BUILD_TLS),module)
|
||||
TLS_MODULE_CFLAGS+=-DUSE_OPENSSL=$(BUILD_MODULE) $(OPENSSL_CFLAGS) -DBUILD_TLS_MODULE=$(BUILD_MODULE)
|
||||
endif
|
||||
|
||||
ifneq ($(SKIP_VEC_SETS),yes)
|
||||
vpath %.c ../modules/vector-sets
|
||||
REDIS_VEC_SETS_OBJ=hnsw.o cJSON.o vset.o
|
||||
FINAL_CFLAGS+=-DINCLUDE_VEC_SETS=1
|
||||
endif
|
||||
|
||||
ifndef V
|
||||
define MAKE_INSTALL
|
||||
@printf ' %b %b\n' $(LINKCOLOR)INSTALL$(ENDCOLOR) $(BINCOLOR)$(1)$(ENDCOLOR) 1>&2
|
||||
@@ -354,14 +362,14 @@ endif
|
||||
|
||||
REDIS_SERVER_NAME=redis-server$(PROG_SUFFIX)
|
||||
REDIS_SENTINEL_NAME=redis-sentinel$(PROG_SUFFIX)
|
||||
REDIS_SERVER_OBJ=threads_mngr.o adlist.o quicklist.o ae.o anet.o dict.o ebuckets.o mstr.o kvstore.o server.o sds.o zmalloc.o lzf_c.o lzf_d.o pqsort.o zipmap.o sha1.o ziplist.o release.o networking.o util.o object.o db.o replication.o rdb.o t_string.o t_list.o t_set.o t_zset.o t_hash.o config.o aof.o pubsub.o multi.o debug.o sort.o intset.o syncio.o cluster.o cluster_legacy.o crc16.o endianconv.o slowlog.o eval.o bio.o rio.o rand.o memtest.o syscheck.o crcspeed.o crc64.o bitops.o sentinel.o notify.o setproctitle.o blocked.o hyperloglog.o latency.o sparkline.o redis-check-rdb.o redis-check-aof.o geo.o lazyfree.o module.o evict.o expire.o geohash.o geohash_helper.o childinfo.o defrag.o siphash.o rax.o t_stream.o listpack.o localtime.o lolwut.o lolwut5.o lolwut6.o acl.o tracking.o socket.o tls.o sha256.o timeout.o setcpuaffinity.o monotonic.o mt19937-64.o resp_parser.o call_reply.o script_lua.o script.o functions.o function_lua.o commands.o strl.o connection.o unix.o logreqres.o
|
||||
REDIS_SERVER_OBJ=threads_mngr.o adlist.o quicklist.o ae.o anet.o dict.o ebuckets.o eventnotifier.o iothread.o mstr.o kvstore.o server.o sds.o zmalloc.o lzf_c.o lzf_d.o pqsort.o zipmap.o sha1.o ziplist.o release.o networking.o util.o object.o db.o replication.o rdb.o t_string.o t_list.o t_set.o t_zset.o t_hash.o config.o aof.o pubsub.o multi.o debug.o sort.o intset.o syncio.o cluster.o cluster_legacy.o crc16.o endianconv.o slowlog.o eval.o bio.o rio.o rand.o memtest.o syscheck.o crcspeed.o crccombine.o crc64.o bitops.o sentinel.o notify.o setproctitle.o blocked.o hyperloglog.o latency.o sparkline.o redis-check-rdb.o redis-check-aof.o geo.o lazyfree.o module.o evict.o expire.o geohash.o geohash_helper.o childinfo.o defrag.o siphash.o rax.o t_stream.o listpack.o localtime.o lolwut.o lolwut5.o lolwut6.o acl.o tracking.o socket.o tls.o sha256.o timeout.o setcpuaffinity.o monotonic.o mt19937-64.o resp_parser.o call_reply.o script_lua.o script.o functions.o function_lua.o commands.o strl.o connection.o unix.o logreqres.o
|
||||
REDIS_CLI_NAME=redis-cli$(PROG_SUFFIX)
|
||||
REDIS_CLI_OBJ=anet.o adlist.o dict.o redis-cli.o zmalloc.o release.o ae.o redisassert.o crcspeed.o crc64.o siphash.o crc16.o monotonic.o cli_common.o mt19937-64.o strl.o cli_commands.o
|
||||
REDIS_CLI_OBJ=anet.o adlist.o dict.o redis-cli.o zmalloc.o release.o ae.o redisassert.o crcspeed.o crccombine.o crc64.o siphash.o crc16.o monotonic.o cli_common.o mt19937-64.o strl.o cli_commands.o
|
||||
REDIS_BENCHMARK_NAME=redis-benchmark$(PROG_SUFFIX)
|
||||
REDIS_BENCHMARK_OBJ=ae.o anet.o redis-benchmark.o adlist.o dict.o zmalloc.o redisassert.o release.o crcspeed.o crc64.o siphash.o crc16.o monotonic.o cli_common.o mt19937-64.o strl.o
|
||||
REDIS_BENCHMARK_OBJ=ae.o anet.o redis-benchmark.o adlist.o dict.o zmalloc.o redisassert.o release.o crcspeed.o crccombine.o crc64.o siphash.o crc16.o monotonic.o cli_common.o mt19937-64.o strl.o
|
||||
REDIS_CHECK_RDB_NAME=redis-check-rdb$(PROG_SUFFIX)
|
||||
REDIS_CHECK_AOF_NAME=redis-check-aof$(PROG_SUFFIX)
|
||||
ALL_SOURCES=$(sort $(patsubst %.o,%.c,$(REDIS_SERVER_OBJ) $(REDIS_CLI_OBJ) $(REDIS_BENCHMARK_OBJ)))
|
||||
ALL_SOURCES=$(sort $(patsubst %.o,%.c,$(REDIS_SERVER_OBJ) $(REDIS_VEC_SETS_OBJ) $(REDIS_CLI_OBJ) $(REDIS_BENCHMARK_OBJ)))
|
||||
|
||||
all: $(REDIS_SERVER_NAME) $(REDIS_SENTINEL_NAME) $(REDIS_CLI_NAME) $(REDIS_BENCHMARK_NAME) $(REDIS_CHECK_RDB_NAME) $(REDIS_CHECK_AOF_NAME) $(TLS_MODULE)
|
||||
@echo ""
|
||||
@@ -408,8 +416,8 @@ ifneq ($(strip $(PREV_FINAL_LDFLAGS)), $(strip $(FINAL_LDFLAGS)))
|
||||
endif
|
||||
|
||||
# redis-server
|
||||
$(REDIS_SERVER_NAME): $(REDIS_SERVER_OBJ)
|
||||
$(REDIS_LD) -o $@ $^ ../deps/hiredis/libhiredis.a ../deps/lua/src/liblua.a ../deps/hdr_histogram/libhdrhistogram.a ../deps/fpconv/libfpconv.a $(FINAL_LIBS)
|
||||
$(REDIS_SERVER_NAME): $(REDIS_SERVER_OBJ) $(REDIS_VEC_SETS_OBJ)
|
||||
$(REDIS_LD) -o $@ $^ ../deps/hiredis/libhiredis.a ../deps/lua/src/liblua.a ../deps/hdr_histogram/libhdrhistogram.a ../deps/fpconv/libfpconv.a ../deps/fast_float/libfast_float.a $(FINAL_LIBS)
|
||||
|
||||
# redis-sentinel
|
||||
$(REDIS_SENTINEL_NAME): $(REDIS_SERVER_NAME)
|
||||
@@ -435,7 +443,7 @@ $(REDIS_CLI_NAME): $(REDIS_CLI_OBJ)
|
||||
$(REDIS_BENCHMARK_NAME): $(REDIS_BENCHMARK_OBJ)
|
||||
$(REDIS_LD) -o $@ $^ ../deps/hiredis/libhiredis.a ../deps/hdr_histogram/libhdrhistogram.a $(FINAL_LIBS) $(TLS_CLIENT_LIBS)
|
||||
|
||||
DEP = $(REDIS_SERVER_OBJ:%.o=%.d) $(REDIS_CLI_OBJ:%.o=%.d) $(REDIS_BENCHMARK_OBJ:%.o=%.d)
|
||||
DEP = $(REDIS_SERVER_OBJ:%.o=%.d) $(REDIS_VEC_SETS_OBJ:%.o=%.d) $(REDIS_CLI_OBJ:%.o=%.d) $(REDIS_BENCHMARK_OBJ:%.o=%.d)
|
||||
-include $(DEP)
|
||||
|
||||
# Because the jemalloc.h header is generated as a part of the jemalloc build,
|
||||
@@ -487,8 +495,9 @@ test-cluster: $(REDIS_SERVER_NAME) $(REDIS_CLI_NAME)
|
||||
check: test
|
||||
|
||||
lcov:
|
||||
@lcov --version
|
||||
$(MAKE) gcov
|
||||
@(set -e; cd ..; ./runtest --clients 1)
|
||||
@(set -e; cd ..; ./runtest)
|
||||
@geninfo -o redis.info .
|
||||
@genhtml --legend -o lcov-html redis.info
|
||||
|
||||
@@ -501,7 +510,7 @@ bench: $(REDIS_BENCHMARK_NAME)
|
||||
@echo ""
|
||||
@echo "WARNING: if it fails under Linux you probably need to install libc6-dev-i386"
|
||||
@echo ""
|
||||
$(MAKE) CFLAGS="-m32" LDFLAGS="-m32"
|
||||
$(MAKE) CFLAGS="-m32" LDFLAGS="-m32" SKIP_VEC_SETS="yes"
|
||||
|
||||
gcov:
|
||||
$(MAKE) REDIS_CFLAGS="-fprofile-arcs -ftest-coverage -DCOVERAGE_TEST" REDIS_LDFLAGS="-fprofile-arcs -ftest-coverage"
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
*/
|
||||
|
||||
#include "server.h"
|
||||
#include "cluster.h"
|
||||
#include "sha256.h"
|
||||
#include <fcntl.h>
|
||||
#include <ctype.h>
|
||||
@@ -277,7 +278,7 @@ int ACLListMatchSds(void *a, void *b) {
|
||||
|
||||
/* Method to free list elements from ACL users password/patterns lists. */
|
||||
void ACLListFreeSds(void *item) {
|
||||
sdsfree(item);
|
||||
sdsfreegeneric(item);
|
||||
}
|
||||
|
||||
/* Method to duplicate list elements from ACL users password/patterns lists. */
|
||||
@@ -469,6 +470,11 @@ void ACLFreeUser(user *u) {
|
||||
zfree(u);
|
||||
}
|
||||
|
||||
/* Generic version of ACLFreeUser. */
|
||||
void ACLFreeUserGeneric(void *u) {
|
||||
ACLFreeUser((user *)u);
|
||||
}
|
||||
|
||||
/* When a user is deleted we need to cycle the active
|
||||
* connections in order to kill all the pending ones that
|
||||
* are authenticated with such user. */
|
||||
@@ -1061,6 +1067,7 @@ int ACLSetSelector(aclSelector *selector, const char* op, size_t oplen) {
|
||||
int flags = 0;
|
||||
size_t offset = 1;
|
||||
if (op[0] == '%') {
|
||||
int perm_ok = 1;
|
||||
for (; offset < oplen; offset++) {
|
||||
if (toupper(op[offset]) == 'R' && !(flags & ACL_READ_PERMISSION)) {
|
||||
flags |= ACL_READ_PERMISSION;
|
||||
@@ -1070,10 +1077,14 @@ int ACLSetSelector(aclSelector *selector, const char* op, size_t oplen) {
|
||||
offset++;
|
||||
break;
|
||||
} else {
|
||||
errno = EINVAL;
|
||||
return C_ERR;
|
||||
perm_ok = 0;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!flags || !perm_ok) {
|
||||
errno = EINVAL;
|
||||
return C_ERR;
|
||||
}
|
||||
} else {
|
||||
flags = ACL_ALL_PERMISSION;
|
||||
}
|
||||
@@ -1577,14 +1588,22 @@ static int ACLSelectorCheckKey(aclSelector *selector, const char *key, int keyle
|
||||
if (keyspec_flags & CMD_KEY_DELETE) key_flags |= ACL_WRITE_PERMISSION;
|
||||
if (keyspec_flags & CMD_KEY_UPDATE) key_flags |= ACL_WRITE_PERMISSION;
|
||||
|
||||
/* Is given key represent a prefix of a set of keys */
|
||||
int prefix = keyspec_flags & CMD_KEY_PREFIX;
|
||||
|
||||
/* Test this key against every pattern. */
|
||||
while((ln = listNext(&li))) {
|
||||
keyPattern *pattern = listNodeValue(ln);
|
||||
if ((pattern->flags & key_flags) != key_flags)
|
||||
continue;
|
||||
size_t plen = sdslen(pattern->pattern);
|
||||
if (stringmatchlen(pattern->pattern,plen,key,keylen,0))
|
||||
return ACL_OK;
|
||||
if (prefix) {
|
||||
if (prefixmatch(pattern->pattern,plen,key,keylen,0))
|
||||
return ACL_OK;
|
||||
} else {
|
||||
if (stringmatchlen(pattern->pattern, plen, key, keylen, 0))
|
||||
return ACL_OK;
|
||||
}
|
||||
}
|
||||
return ACL_DENIED_KEY;
|
||||
}
|
||||
@@ -2446,12 +2465,12 @@ sds ACLLoadFromFile(const char *filename) {
|
||||
}
|
||||
|
||||
if (user_channels)
|
||||
raxFreeWithCallback(user_channels, (void(*)(void*))listRelease);
|
||||
raxFreeWithCallback(old_users,(void(*)(void*))ACLFreeUser);
|
||||
raxFreeWithCallback(user_channels, listReleaseGeneric);
|
||||
raxFreeWithCallback(old_users, ACLFreeUserGeneric);
|
||||
sdsfree(errors);
|
||||
return NULL;
|
||||
} else {
|
||||
raxFreeWithCallback(Users,(void(*)(void*))ACLFreeUser);
|
||||
raxFreeWithCallback(Users, ACLFreeUserGeneric);
|
||||
Users = old_users;
|
||||
errors = sdscat(errors,"WARNING: ACL errors detected, no change to the previously active ACL rules was performed");
|
||||
return errors;
|
||||
@@ -2762,7 +2781,6 @@ void aclCatWithFlags(client *c, dict *commands, uint64_t cflag, int *arraylen) {
|
||||
|
||||
while ((de = dictNext(di)) != NULL) {
|
||||
struct redisCommand *cmd = dictGetVal(de);
|
||||
if (cmd->flags & CMD_MODULE) continue;
|
||||
if (cmd->acl_categories & cflag) {
|
||||
addReplyBulkCBuffer(c, cmd->fullname, sdslen(cmd->fullname));
|
||||
(*arraylen)++;
|
||||
@@ -3177,6 +3195,38 @@ void addReplyCommandCategories(client *c, struct redisCommand *cmd) {
|
||||
setDeferredSetLen(c, flaglen, flagcount);
|
||||
}
|
||||
|
||||
/* When successful, initiates an internal connection, that is able to execute
|
||||
* internal commands (see CMD_INTERNAL). */
|
||||
static void internalAuth(client *c) {
|
||||
if (server.cluster == NULL) {
|
||||
addReplyError(c, "Cannot authenticate as an internal connection on non-cluster instances");
|
||||
return;
|
||||
}
|
||||
|
||||
sds password = c->argv[2]->ptr;
|
||||
|
||||
/* Get internal secret. */
|
||||
size_t len = -1;
|
||||
const char *internal_secret = clusterGetSecret(&len);
|
||||
if (sdslen(password) != len) {
|
||||
addReplyError(c, "-WRONGPASS invalid internal password");
|
||||
return;
|
||||
}
|
||||
if (!time_independent_strcmp((char *)internal_secret, (char *)password, len)) {
|
||||
c->flags |= CLIENT_INTERNAL;
|
||||
/* No further authentication is needed. */
|
||||
c->authenticated = 1;
|
||||
/* Set the user to the unrestricted user, if it is not already set (default). */
|
||||
if (c->user != NULL) {
|
||||
c->user = NULL;
|
||||
moduleNotifyUserChanged(c);
|
||||
}
|
||||
addReply(c, shared.ok);
|
||||
} else {
|
||||
addReplyError(c, "-WRONGPASS invalid internal password");
|
||||
}
|
||||
}
|
||||
|
||||
/* AUTH <password>
|
||||
* AUTH <username> <password> (Redis >= 6.0 form)
|
||||
*
|
||||
@@ -3210,6 +3260,14 @@ void authCommand(client *c) {
|
||||
username = c->argv[1];
|
||||
password = c->argv[2];
|
||||
redactClientCommandArgument(c, 2);
|
||||
|
||||
/* Handle internal authentication commands.
|
||||
* Note: No user-defined ACL user can have this username (no spaces
|
||||
* allowed), thus no conflicts with ACL possible. */
|
||||
if (!strcmp(username->ptr, "internal connection")) {
|
||||
internalAuth(c);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
robj *err = NULL;
|
||||
|
||||
@@ -61,6 +61,11 @@ void listRelease(list *list)
|
||||
zfree(list);
|
||||
}
|
||||
|
||||
/* Generic version of listRelease. */
|
||||
void listReleaseGeneric(void *list) {
|
||||
listRelease((struct list*)list);
|
||||
}
|
||||
|
||||
/* Add a new node to the list, to head, containing the specified 'value'
|
||||
* pointer as value.
|
||||
*
|
||||
|
||||
@@ -51,6 +51,7 @@ typedef struct list {
|
||||
/* Prototypes */
|
||||
list *listCreate(void);
|
||||
void listRelease(list *list);
|
||||
void listReleaseGeneric(void *list);
|
||||
void listEmpty(list *list);
|
||||
list *listAddNodeHead(list *list, void *value);
|
||||
list *listAddNodeTail(list *list, void *value);
|
||||
|
||||
@@ -42,7 +42,7 @@
|
||||
#endif
|
||||
#endif
|
||||
|
||||
|
||||
#define INITIAL_EVENT 1024
|
||||
aeEventLoop *aeCreateEventLoop(int setsize) {
|
||||
aeEventLoop *eventLoop;
|
||||
int i;
|
||||
@@ -50,8 +50,9 @@ aeEventLoop *aeCreateEventLoop(int setsize) {
|
||||
monotonicInit(); /* just in case the calling app didn't initialize */
|
||||
|
||||
if ((eventLoop = zmalloc(sizeof(*eventLoop))) == NULL) goto err;
|
||||
eventLoop->events = zmalloc(sizeof(aeFileEvent)*setsize);
|
||||
eventLoop->fired = zmalloc(sizeof(aeFiredEvent)*setsize);
|
||||
eventLoop->nevents = setsize < INITIAL_EVENT ? setsize : INITIAL_EVENT;
|
||||
eventLoop->events = zmalloc(sizeof(aeFileEvent)*eventLoop->nevents);
|
||||
eventLoop->fired = zmalloc(sizeof(aeFiredEvent)*eventLoop->nevents);
|
||||
if (eventLoop->events == NULL || eventLoop->fired == NULL) goto err;
|
||||
eventLoop->setsize = setsize;
|
||||
eventLoop->timeEventHead = NULL;
|
||||
@@ -61,10 +62,11 @@ aeEventLoop *aeCreateEventLoop(int setsize) {
|
||||
eventLoop->beforesleep = NULL;
|
||||
eventLoop->aftersleep = NULL;
|
||||
eventLoop->flags = 0;
|
||||
memset(eventLoop->privdata, 0, sizeof(eventLoop->privdata));
|
||||
if (aeApiCreate(eventLoop) == -1) goto err;
|
||||
/* Events with mask == AE_NONE are not set. So let's initialize the
|
||||
* vector with it. */
|
||||
for (i = 0; i < setsize; i++)
|
||||
for (i = 0; i < eventLoop->nevents; i++)
|
||||
eventLoop->events[i].mask = AE_NONE;
|
||||
return eventLoop;
|
||||
|
||||
@@ -102,20 +104,19 @@ void aeSetDontWait(aeEventLoop *eventLoop, int noWait) {
|
||||
*
|
||||
* Otherwise AE_OK is returned and the operation is successful. */
|
||||
int aeResizeSetSize(aeEventLoop *eventLoop, int setsize) {
|
||||
int i;
|
||||
|
||||
if (setsize == eventLoop->setsize) return AE_OK;
|
||||
if (eventLoop->maxfd >= setsize) return AE_ERR;
|
||||
if (aeApiResize(eventLoop,setsize) == -1) return AE_ERR;
|
||||
|
||||
eventLoop->events = zrealloc(eventLoop->events,sizeof(aeFileEvent)*setsize);
|
||||
eventLoop->fired = zrealloc(eventLoop->fired,sizeof(aeFiredEvent)*setsize);
|
||||
eventLoop->setsize = setsize;
|
||||
|
||||
/* Make sure that if we created new slots, they are initialized with
|
||||
* an AE_NONE mask. */
|
||||
for (i = eventLoop->maxfd+1; i < setsize; i++)
|
||||
eventLoop->events[i].mask = AE_NONE;
|
||||
/* If the current allocated space is larger than the requested size,
|
||||
* we need to shrink it to the requested size. */
|
||||
if (setsize < eventLoop->nevents) {
|
||||
eventLoop->events = zrealloc(eventLoop->events,sizeof(aeFileEvent)*setsize);
|
||||
eventLoop->fired = zrealloc(eventLoop->fired,sizeof(aeFiredEvent)*setsize);
|
||||
eventLoop->nevents = setsize;
|
||||
}
|
||||
return AE_OK;
|
||||
}
|
||||
|
||||
@@ -147,6 +148,22 @@ int aeCreateFileEvent(aeEventLoop *eventLoop, int fd, int mask,
|
||||
errno = ERANGE;
|
||||
return AE_ERR;
|
||||
}
|
||||
|
||||
/* Resize the events and fired arrays if the file
|
||||
* descriptor exceeds the current number of events. */
|
||||
if (unlikely(fd >= eventLoop->nevents)) {
|
||||
int newnevents = eventLoop->nevents;
|
||||
newnevents = (newnevents * 2 > fd + 1) ? newnevents * 2 : fd + 1;
|
||||
newnevents = (newnevents > eventLoop->setsize) ? eventLoop->setsize : newnevents;
|
||||
eventLoop->events = zrealloc(eventLoop->events, sizeof(aeFileEvent) * newnevents);
|
||||
eventLoop->fired = zrealloc(eventLoop->fired, sizeof(aeFiredEvent) * newnevents);
|
||||
|
||||
/* Initialize new slots with an AE_NONE mask */
|
||||
for (int i = eventLoop->nevents; i < newnevents; i++)
|
||||
eventLoop->events[i].mask = AE_NONE;
|
||||
eventLoop->nevents = newnevents;
|
||||
}
|
||||
|
||||
aeFileEvent *fe = &eventLoop->events[fd];
|
||||
|
||||
if (aeApiAddEvent(eventLoop, fd, mask) == -1)
|
||||
|
||||
@@ -79,6 +79,7 @@ typedef struct aeEventLoop {
|
||||
int maxfd; /* highest file descriptor currently registered */
|
||||
int setsize; /* max number of file descriptors tracked */
|
||||
long long timeEventNextId;
|
||||
int nevents; /* Size of Registered events */
|
||||
aeFileEvent *events; /* Registered events */
|
||||
aeFiredEvent *fired; /* Fired events */
|
||||
aeTimeEvent *timeEventHead;
|
||||
@@ -87,6 +88,7 @@ typedef struct aeEventLoop {
|
||||
aeBeforeSleepProc *beforesleep;
|
||||
aeBeforeSleepProc *aftersleep;
|
||||
int flags;
|
||||
void *privdata[2];
|
||||
} aeEventLoop;
|
||||
|
||||
/* Prototypes */
|
||||
|
||||
@@ -30,6 +30,13 @@ aofManifest *aofLoadManifestFromFile(sds am_filepath);
|
||||
void aofManifestFreeAndUpdate(aofManifest *am);
|
||||
void aof_background_fsync_and_close(int fd);
|
||||
|
||||
/* When we call 'startAppendOnly', we will create a temp INCR AOF, and rename
|
||||
* it to the real INCR AOF name when the AOFRW is done, so if want to know the
|
||||
* accurate start offset of the INCR AOF, we need to record it when we create
|
||||
* the temp INCR AOF. This variable is used to record the start offset, and
|
||||
* set the start offset of the real INCR AOF when the AOFRW is done. */
|
||||
static long long tempIncAofStartReplOffset = 0;
|
||||
|
||||
/* ----------------------------------------------------------------------------
|
||||
* AOF Manifest file implementation.
|
||||
*
|
||||
@@ -73,10 +80,15 @@ void aof_background_fsync_and_close(int fd);
|
||||
#define AOF_MANIFEST_KEY_FILE_NAME "file"
|
||||
#define AOF_MANIFEST_KEY_FILE_SEQ "seq"
|
||||
#define AOF_MANIFEST_KEY_FILE_TYPE "type"
|
||||
#define AOF_MANIFEST_KEY_FILE_STARTOFFSET "startoffset"
|
||||
#define AOF_MANIFEST_KEY_FILE_ENDOFFSET "endoffset"
|
||||
|
||||
/* Create an empty aofInfo. */
|
||||
aofInfo *aofInfoCreate(void) {
|
||||
return zcalloc(sizeof(aofInfo));
|
||||
aofInfo *ai = zcalloc(sizeof(aofInfo));
|
||||
ai->start_offset = -1;
|
||||
ai->end_offset = -1;
|
||||
return ai;
|
||||
}
|
||||
|
||||
/* Free the aofInfo structure (pointed to by ai) and its embedded file_name. */
|
||||
@@ -93,6 +105,8 @@ aofInfo *aofInfoDup(aofInfo *orig) {
|
||||
ai->file_name = sdsdup(orig->file_name);
|
||||
ai->file_seq = orig->file_seq;
|
||||
ai->file_type = orig->file_type;
|
||||
ai->start_offset = orig->start_offset;
|
||||
ai->end_offset = orig->end_offset;
|
||||
return ai;
|
||||
}
|
||||
|
||||
@@ -105,10 +119,19 @@ sds aofInfoFormat(sds buf, aofInfo *ai) {
|
||||
if (sdsneedsrepr(ai->file_name))
|
||||
filename_repr = sdscatrepr(sdsempty(), ai->file_name, sdslen(ai->file_name));
|
||||
|
||||
sds ret = sdscatprintf(buf, "%s %s %s %lld %s %c\n",
|
||||
sds ret = sdscatprintf(buf, "%s %s %s %lld %s %c",
|
||||
AOF_MANIFEST_KEY_FILE_NAME, filename_repr ? filename_repr : ai->file_name,
|
||||
AOF_MANIFEST_KEY_FILE_SEQ, ai->file_seq,
|
||||
AOF_MANIFEST_KEY_FILE_TYPE, ai->file_type);
|
||||
|
||||
if (ai->start_offset != -1) {
|
||||
ret = sdscatprintf(ret, " %s %lld", AOF_MANIFEST_KEY_FILE_STARTOFFSET, ai->start_offset);
|
||||
if (ai->end_offset != -1) {
|
||||
ret = sdscatprintf(ret, " %s %lld", AOF_MANIFEST_KEY_FILE_ENDOFFSET, ai->end_offset);
|
||||
}
|
||||
}
|
||||
|
||||
ret = sdscatlen(ret, "\n", 1);
|
||||
sdsfree(filename_repr);
|
||||
|
||||
return ret;
|
||||
@@ -304,6 +327,10 @@ aofManifest *aofLoadManifestFromFile(sds am_filepath) {
|
||||
ai->file_seq = atoll(argv[i+1]);
|
||||
} else if (!strcasecmp(argv[i], AOF_MANIFEST_KEY_FILE_TYPE)) {
|
||||
ai->file_type = (argv[i+1])[0];
|
||||
} else if (!strcasecmp(argv[i], AOF_MANIFEST_KEY_FILE_STARTOFFSET)) {
|
||||
ai->start_offset = atoll(argv[i+1]);
|
||||
} else if (!strcasecmp(argv[i], AOF_MANIFEST_KEY_FILE_ENDOFFSET)) {
|
||||
ai->end_offset = atoll(argv[i+1]);
|
||||
}
|
||||
/* else if (!strcasecmp(argv[i], AOF_MANIFEST_KEY_OTHER)) {} */
|
||||
}
|
||||
@@ -433,12 +460,13 @@ sds getNewBaseFileNameAndMarkPreAsHistory(aofManifest *am) {
|
||||
* for example:
|
||||
* appendonly.aof.1.incr.aof
|
||||
*/
|
||||
sds getNewIncrAofName(aofManifest *am) {
|
||||
sds getNewIncrAofName(aofManifest *am, long long start_reploff) {
|
||||
aofInfo *ai = aofInfoCreate();
|
||||
ai->file_type = AOF_FILE_TYPE_INCR;
|
||||
ai->file_name = sdscatprintf(sdsempty(), "%s.%lld%s%s", server.aof_filename,
|
||||
++am->curr_incr_file_seq, INCR_FILE_SUFFIX, AOF_FORMAT_SUFFIX);
|
||||
ai->file_seq = am->curr_incr_file_seq;
|
||||
ai->start_offset = start_reploff;
|
||||
listAddNodeTail(am->incr_aof_list, ai);
|
||||
am->dirty = 1;
|
||||
return ai->file_name;
|
||||
@@ -456,7 +484,7 @@ sds getLastIncrAofName(aofManifest *am) {
|
||||
|
||||
/* If 'incr_aof_list' is empty, just create a new one. */
|
||||
if (!listLength(am->incr_aof_list)) {
|
||||
return getNewIncrAofName(am);
|
||||
return getNewIncrAofName(am, server.master_repl_offset);
|
||||
}
|
||||
|
||||
/* Or return the last one. */
|
||||
@@ -781,10 +809,11 @@ int openNewIncrAofForAppend(void) {
|
||||
if (server.aof_state == AOF_WAIT_REWRITE) {
|
||||
/* Use a temporary INCR AOF file to accumulate data during AOF_WAIT_REWRITE. */
|
||||
new_aof_name = getTempIncrAofName();
|
||||
tempIncAofStartReplOffset = server.master_repl_offset;
|
||||
} else {
|
||||
/* Dup a temp aof_manifest to modify. */
|
||||
temp_am = aofManifestDup(server.aof_manifest);
|
||||
new_aof_name = sdsdup(getNewIncrAofName(temp_am));
|
||||
new_aof_name = sdsdup(getNewIncrAofName(temp_am, server.master_repl_offset));
|
||||
}
|
||||
sds new_aof_filepath = makePath(server.aof_dirname, new_aof_name);
|
||||
newfd = open(new_aof_filepath, O_WRONLY|O_TRUNC|O_CREAT, 0644);
|
||||
@@ -833,6 +862,50 @@ cleanup:
|
||||
return C_ERR;
|
||||
}
|
||||
|
||||
/* When we close gracefully the AOF file, we have the chance to persist the
|
||||
* end replication offset of current INCR AOF. */
|
||||
void updateCurIncrAofEndOffset(void) {
|
||||
if (server.aof_state != AOF_ON) return;
|
||||
serverAssert(server.aof_manifest != NULL);
|
||||
|
||||
if (listLength(server.aof_manifest->incr_aof_list) == 0) return;
|
||||
aofInfo *ai = listNodeValue(listLast(server.aof_manifest->incr_aof_list));
|
||||
ai->end_offset = server.master_repl_offset;
|
||||
server.aof_manifest->dirty = 1;
|
||||
/* It doesn't matter if the persistence fails since this information is not
|
||||
* critical, we can get an approximate value by start offset plus file size. */
|
||||
persistAofManifest(server.aof_manifest);
|
||||
}
|
||||
|
||||
/* After loading AOF data, we need to update the `server.master_repl_offset`
|
||||
* based on the information of the last INCR AOF, to avoid the rollback of
|
||||
* the start offset of new INCR AOF. */
|
||||
void updateReplOffsetAndResetEndOffset(void) {
|
||||
if (server.aof_state != AOF_ON) return;
|
||||
serverAssert(server.aof_manifest != NULL);
|
||||
|
||||
/* If the INCR file has an end offset, we directly use it, and clear it
|
||||
* to avoid the next time we load the manifest file, we will use the same
|
||||
* offset, but the real offset may have advanced. */
|
||||
if (listLength(server.aof_manifest->incr_aof_list) == 0) return;
|
||||
aofInfo *ai = listNodeValue(listLast(server.aof_manifest->incr_aof_list));
|
||||
if (ai->end_offset != -1) {
|
||||
server.master_repl_offset = ai->end_offset;
|
||||
ai->end_offset = -1;
|
||||
server.aof_manifest->dirty = 1;
|
||||
/* We must update the end offset of INCR file correctly, otherwise we
|
||||
* may keep wrong information in the manifest file, since we continue
|
||||
* to append data to the same INCR file. */
|
||||
if (persistAofManifest(server.aof_manifest) != AOF_OK)
|
||||
exit(1);
|
||||
} else {
|
||||
/* If the INCR file doesn't have an end offset, we need to calculate
|
||||
* the replication offset by the start offset plus the file size. */
|
||||
server.master_repl_offset = (ai->start_offset == -1 ? 0 : ai->start_offset) +
|
||||
getAppendOnlyFileSize(ai->file_name, NULL);
|
||||
}
|
||||
}
|
||||
|
||||
/* Whether to limit the execution of Background AOF rewrite.
|
||||
*
|
||||
* At present, if AOFRW fails, redis will automatically retry. If it continues
|
||||
@@ -938,6 +1011,7 @@ void stopAppendOnly(void) {
|
||||
server.aof_last_fsync = server.mstime;
|
||||
}
|
||||
close(server.aof_fd);
|
||||
updateCurIncrAofEndOffset();
|
||||
|
||||
server.aof_fd = -1;
|
||||
server.aof_selected_db = -1;
|
||||
@@ -997,6 +1071,29 @@ int startAppendOnly(void) {
|
||||
return C_OK;
|
||||
}
|
||||
|
||||
void startAppendOnlyWithRetry(void) {
|
||||
unsigned int tries, max_tries = 10;
|
||||
for (tries = 0; tries < max_tries; ++tries) {
|
||||
if (startAppendOnly() == C_OK)
|
||||
break;
|
||||
serverLog(LL_WARNING, "Failed to enable AOF! Trying it again in one second.");
|
||||
sleep(1);
|
||||
}
|
||||
if (tries == max_tries) {
|
||||
serverLog(LL_WARNING, "FATAL: AOF can't be turned on. Exiting now.");
|
||||
exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
/* Called after "appendonly" config is changed. */
|
||||
void applyAppendOnlyConfig(void) {
|
||||
if (!server.aof_enabled && server.aof_state != AOF_OFF) {
|
||||
stopAppendOnly();
|
||||
} else if (server.aof_enabled && server.aof_state == AOF_OFF) {
|
||||
startAppendOnlyWithRetry();
|
||||
}
|
||||
}
|
||||
|
||||
/* This is a wrapper to the write syscall in order to retry on short writes
|
||||
* or if the syscall gets interrupted. It could look strange that we retry
|
||||
* on short writes given that we are writing to a block device: normally if
|
||||
@@ -1048,35 +1145,34 @@ void flushAppendOnlyFile(int force) {
|
||||
mstime_t latency;
|
||||
|
||||
if (sdslen(server.aof_buf) == 0) {
|
||||
/* Check if we need to do fsync even the aof buffer is empty,
|
||||
* because previously in AOF_FSYNC_EVERYSEC mode, fsync is
|
||||
* called only when aof buffer is not empty, so if users
|
||||
* stop write commands before fsync called in one second,
|
||||
* the data in page cache cannot be flushed in time. */
|
||||
if (server.aof_fsync == AOF_FSYNC_EVERYSEC &&
|
||||
server.aof_last_incr_fsync_offset != server.aof_last_incr_size &&
|
||||
server.mstime - server.aof_last_fsync >= 1000 &&
|
||||
!(sync_in_progress = aofFsyncInProgress())) {
|
||||
goto try_fsync;
|
||||
|
||||
/* Check if we need to do fsync even the aof buffer is empty,
|
||||
* the reason is described in the previous AOF_FSYNC_EVERYSEC block,
|
||||
* and AOF_FSYNC_ALWAYS is also checked here to handle a case where
|
||||
* aof_fsync is changed from everysec to always. */
|
||||
} else if (server.aof_fsync == AOF_FSYNC_ALWAYS &&
|
||||
server.aof_last_incr_fsync_offset != server.aof_last_incr_size)
|
||||
{
|
||||
goto try_fsync;
|
||||
} else {
|
||||
if (server.aof_last_incr_fsync_offset == server.aof_last_incr_size) {
|
||||
/* All data is fsync'd already: Update fsynced_reploff_pending just in case.
|
||||
* This is needed to avoid a WAITAOF hang in case a module used RM_Call with the NO_AOF flag,
|
||||
* in which case master_repl_offset will increase but fsynced_reploff_pending won't be updated
|
||||
* (because there's no reason, from the AOF POV, to call fsync) and then WAITAOF may wait on
|
||||
* the higher offset (which contains data that was only propagated to replicas, and not to AOF) */
|
||||
if (!sync_in_progress && server.aof_fsync != AOF_FSYNC_NO)
|
||||
* This is needed to avoid a WAITAOF hang in case a module used RM_Call
|
||||
* with the NO_AOF flag, in which case master_repl_offset will increase but
|
||||
* fsynced_reploff_pending won't be updated (because there's no reason, from
|
||||
* the AOF POV, to call fsync) and then WAITAOF may wait on the higher offset
|
||||
* (which contains data that was only propagated to replicas, and not to AOF) */
|
||||
if (!aofFsyncInProgress())
|
||||
atomicSet(server.fsynced_reploff_pending, server.master_repl_offset);
|
||||
return;
|
||||
} else {
|
||||
/* Check if we need to do fsync even the aof buffer is empty,
|
||||
* because previously in AOF_FSYNC_EVERYSEC mode, fsync is
|
||||
* called only when aof buffer is not empty, so if users
|
||||
* stop write commands before fsync called in one second,
|
||||
* the data in page cache cannot be flushed in time. */
|
||||
if (server.aof_fsync == AOF_FSYNC_EVERYSEC &&
|
||||
server.mstime - server.aof_last_fsync >= 1000 &&
|
||||
!(sync_in_progress = aofFsyncInProgress()))
|
||||
goto try_fsync;
|
||||
|
||||
/* Check if we need to do fsync even the aof buffer is empty,
|
||||
* the reason is described in the previous AOF_FSYNC_EVERYSEC block,
|
||||
* and AOF_FSYNC_ALWAYS is also checked here to handle a case where
|
||||
* aof_fsync is changed from everysec to always. */
|
||||
if (server.aof_fsync == AOF_FSYNC_ALWAYS)
|
||||
goto try_fsync;
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
if (server.aof_fsync == AOF_FSYNC_EVERYSEC)
|
||||
@@ -2642,7 +2738,7 @@ void backgroundRewriteDoneHandler(int exitcode, int bysignal) {
|
||||
sds temp_incr_aof_name = getTempIncrAofName();
|
||||
sds temp_incr_filepath = makePath(server.aof_dirname, temp_incr_aof_name);
|
||||
/* Get next new incr aof name. */
|
||||
sds new_incr_filename = getNewIncrAofName(temp_am);
|
||||
sds new_incr_filename = getNewIncrAofName(temp_am, tempIncAofStartReplOffset);
|
||||
new_incr_filepath = makePath(server.aof_dirname, new_incr_filename);
|
||||
latencyStartMonitor(latency);
|
||||
if (rename(temp_incr_filepath, new_incr_filepath) == -1) {
|
||||
|
||||
+1
-1
@@ -10,7 +10,7 @@ const char *ascii_logo =
|
||||
" _._ \n"
|
||||
" _.-``__ ''-._ \n"
|
||||
" _.-`` `. `_. ''-._ Redis Community Edition \n"
|
||||
" .-`` .-```. ```\\/ _.,_ ''-._ %s (%s/%d) %s bit\n"
|
||||
" .-`` .-```. ```\\/ _.,_ ''-._ %s (%s/%d) %s bit\n"
|
||||
" ( ' , .-` | `, ) Running in %s mode\n"
|
||||
" |`-._`-...-` __...-.``-._|'` _.-'| Port: %d\n"
|
||||
" | `-._ `._ / _.-' | PID: %ld\n"
|
||||
|
||||
+1
-1
@@ -32,7 +32,7 @@
|
||||
* (if the flag was 0 -> set to 1, if it's already 1 -> do nothing, but the final result is that the flag is set),
|
||||
* and also it has a full barrier (__sync_lock_test_and_set has acquire barrier).
|
||||
*
|
||||
* NOTE2: Unlike other atomic type, which aren't guaranteed to be lock free, c11 atmoic_flag does.
|
||||
* NOTE2: Unlike other atomic type, which aren't guaranteed to be lock free, c11 atomic_flag does.
|
||||
* To check whether a type is lock free, atomic_is_lock_free() can be used.
|
||||
* It can be considered to limit the flag type to atomic_flag to improve performance.
|
||||
*
|
||||
|
||||
@@ -81,6 +81,7 @@ static int job_comp_pipe[2]; /* Pipe used to awake the event loop */
|
||||
typedef struct bio_comp_item {
|
||||
comp_fn *func; /* callback after completion job will be processed */
|
||||
uint64_t arg; /* user data to be passed to the function */
|
||||
void *ptr; /* user pointer to be passed to the function */
|
||||
} bio_comp_item;
|
||||
|
||||
/* This structure represents a background Job. It is only used locally to this
|
||||
@@ -110,6 +111,7 @@ typedef union bio_job {
|
||||
int type; /* header */
|
||||
comp_fn *fn; /* callback. Handover to main thread to cb as notify for job completion */
|
||||
uint64_t arg; /* callback arguments */
|
||||
void *ptr; /* callback pointer */
|
||||
} comp_rq;
|
||||
} bio_job;
|
||||
|
||||
@@ -200,7 +202,7 @@ void bioCreateLazyFreeJob(lazy_free_fn free_fn, int arg_count, ...) {
|
||||
bioSubmitJob(BIO_LAZY_FREE, job);
|
||||
}
|
||||
|
||||
void bioCreateCompRq(bio_worker_t assigned_worker, comp_fn *func, uint64_t user_data) {
|
||||
void bioCreateCompRq(bio_worker_t assigned_worker, comp_fn *func, uint64_t user_data, void *user_ptr) {
|
||||
int type;
|
||||
switch (assigned_worker) {
|
||||
case BIO_WORKER_CLOSE_FILE:
|
||||
@@ -219,6 +221,7 @@ void bioCreateCompRq(bio_worker_t assigned_worker, comp_fn *func, uint64_t user_
|
||||
bio_job *job = zmalloc(sizeof(*job));
|
||||
job->comp_rq.fn = func;
|
||||
job->comp_rq.arg = user_data;
|
||||
job->comp_rq.ptr = user_ptr;
|
||||
bioSubmitJob(type, job);
|
||||
}
|
||||
|
||||
@@ -339,6 +342,7 @@ void *bioProcessBackgroundJobs(void *arg) {
|
||||
bio_comp_item *comp_rsp = zmalloc(sizeof(bio_comp_item));
|
||||
comp_rsp->func = job->comp_rq.fn;
|
||||
comp_rsp->arg = job->comp_rq.arg;
|
||||
comp_rsp->ptr = job->comp_rq.ptr;
|
||||
|
||||
/* just write it to completion job responses */
|
||||
pthread_mutex_lock(&bio_mutex_comp);
|
||||
@@ -432,7 +436,7 @@ void bioPipeReadJobCompList(aeEventLoop *el, int fd, void *privdata, int mask) {
|
||||
listNode *ln = listFirst(tmp_list);
|
||||
bio_comp_item *rsp = ln->value;
|
||||
listDelNode(tmp_list, ln);
|
||||
rsp->func(rsp->arg);
|
||||
rsp->func(rsp->arg, rsp->ptr);
|
||||
zfree(rsp);
|
||||
}
|
||||
listRelease(tmp_list);
|
||||
|
||||
@@ -10,7 +10,7 @@
|
||||
#define __BIO_H
|
||||
|
||||
typedef void lazy_free_fn(void *args[]);
|
||||
typedef void comp_fn(uint64_t user_data);
|
||||
typedef void comp_fn(uint64_t user_data, void *user_ptr);
|
||||
|
||||
typedef enum bio_worker_t {
|
||||
BIO_WORKER_CLOSE_FILE = 0,
|
||||
@@ -40,7 +40,7 @@ void bioCreateCloseJob(int fd, int need_fsync, int need_reclaim_cache);
|
||||
void bioCreateCloseAofJob(int fd, long long offset, int need_reclaim_cache);
|
||||
void bioCreateFsyncJob(int fd, long long offset, int need_reclaim_cache);
|
||||
void bioCreateLazyFreeJob(lazy_free_fn free_fn, int arg_count, ...);
|
||||
void bioCreateCompRq(bio_worker_t assigned_worker, comp_fn *func, uint64_t user_data);
|
||||
void bioCreateCompRq(bio_worker_t assigned_worker, comp_fn *func, uint64_t user_data, void *user_ptr);
|
||||
|
||||
|
||||
#endif
|
||||
|
||||
+72
-17
@@ -16,18 +16,49 @@
|
||||
/* Count number of bits set in the binary array pointed by 's' and long
|
||||
* 'count' bytes. The implementation of this function is required to
|
||||
* work with an input string length up to 512 MB or more (server.proto_max_bulk_len) */
|
||||
ATTRIBUTE_TARGET_POPCNT
|
||||
long long redisPopcount(void *s, long count) {
|
||||
long long bits = 0;
|
||||
unsigned char *p = s;
|
||||
uint32_t *p4;
|
||||
#if defined(HAVE_POPCNT)
|
||||
int use_popcnt = __builtin_cpu_supports("popcnt"); /* Check if CPU supports POPCNT instruction. */
|
||||
#else
|
||||
int use_popcnt = 0; /* Assume CPU does not support POPCNT if
|
||||
* __builtin_cpu_supports() is not available. */
|
||||
#endif
|
||||
static const unsigned char bitsinbyte[256] = {0,1,1,2,1,2,2,3,1,2,2,3,2,3,3,4,1,2,2,3,2,3,3,4,2,3,3,4,3,4,4,5,1,2,2,3,2,3,3,4,2,3,3,4,3,4,4,5,2,3,3,4,3,4,4,5,3,4,4,5,4,5,5,6,1,2,2,3,2,3,3,4,2,3,3,4,3,4,4,5,2,3,3,4,3,4,4,5,3,4,4,5,4,5,5,6,2,3,3,4,3,4,4,5,3,4,4,5,4,5,5,6,3,4,4,5,4,5,5,6,4,5,5,6,5,6,6,7,1,2,2,3,2,3,3,4,2,3,3,4,3,4,4,5,2,3,3,4,3,4,4,5,3,4,4,5,4,5,5,6,2,3,3,4,3,4,4,5,3,4,4,5,4,5,5,6,3,4,4,5,4,5,5,6,4,5,5,6,5,6,6,7,2,3,3,4,3,4,4,5,3,4,4,5,4,5,5,6,3,4,4,5,4,5,5,6,4,5,5,6,5,6,6,7,3,4,4,5,4,5,5,6,4,5,5,6,5,6,6,7,4,5,5,6,5,6,6,7,5,6,6,7,6,7,7,8};
|
||||
|
||||
/* Count initial bytes not aligned to 32 bit. */
|
||||
while((unsigned long)p & 3 && count) {
|
||||
|
||||
/* Count initial bytes not aligned to 64-bit when using the POPCNT instruction,
|
||||
* otherwise align to 32-bit. */
|
||||
int align = use_popcnt ? 7 : 3;
|
||||
while ((unsigned long)p & align && count) {
|
||||
bits += bitsinbyte[*p++];
|
||||
count--;
|
||||
}
|
||||
|
||||
if (likely(use_popcnt)) {
|
||||
/* Use separate counters to make the CPU think there are no
|
||||
* dependencies between these popcnt operations. */
|
||||
uint64_t cnt[4];
|
||||
memset(cnt, 0, sizeof(cnt));
|
||||
|
||||
/* Count bits 32 bytes at a time by using popcnt.
|
||||
* Unroll the loop to avoid the overhead of a single popcnt per iteration,
|
||||
* allowing the CPU to extract more instruction-level parallelism.
|
||||
* Reference: https://danluu.com/assembly-intrinsics/ */
|
||||
while (count >= 32) {
|
||||
cnt[0] += __builtin_popcountll(*(uint64_t*)(p));
|
||||
cnt[1] += __builtin_popcountll(*(uint64_t*)(p + 8));
|
||||
cnt[2] += __builtin_popcountll(*(uint64_t*)(p + 16));
|
||||
cnt[3] += __builtin_popcountll(*(uint64_t*)(p + 24));
|
||||
count -= 32;
|
||||
p += 32;
|
||||
}
|
||||
bits += cnt[0] + cnt[1] + cnt[2] + cnt[3];
|
||||
goto remain;
|
||||
}
|
||||
|
||||
/* Count bits 28 bytes at a time */
|
||||
p4 = (uint32_t*)p;
|
||||
while(count>=28) {
|
||||
@@ -64,8 +95,10 @@ long long redisPopcount(void *s, long count) {
|
||||
((aux6 + (aux6 >> 4)) & 0x0F0F0F0F) +
|
||||
((aux7 + (aux7 >> 4)) & 0x0F0F0F0F))* 0x01010101) >> 24;
|
||||
}
|
||||
/* Count the remaining bytes. */
|
||||
p = (unsigned char*)p4;
|
||||
|
||||
remain:
|
||||
/* Count the remaining bytes. */
|
||||
while(count--) bits += bitsinbyte[*p++];
|
||||
return bits;
|
||||
}
|
||||
@@ -456,22 +489,27 @@ int getBitfieldTypeFromArgument(client *c, robj *o, int *sign, int *bits) {
|
||||
* bits to a string object. The command creates or pad with zeroes the string
|
||||
* so that the 'maxbit' bit can be addressed. The object is finally
|
||||
* returned. Otherwise if the key holds a wrong type NULL is returned and
|
||||
* an error is sent to the client. */
|
||||
robj *lookupStringForBitCommand(client *c, uint64_t maxbit, int *dirty) {
|
||||
* an error is sent to the client.
|
||||
*
|
||||
* (Must provide all the arguments to the function)
|
||||
*/
|
||||
static robj *lookupStringForBitCommand(client *c, uint64_t maxbit,
|
||||
size_t *strOldSize, size_t *strGrowSize)
|
||||
{
|
||||
size_t byte = maxbit >> 3;
|
||||
robj *o = lookupKeyWrite(c->db,c->argv[1]);
|
||||
if (checkType(c,o,OBJ_STRING)) return NULL;
|
||||
if (dirty) *dirty = 0;
|
||||
|
||||
if (o == NULL) {
|
||||
o = createObject(OBJ_STRING,sdsnewlen(NULL, byte+1));
|
||||
dbAdd(c->db,c->argv[1],o);
|
||||
if (dirty) *dirty = 1;
|
||||
*strGrowSize = byte + 1;
|
||||
*strOldSize = 0;
|
||||
} else {
|
||||
o = dbUnshareStringValue(c->db,c->argv[1],o);
|
||||
size_t oldlen = sdslen(o->ptr);
|
||||
*strOldSize = sdslen(o->ptr);
|
||||
o->ptr = sdsgrowzero(o->ptr,byte+1);
|
||||
if (dirty && oldlen != sdslen(o->ptr)) *dirty = 1;
|
||||
*strGrowSize = sdslen(o->ptr) - *strOldSize;
|
||||
}
|
||||
return o;
|
||||
}
|
||||
@@ -528,8 +566,9 @@ void setbitCommand(client *c) {
|
||||
return;
|
||||
}
|
||||
|
||||
int dirty;
|
||||
if ((o = lookupStringForBitCommand(c,bitoffset,&dirty)) == NULL) return;
|
||||
size_t strOldSize, strGrowSize;
|
||||
if ((o = lookupStringForBitCommand(c,bitoffset,&strOldSize,&strGrowSize)) == NULL)
|
||||
return;
|
||||
|
||||
/* Get current values */
|
||||
byte = bitoffset >> 3;
|
||||
@@ -540,7 +579,7 @@ void setbitCommand(client *c) {
|
||||
/* Either it is newly created, changed length, or the bit changes before and after.
|
||||
* Note that the bitval here is actually a decimal number.
|
||||
* So we need to use `!!` to convert it to 0 or 1 for comparison. */
|
||||
if (dirty || (!!bitval != on)) {
|
||||
if (strGrowSize || (!!bitval != on)) {
|
||||
/* Update byte with new bit value. */
|
||||
byteval &= ~(1 << bit);
|
||||
byteval |= ((on & 0x1) << bit);
|
||||
@@ -548,6 +587,13 @@ void setbitCommand(client *c) {
|
||||
signalModifiedKey(c,c->db,c->argv[1]);
|
||||
notifyKeyspaceEvent(NOTIFY_STRING,"setbit",c->argv[1],c->db->id);
|
||||
server.dirty++;
|
||||
|
||||
/* If this is not a new key (old size not 0) and size changed, then
|
||||
* update the keysizes histogram. Otherwise, the histogram already
|
||||
* updated in lookupStringForBitCommand() by calling dbAdd(). */
|
||||
if ((strOldSize > 0) && (strGrowSize != 0))
|
||||
updateKeysizesHist(c->db, getKeySlot(c->argv[1]->ptr), OBJ_STRING,
|
||||
strOldSize, strOldSize + strGrowSize);
|
||||
}
|
||||
|
||||
/* Return original value. */
|
||||
@@ -1032,7 +1078,8 @@ struct bitfieldOp {
|
||||
void bitfieldGeneric(client *c, int flags) {
|
||||
robj *o;
|
||||
uint64_t bitoffset;
|
||||
int j, numops = 0, changes = 0, dirty = 0;
|
||||
int j, numops = 0, changes = 0;
|
||||
size_t strOldSize, strGrowSize = 0;
|
||||
struct bitfieldOp *ops = NULL; /* Array of ops to execute at end. */
|
||||
int owtype = BFOVERFLOW_WRAP; /* Overflow type. */
|
||||
int readonly = 1;
|
||||
@@ -1126,7 +1173,7 @@ void bitfieldGeneric(client *c, int flags) {
|
||||
/* Lookup by making room up to the farthest bit reached by
|
||||
* this operation. */
|
||||
if ((o = lookupStringForBitCommand(c,
|
||||
highest_write_offset,&dirty)) == NULL) {
|
||||
highest_write_offset,&strOldSize,&strGrowSize)) == NULL) {
|
||||
zfree(ops);
|
||||
return;
|
||||
}
|
||||
@@ -1176,7 +1223,7 @@ void bitfieldGeneric(client *c, int flags) {
|
||||
setSignedBitfield(o->ptr,thisop->offset,
|
||||
thisop->bits,newval);
|
||||
|
||||
if (dirty || (oldval != newval))
|
||||
if (strGrowSize || (oldval != newval))
|
||||
changes++;
|
||||
} else {
|
||||
addReplyNull(c);
|
||||
@@ -1210,7 +1257,7 @@ void bitfieldGeneric(client *c, int flags) {
|
||||
setUnsignedBitfield(o->ptr,thisop->offset,
|
||||
thisop->bits,newval);
|
||||
|
||||
if (dirty || (oldval != newval))
|
||||
if (strGrowSize || (oldval != newval))
|
||||
changes++;
|
||||
} else {
|
||||
addReplyNull(c);
|
||||
@@ -1253,6 +1300,14 @@ void bitfieldGeneric(client *c, int flags) {
|
||||
}
|
||||
|
||||
if (changes) {
|
||||
|
||||
/* If this is not a new key (old size not 0) and size changed, then
|
||||
* update the keysizes histogram. Otherwise, the histogram already
|
||||
* updated in lookupStringForBitCommand() by calling dbAdd(). */
|
||||
if ((strOldSize > 0) && (strGrowSize != 0))
|
||||
updateKeysizesHist(c->db, getKeySlot(c->argv[1]->ptr), OBJ_STRING,
|
||||
strOldSize, strOldSize + strGrowSize);
|
||||
|
||||
signalModifiedKey(c,c->db,c->argv[1]);
|
||||
notifyKeyspaceEvent(NOTIFY_STRING,"setbit",c->argv[1],c->db->id);
|
||||
server.dirty += changes;
|
||||
|
||||
@@ -270,6 +270,7 @@ void disconnectAllBlockedClients(void) {
|
||||
|
||||
if (c->bstate.btype == BLOCKED_LAZYFREE) {
|
||||
addReply(c, shared.ok); /* No reason lazy-free to fail */
|
||||
updateStatsOnUnblock(c, 0, 0, 0);
|
||||
c->flags &= ~CLIENT_PENDING_COMMAND;
|
||||
unblockClient(c, 1);
|
||||
} else {
|
||||
|
||||
+1
-1
@@ -533,7 +533,7 @@ CallReply *callReplyCreateError(sds reply, void *private_data) {
|
||||
sdsfree(reply);
|
||||
}
|
||||
list *deferred_error_list = listCreate();
|
||||
listSetFreeMethod(deferred_error_list, (void (*)(void*))sdsfree);
|
||||
listSetFreeMethod(deferred_error_list, sdsfreegeneric);
|
||||
listAddNodeTail(deferred_error_list, sdsnew(err_buff));
|
||||
return callReplyCreate(err_buff, deferred_error_list, private_data);
|
||||
}
|
||||
|
||||
+245
-1
@@ -317,7 +317,7 @@ migrateCachedSocket* migrateGetSocket(client *c, robj *host, robj *port, long ti
|
||||
}
|
||||
|
||||
/* Create the connection */
|
||||
conn = connCreate(connTypeOfCluster());
|
||||
conn = connCreate(server.el, connTypeOfCluster());
|
||||
if (connBlockingConnect(conn, host->ptr, atoi(port->ptr), timeout)
|
||||
!= C_OK) {
|
||||
addReplyError(c,"-IOERR error or timeout connecting to the client");
|
||||
@@ -783,6 +783,130 @@ unsigned int countKeysInSlot(unsigned int slot) {
|
||||
return kvstoreDictSize(server.db->keys, slot);
|
||||
}
|
||||
|
||||
/* Add detailed information of a node to the output buffer of the given client. */
|
||||
void addNodeDetailsToShardReply(client *c, clusterNode *node) {
|
||||
|
||||
int reply_count = 0;
|
||||
char *hostname;
|
||||
void *node_replylen = addReplyDeferredLen(c);
|
||||
|
||||
addReplyBulkCString(c, "id");
|
||||
addReplyBulkCBuffer(c, clusterNodeGetName(node), CLUSTER_NAMELEN);
|
||||
reply_count++;
|
||||
|
||||
if (clusterNodeTcpPort(node)) {
|
||||
addReplyBulkCString(c, "port");
|
||||
addReplyLongLong(c, clusterNodeTcpPort(node));
|
||||
reply_count++;
|
||||
}
|
||||
|
||||
if (clusterNodeTlsPort(node)) {
|
||||
addReplyBulkCString(c, "tls-port");
|
||||
addReplyLongLong(c, clusterNodeTlsPort(node));
|
||||
reply_count++;
|
||||
}
|
||||
|
||||
addReplyBulkCString(c, "ip");
|
||||
addReplyBulkCString(c, clusterNodeIp(node));
|
||||
reply_count++;
|
||||
|
||||
addReplyBulkCString(c, "endpoint");
|
||||
addReplyBulkCString(c, clusterNodePreferredEndpoint(node));
|
||||
reply_count++;
|
||||
|
||||
hostname = clusterNodeHostname(node);
|
||||
if (hostname != NULL && *hostname != '\0') {
|
||||
addReplyBulkCString(c, "hostname");
|
||||
addReplyBulkCString(c, hostname);
|
||||
reply_count++;
|
||||
}
|
||||
|
||||
long long node_offset;
|
||||
if (clusterNodeIsMyself(node)) {
|
||||
node_offset = clusterNodeIsSlave(node) ? replicationGetSlaveOffset() : server.master_repl_offset;
|
||||
} else {
|
||||
node_offset = clusterNodeReplOffset(node);
|
||||
}
|
||||
|
||||
addReplyBulkCString(c, "role");
|
||||
addReplyBulkCString(c, clusterNodeIsSlave(node) ? "replica" : "master");
|
||||
reply_count++;
|
||||
|
||||
addReplyBulkCString(c, "replication-offset");
|
||||
addReplyLongLong(c, node_offset);
|
||||
reply_count++;
|
||||
|
||||
addReplyBulkCString(c, "health");
|
||||
const char *health_msg = NULL;
|
||||
if (clusterNodeIsFailing(node)) {
|
||||
health_msg = "fail";
|
||||
} else if (clusterNodeIsSlave(node) && node_offset == 0) {
|
||||
health_msg = "loading";
|
||||
} else {
|
||||
health_msg = "online";
|
||||
}
|
||||
addReplyBulkCString(c, health_msg);
|
||||
reply_count++;
|
||||
|
||||
setDeferredMapLen(c, node_replylen, reply_count);
|
||||
}
|
||||
|
||||
static clusterNode *clusterGetMasterFromShard(void *shard_handle) {
|
||||
clusterNode *n = NULL;
|
||||
void *node_it = clusterShardHandleGetNodeIterator(shard_handle);
|
||||
while((n = clusterShardNodeIteratorNext(node_it)) != NULL) {
|
||||
if (!clusterNodeIsFailing(n)) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
clusterShardNodeIteratorFree(node_it);
|
||||
if (!n) return NULL;
|
||||
return clusterNodeGetMaster(n);
|
||||
}
|
||||
|
||||
/* Add the shard reply of a single shard based off the given primary node. */
|
||||
void addShardReplyForClusterShards(client *c, void *shard_handle) {
|
||||
serverAssert(clusterGetShardNodeCount(shard_handle) > 0);
|
||||
addReplyMapLen(c, 2);
|
||||
addReplyBulkCString(c, "slots");
|
||||
|
||||
/* Use slot_info_pairs from the primary only */
|
||||
clusterNode *master_node = clusterGetMasterFromShard(shard_handle);
|
||||
|
||||
if (master_node && clusterNodeHasSlotInfo(master_node)) {
|
||||
serverAssert((clusterNodeSlotInfoCount(master_node) % 2) == 0);
|
||||
addReplyArrayLen(c, clusterNodeSlotInfoCount(master_node));
|
||||
for (int i = 0; i < clusterNodeSlotInfoCount(master_node); i++)
|
||||
addReplyLongLong(c, (unsigned long)clusterNodeSlotInfoEntry(master_node, i));
|
||||
} else {
|
||||
/* If no slot info pair is provided, the node owns no slots */
|
||||
addReplyArrayLen(c, 0);
|
||||
}
|
||||
|
||||
addReplyBulkCString(c, "nodes");
|
||||
addReplyArrayLen(c, clusterGetShardNodeCount(shard_handle));
|
||||
void *node_it = clusterShardHandleGetNodeIterator(shard_handle);
|
||||
for (clusterNode *n = clusterShardNodeIteratorNext(node_it); n != NULL; n = clusterShardNodeIteratorNext(node_it)) {
|
||||
addNodeDetailsToShardReply(c, n);
|
||||
clusterFreeNodesSlotsInfo(n);
|
||||
}
|
||||
clusterShardNodeIteratorFree(node_it);
|
||||
}
|
||||
|
||||
/* Add to the output buffer of the given client, an array of slot (start, end)
|
||||
* pair owned by the shard, also the primary and set of replica(s) along with
|
||||
* information about each node. */
|
||||
void clusterCommandShards(client *c) {
|
||||
addReplyArrayLen(c, clusterGetShardCount());
|
||||
/* This call will add slot_info_pairs to all nodes */
|
||||
clusterGenNodesSlotsInfo(0);
|
||||
dictIterator *shard_it = clusterGetShardIterator();
|
||||
for(void *shard_handle = clusterNextShardHandle(shard_it); shard_handle != NULL; shard_handle = clusterNextShardHandle(shard_it)) {
|
||||
addShardReplyForClusterShards(c, shard_handle);
|
||||
}
|
||||
clusterFreeShardIterator(shard_it);
|
||||
}
|
||||
|
||||
void clusterCommandHelp(client *c) {
|
||||
const char *help[] = {
|
||||
"COUNTKEYSINSLOT <slot>",
|
||||
@@ -1434,6 +1558,126 @@ void readonlyCommand(client *c) {
|
||||
addReply(c,shared.ok);
|
||||
}
|
||||
|
||||
void replySlotsFlushAndFree(client *c, SlotsFlush *sflush) {
|
||||
addReplyArrayLen(c, sflush->numRanges);
|
||||
for (int i = 0 ; i < sflush->numRanges ; i++) {
|
||||
addReplyArrayLen(c, 2);
|
||||
addReplyLongLong(c, sflush->ranges[i].first);
|
||||
addReplyLongLong(c, sflush->ranges[i].last);
|
||||
}
|
||||
zfree(sflush);
|
||||
}
|
||||
|
||||
/* Partially flush destination DB in a cluster node, based on the slot range.
|
||||
*
|
||||
* Usage: SFLUSH <start-slot> <end slot> [<start-slot> <end slot>]* [SYNC|ASYNC]
|
||||
*
|
||||
* This is an initial implementation of SFLUSH (slots flush) which is limited to
|
||||
* flushing a single shard as a whole, but in the future the same command may be
|
||||
* used to partially flush a shard based on hash slots. Currently only if provided
|
||||
* slots cover entirely the slots of a node, the node will be flushed and the
|
||||
* return value will be pairs of slot ranges. Otherwise, a single empty set will
|
||||
* be returned. If possible, SFLUSH SYNC will be run as blocking ASYNC as an
|
||||
* optimization.
|
||||
*/
|
||||
void sflushCommand(client *c) {
|
||||
int flags = EMPTYDB_NO_FLAGS, argc = c->argc;
|
||||
|
||||
if (server.cluster_enabled == 0) {
|
||||
addReplyError(c,"This instance has cluster support disabled");
|
||||
return;
|
||||
}
|
||||
|
||||
/* check if last argument is SYNC or ASYNC */
|
||||
if (!strcasecmp(c->argv[c->argc-1]->ptr,"sync")) {
|
||||
flags = EMPTYDB_NO_FLAGS;
|
||||
argc--;
|
||||
} else if (!strcasecmp(c->argv[c->argc-1]->ptr,"async")) {
|
||||
flags = EMPTYDB_ASYNC;
|
||||
argc--;
|
||||
} else if (server.lazyfree_lazy_user_flush) {
|
||||
flags = EMPTYDB_ASYNC;
|
||||
}
|
||||
|
||||
/* parse the slot range */
|
||||
if (argc % 2 == 0) {
|
||||
addReplyErrorArity(c);
|
||||
return;
|
||||
}
|
||||
|
||||
/* Verify <first, last> slot pairs are valid and not overlapping */
|
||||
long long j, first, last;
|
||||
unsigned char slotsToFlushRq[CLUSTER_SLOTS] = {0};
|
||||
for (j = 1; j < argc; j += 2) {
|
||||
/* check if the first slot is valid */
|
||||
if (getLongLongFromObject(c->argv[j], &first) != C_OK || first < 0 || first >= CLUSTER_SLOTS) {
|
||||
addReplyError(c,"Invalid or out of range slot");
|
||||
return;
|
||||
}
|
||||
|
||||
/* check if the last slot is valid */
|
||||
if (getLongLongFromObject(c->argv[j+1], &last) != C_OK || last < 0 || last >= CLUSTER_SLOTS) {
|
||||
addReplyError(c,"Invalid or out of range slot");
|
||||
return;
|
||||
}
|
||||
|
||||
if (first > last) {
|
||||
addReplyErrorFormat(c,"start slot number %lld is greater than end slot number %lld", first, last);
|
||||
return;
|
||||
}
|
||||
|
||||
/* Mark the slots in slotsToFlushRq[] */
|
||||
for (int i = first; i <= last; i++) {
|
||||
if (slotsToFlushRq[i]) {
|
||||
addReplyErrorFormat(c, "Slot %d specified multiple times", i);
|
||||
return;
|
||||
}
|
||||
slotsToFlushRq[i] = 1;
|
||||
}
|
||||
}
|
||||
|
||||
/* Verify slotsToFlushRq[] covers ALL slots of myNode. */
|
||||
clusterNode *myNode = getMyClusterNode();
|
||||
/* During iteration trace also the slot range pairs and save in SlotsFlush.
|
||||
* It is allocated on heap since there is a chance that FLUSH SYNC will be
|
||||
* running as blocking ASYNC and only later reply with slot ranges */
|
||||
int capacity = 32; /* Initial capacity */
|
||||
SlotsFlush *sflush = zmalloc(sizeof(SlotsFlush) + sizeof(SlotRange) * capacity);
|
||||
sflush->numRanges = 0;
|
||||
int inSlotRange = 0;
|
||||
for (int i = 0; i < CLUSTER_SLOTS; i++) {
|
||||
if (myNode == getNodeBySlot(i)) {
|
||||
if (!slotsToFlushRq[i]) {
|
||||
addReplySetLen(c, 0); /* Not all slots of mynode got covered. See sflushCommand() comment. */
|
||||
zfree(sflush);
|
||||
return;
|
||||
}
|
||||
|
||||
if (!inSlotRange) { /* If start another slot range */
|
||||
sflush->ranges[sflush->numRanges].first = i;
|
||||
inSlotRange = 1;
|
||||
}
|
||||
} else {
|
||||
if (inSlotRange) { /* If end another slot range */
|
||||
sflush->ranges[sflush->numRanges++].last = i - 1;
|
||||
inSlotRange = 0;
|
||||
/* If reached 'sflush' capacity, double the capacity */
|
||||
if (sflush->numRanges >= capacity) {
|
||||
capacity *= 2;
|
||||
sflush = zrealloc(sflush, sizeof(SlotsFlush) + sizeof(SlotRange) * capacity);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* Update last pair if last cluster slot is also end of last range */
|
||||
if (inSlotRange) sflush->ranges[sflush->numRanges++].last = CLUSTER_SLOTS - 1;
|
||||
|
||||
/* Flush selected slots. If not flush as blocking async, then reply immediately */
|
||||
if (flushCommandCommon(c, FLUSH_TYPE_SLOTS, flags, sflush) == 0)
|
||||
replySlotsFlushAndFree(c, sflush);
|
||||
}
|
||||
|
||||
/* The READWRITE command just clears the READONLY command state. */
|
||||
void readwriteCommand(client *c) {
|
||||
if (server.cluster_enabled == 0) {
|
||||
|
||||
+21
-1
@@ -97,7 +97,7 @@ int clusterManualFailoverTimeLimit(void);
|
||||
void clusterCommandSlots(client * c);
|
||||
void clusterCommandMyId(client *c);
|
||||
void clusterCommandMyShardId(client *c);
|
||||
void clusterCommandShards(client *c);
|
||||
|
||||
sds clusterGenNodeDescription(client *c, clusterNode *node, int tls_primary);
|
||||
|
||||
int clusterNodeCoversSlot(clusterNode *n, int slot);
|
||||
@@ -131,6 +131,7 @@ char *clusterNodeHostname(clusterNode *node);
|
||||
const char *clusterNodePreferredEndpoint(clusterNode *n);
|
||||
long long clusterNodeReplOffset(clusterNode *node);
|
||||
clusterNode *clusterLookupNode(const char *name, int length);
|
||||
const char *clusterGetSecret(size_t *len);
|
||||
|
||||
/* functions with shared implementations */
|
||||
clusterNode *getNodeByQuery(client *c, struct redisCommand *cmd, robj **argv, int argc, int *hashslot, uint64_t cmd_flags, int *error_code);
|
||||
@@ -142,4 +143,23 @@ int isValidAuxString(char *s, unsigned int length);
|
||||
void migrateCommand(client *c);
|
||||
void clusterCommand(client *c);
|
||||
ConnectionType *connTypeOfCluster(void);
|
||||
|
||||
void clusterGenNodesSlotsInfo(int filter);
|
||||
void clusterFreeNodesSlotsInfo(clusterNode *n);
|
||||
int clusterNodeSlotInfoCount(clusterNode *n);
|
||||
uint16_t clusterNodeSlotInfoEntry(clusterNode *n, int idx);
|
||||
int clusterNodeHasSlotInfo(clusterNode *n);
|
||||
|
||||
int clusterGetShardCount(void);
|
||||
void *clusterGetShardIterator(void);
|
||||
void *clusterNextShardHandle(void *shard_iterator);
|
||||
void clusterFreeShardIterator(void *shard_iterator);
|
||||
int clusterGetShardNodeCount(void *shard);
|
||||
void *clusterShardHandleGetNodeIterator(void *shard);
|
||||
clusterNode *clusterShardNodeIteratorNext(void *node_iterator);
|
||||
void clusterShardNodeIteratorFree(void *node_iterator);
|
||||
clusterNode *clusterShardNodeFirst(void *shard);
|
||||
|
||||
int clusterNodeTcpPort(clusterNode *node);
|
||||
int clusterNodeTlsPort(clusterNode *node);
|
||||
#endif /* __CLUSTER_H */
|
||||
|
||||
+144
-141
@@ -2,8 +2,13 @@
|
||||
* Copyright (c) 2009-Present, Redis Ltd.
|
||||
* All rights reserved.
|
||||
*
|
||||
* Copyright (c) 2024-present, Valkey contributors.
|
||||
* All rights reserved.
|
||||
*
|
||||
* Licensed under your choice of the Redis Source Available License 2.0
|
||||
* (RSALv2) or the Server Side Public License v1 (SSPLv1).
|
||||
*
|
||||
* Portions of this file are available under BSD3 terms; see REDISCONTRIBUTIONS for more information.
|
||||
*/
|
||||
|
||||
/*
|
||||
@@ -88,6 +93,7 @@ int auxTlsPortPresent(clusterNode *n);
|
||||
static void clusterBuildMessageHdr(clusterMsg *hdr, int type, size_t msglen);
|
||||
void freeClusterLink(clusterLink *link);
|
||||
int verifyClusterNodeId(const char *name, int length);
|
||||
static void updateShardId(clusterNode *node, const char *shard_id);
|
||||
|
||||
int getNodeDefaultClientPort(clusterNode *n) {
|
||||
return server.tls_cluster ? n->tls_port : n->tcp_port;
|
||||
@@ -198,12 +204,11 @@ int auxShardIdSetter(clusterNode *n, void *value, int length) {
|
||||
return C_ERR;
|
||||
}
|
||||
memcpy(n->shard_id, value, CLUSTER_NAMELEN);
|
||||
/* if n already has replicas, make sure they all agree
|
||||
* on the shard id */
|
||||
/* if n already has replicas, make sure they all use
|
||||
* the primary shard id */
|
||||
for (int i = 0; i < n->numslaves; i++) {
|
||||
if (memcmp(n->slaves[i]->shard_id, n->shard_id, CLUSTER_NAMELEN) != 0) {
|
||||
return C_ERR;
|
||||
}
|
||||
if (memcmp(n->slaves[i]->shard_id, n->shard_id, CLUSTER_NAMELEN) != 0)
|
||||
updateShardId(n->slaves[i], n->shard_id);
|
||||
}
|
||||
clusterAddNodeToShard(value, n);
|
||||
return C_OK;
|
||||
@@ -545,18 +550,12 @@ int clusterLoadConfig(char *filename) {
|
||||
clusterAddNode(master);
|
||||
}
|
||||
/* shard_id can be absent if we are loading a nodes.conf generated
|
||||
* by an older version of Redis; we should follow the primary's
|
||||
* shard_id in this case */
|
||||
if (auxFieldHandlers[af_shard_id].isPresent(n) == 0) {
|
||||
memcpy(n->shard_id, master->shard_id, CLUSTER_NAMELEN);
|
||||
clusterAddNodeToShard(master->shard_id, n);
|
||||
} else if (clusterGetNodesInMyShard(master) != NULL &&
|
||||
memcmp(master->shard_id, n->shard_id, CLUSTER_NAMELEN) != 0)
|
||||
{
|
||||
/* If the primary has been added to a shard, make sure this
|
||||
* node has the same persisted shard id as the primary. */
|
||||
goto fmterr;
|
||||
}
|
||||
* by an older version of Redis;
|
||||
* ignore replica's shard_id in the file, only use the primary's.
|
||||
* If replica precedes primary in file, it will be corrected
|
||||
* later by the auxShardIdSetter */
|
||||
memcpy(n->shard_id, master->shard_id, CLUSTER_NAMELEN);
|
||||
clusterAddNodeToShard(master->shard_id, n);
|
||||
n->slaveof = master;
|
||||
clusterNodeAddSlave(master,n);
|
||||
} else if (auxFieldHandlers[af_shard_id].isPresent(n) == 0) {
|
||||
@@ -634,6 +633,8 @@ int clusterLoadConfig(char *filename) {
|
||||
}
|
||||
/* Config sanity check */
|
||||
if (server.cluster->myself == NULL) goto fmterr;
|
||||
if (!(myself->flags & (CLUSTER_NODE_MASTER | CLUSTER_NODE_SLAVE))) goto fmterr;
|
||||
if (nodeIsSlave(myself) && myself->slaveof == NULL) goto fmterr;
|
||||
|
||||
zfree(line);
|
||||
fclose(fp);
|
||||
@@ -901,22 +902,33 @@ static void updateAnnouncedHumanNodename(clusterNode *node, char *new) {
|
||||
clusterDoBeforeSleep(CLUSTER_TODO_SAVE_CONFIG);
|
||||
}
|
||||
|
||||
static void assignShardIdToNode(clusterNode *node, const char *shard_id, int flag) {
|
||||
clusterRemoveNodeFromShard(node);
|
||||
memcpy(node->shard_id, shard_id, CLUSTER_NAMELEN);
|
||||
clusterAddNodeToShard(shard_id, node);
|
||||
clusterDoBeforeSleep(flag);
|
||||
}
|
||||
|
||||
static void updateShardId(clusterNode *node, const char *shard_id) {
|
||||
if (shard_id && memcmp(node->shard_id, shard_id, CLUSTER_NAMELEN) != 0) {
|
||||
clusterRemoveNodeFromShard(node);
|
||||
memcpy(node->shard_id, shard_id, CLUSTER_NAMELEN);
|
||||
clusterAddNodeToShard(shard_id, node);
|
||||
clusterDoBeforeSleep(CLUSTER_TODO_SAVE_CONFIG);
|
||||
}
|
||||
if (shard_id && myself != node && myself->slaveof == node) {
|
||||
if (memcmp(myself->shard_id, shard_id, CLUSTER_NAMELEN) != 0) {
|
||||
/* shard-id can diverge right after a rolling upgrade
|
||||
* from pre-7.2 releases */
|
||||
clusterRemoveNodeFromShard(myself);
|
||||
memcpy(myself->shard_id, shard_id, CLUSTER_NAMELEN);
|
||||
clusterAddNodeToShard(shard_id, myself);
|
||||
clusterDoBeforeSleep(CLUSTER_TODO_SAVE_CONFIG|CLUSTER_TODO_FSYNC_CONFIG);
|
||||
/* We always make our best effort to keep the shard-id consistent
|
||||
* between the master and its replicas:
|
||||
*
|
||||
* 1. When updating the master's shard-id, we simultaneously update the
|
||||
* shard-id of all its replicas to ensure consistency.
|
||||
* 2. When updating replica's shard-id, if it differs from its master's shard-id,
|
||||
* we discard this replica's shard-id and continue using master's shard-id.
|
||||
* This applies even if the master does not support shard-id, in which
|
||||
* case we rely on the master's randomly generated shard-id. */
|
||||
if (node->slaveof == NULL) {
|
||||
assignShardIdToNode(node, shard_id, CLUSTER_TODO_SAVE_CONFIG);
|
||||
for (int i = 0; i < clusterNodeNumSlaves(node); i++) {
|
||||
clusterNode *slavenode = clusterNodeGetSlave(node, i);
|
||||
if (memcmp(slavenode->shard_id, shard_id, CLUSTER_NAMELEN) != 0)
|
||||
assignShardIdToNode(slavenode, shard_id, CLUSTER_TODO_SAVE_CONFIG|CLUSTER_TODO_FSYNC_CONFIG);
|
||||
}
|
||||
} else if (memcmp(node->slaveof->shard_id, shard_id, CLUSTER_NAMELEN) == 0) {
|
||||
assignShardIdToNode(node, shard_id, CLUSTER_TODO_SAVE_CONFIG);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1012,6 +1024,8 @@ void clusterInit(void) {
|
||||
clusterUpdateMyselfIp();
|
||||
clusterUpdateMyselfHostname();
|
||||
clusterUpdateMyselfHumanNodename();
|
||||
|
||||
getRandomHexChars(server.cluster->internal_secret, CLUSTER_INTERNALSECRETLEN);
|
||||
}
|
||||
|
||||
void clusterInitLast(void) {
|
||||
@@ -1244,7 +1258,7 @@ void clusterAcceptHandler(aeEventLoop *el, int fd, void *privdata, int mask) {
|
||||
return;
|
||||
}
|
||||
|
||||
connection *conn = connCreateAccepted(connTypeOfCluster(), cfd, &require_auth);
|
||||
connection *conn = connCreateAccepted(server.el, connTypeOfCluster(), cfd, &require_auth);
|
||||
|
||||
/* Make sure connection is not in an error state */
|
||||
if (connGetState(conn) != CONN_STATE_ACCEPTING) {
|
||||
@@ -1561,6 +1575,14 @@ clusterNode *clusterLookupNode(const char *name, int length) {
|
||||
return dictGetVal(de);
|
||||
}
|
||||
|
||||
const char *clusterGetSecret(size_t *len) {
|
||||
if (!server.cluster) {
|
||||
return NULL;
|
||||
}
|
||||
*len = CLUSTER_INTERNALSECRETLEN;
|
||||
return server.cluster->internal_secret;
|
||||
}
|
||||
|
||||
/* Get all the nodes in my shard.
|
||||
* Note that the list returned is not computed on the fly
|
||||
* via slaveof; rather, it is maintained permanently to
|
||||
@@ -2485,6 +2507,10 @@ uint32_t getShardIdPingExtSize(void) {
|
||||
return getAlignedPingExtSize(sizeof(clusterMsgPingExtShardId));
|
||||
}
|
||||
|
||||
uint32_t getInternalSecretPingExtSize(void) {
|
||||
return getAlignedPingExtSize(sizeof(clusterMsgPingExtInternalSecret));
|
||||
}
|
||||
|
||||
uint32_t getForgottenNodeExtSize(void) {
|
||||
return getAlignedPingExtSize(sizeof(clusterMsgPingExtForgottenNode));
|
||||
}
|
||||
@@ -2576,10 +2602,18 @@ uint32_t writePingExt(clusterMsg *hdr, int gossipcount) {
|
||||
totlen += getShardIdPingExtSize();
|
||||
extensions++;
|
||||
|
||||
/* Populate insternal secret */
|
||||
if (cursor != NULL) {
|
||||
clusterMsgPingExtInternalSecret *ext = preparePingExt(cursor, CLUSTERMSG_EXT_TYPE_INTERNALSECRET, getInternalSecretPingExtSize());
|
||||
memcpy(ext->internal_secret, server.cluster->internal_secret, CLUSTER_INTERNALSECRETLEN);
|
||||
|
||||
/* Move the write cursor */
|
||||
cursor = nextPingExt(cursor);
|
||||
}
|
||||
totlen += getInternalSecretPingExtSize();
|
||||
extensions++;
|
||||
|
||||
if (hdr != NULL) {
|
||||
if (extensions != 0) {
|
||||
hdr->mflags[0] |= CLUSTERMSG_FLAG0_EXT_DATA;
|
||||
}
|
||||
hdr->extensions = htons(extensions);
|
||||
}
|
||||
|
||||
@@ -2619,9 +2653,14 @@ void clusterProcessPingExtensions(clusterMsg *hdr, clusterLink *link) {
|
||||
} else if (type == CLUSTERMSG_EXT_TYPE_SHARDID) {
|
||||
clusterMsgPingExtShardId *shardid_ext = (clusterMsgPingExtShardId *) &(ext->ext[0].shard_id);
|
||||
ext_shardid = shardid_ext->shard_id;
|
||||
} else if (type == CLUSTERMSG_EXT_TYPE_INTERNALSECRET) {
|
||||
clusterMsgPingExtInternalSecret *internal_secret_ext = (clusterMsgPingExtInternalSecret *) &(ext->ext[0].internal_secret);
|
||||
if (memcmp(server.cluster->internal_secret, internal_secret_ext->internal_secret, CLUSTER_INTERNALSECRETLEN) > 0 ) {
|
||||
memcpy(server.cluster->internal_secret, internal_secret_ext->internal_secret, CLUSTER_INTERNALSECRETLEN);
|
||||
}
|
||||
} else {
|
||||
/* Unknown type, we will ignore it but log what happened. */
|
||||
serverLog(LL_WARNING, "Received unknown extension type %d", type);
|
||||
serverLog(LL_VERBOSE, "Received unknown extension type %d", type);
|
||||
}
|
||||
|
||||
/* We know this will be valid since we validated it ahead of time */
|
||||
@@ -2769,6 +2808,9 @@ int clusterProcessPacket(clusterLink *link) {
|
||||
}
|
||||
|
||||
sender = getNodeFromLinkAndMsg(link, hdr);
|
||||
if (sender && (hdr->mflags[0] & CLUSTERMSG_FLAG0_EXT_DATA)) {
|
||||
sender->flags |= CLUSTER_NODE_EXTENSIONS_SUPPORTED;
|
||||
}
|
||||
|
||||
/* Update the last time we saw any data from this node. We
|
||||
* use this in order to avoid detecting a timeout from a node that
|
||||
@@ -3534,6 +3576,8 @@ static void clusterBuildMessageHdr(clusterMsg *hdr, int type, size_t msglen) {
|
||||
/* Set the message flags. */
|
||||
if (clusterNodeIsMaster(myself) && server.cluster->mf_end)
|
||||
hdr->mflags[0] |= CLUSTERMSG_FLAG0_PAUSED;
|
||||
hdr->mflags[0] |= CLUSTERMSG_FLAG0_EXT_DATA; /* Always make other nodes know that
|
||||
* this node supports extension data. */
|
||||
|
||||
hdr->totlen = htonl(msglen);
|
||||
}
|
||||
@@ -3612,7 +3656,9 @@ void clusterSendPing(clusterLink *link, int type) {
|
||||
* to put inside the packet. */
|
||||
estlen = sizeof(clusterMsg) - sizeof(union clusterMsgData);
|
||||
estlen += (sizeof(clusterMsgDataGossip)*(wanted + pfail_wanted));
|
||||
estlen += writePingExt(NULL, 0);
|
||||
if (link->node && nodeSupportsExtensions(link->node)) {
|
||||
estlen += writePingExt(NULL, 0);
|
||||
}
|
||||
/* Note: clusterBuildMessageHdr() expects the buffer to be always at least
|
||||
* sizeof(clusterMsg) or more. */
|
||||
if (estlen < (int)sizeof(clusterMsg)) estlen = sizeof(clusterMsg);
|
||||
@@ -3682,7 +3728,9 @@ void clusterSendPing(clusterLink *link, int type) {
|
||||
|
||||
/* Compute the actual total length and send! */
|
||||
uint32_t totlen = 0;
|
||||
totlen += writePingExt(hdr, gossipcount);
|
||||
if (link->node && nodeSupportsExtensions(link->node)) {
|
||||
totlen += writePingExt(hdr, gossipcount);
|
||||
}
|
||||
totlen += sizeof(clusterMsg)-sizeof(union clusterMsgData);
|
||||
totlen += (sizeof(clusterMsgDataGossip)*gossipcount);
|
||||
serverAssert(gossipcount < USHRT_MAX);
|
||||
@@ -4559,7 +4607,7 @@ static int clusterNodeCronHandleReconnect(clusterNode *node, mstime_t handshake_
|
||||
|
||||
if (node->link == NULL) {
|
||||
clusterLink *link = createClusterLink(node);
|
||||
link->conn = connCreate(connTypeOfCluster());
|
||||
link->conn = connCreate(server.el, connTypeOfCluster());
|
||||
connSetPrivateData(link->conn, link);
|
||||
if (connConnect(link->conn, node->ip, node->cport, server.bind_source_addr,
|
||||
clusterLinkConnectHandler) == C_ERR) {
|
||||
@@ -5582,113 +5630,68 @@ void clusterUpdateSlots(client *c, unsigned char *slots, int del) {
|
||||
}
|
||||
}
|
||||
|
||||
/* Add detailed information of a node to the output buffer of the given client. */
|
||||
void addNodeDetailsToShardReply(client *c, clusterNode *node) {
|
||||
int reply_count = 0;
|
||||
void *node_replylen = addReplyDeferredLen(c);
|
||||
addReplyBulkCString(c, "id");
|
||||
addReplyBulkCBuffer(c, node->name, CLUSTER_NAMELEN);
|
||||
reply_count++;
|
||||
|
||||
if (node->tcp_port) {
|
||||
addReplyBulkCString(c, "port");
|
||||
addReplyLongLong(c, node->tcp_port);
|
||||
reply_count++;
|
||||
}
|
||||
|
||||
if (node->tls_port) {
|
||||
addReplyBulkCString(c, "tls-port");
|
||||
addReplyLongLong(c, node->tls_port);
|
||||
reply_count++;
|
||||
}
|
||||
|
||||
addReplyBulkCString(c, "ip");
|
||||
addReplyBulkCString(c, node->ip);
|
||||
reply_count++;
|
||||
|
||||
addReplyBulkCString(c, "endpoint");
|
||||
addReplyBulkCString(c, clusterNodePreferredEndpoint(node));
|
||||
reply_count++;
|
||||
|
||||
if (sdslen(node->hostname) != 0) {
|
||||
addReplyBulkCString(c, "hostname");
|
||||
addReplyBulkCBuffer(c, node->hostname, sdslen(node->hostname));
|
||||
reply_count++;
|
||||
}
|
||||
|
||||
long long node_offset;
|
||||
if (node->flags & CLUSTER_NODE_MYSELF) {
|
||||
node_offset = nodeIsSlave(node) ? replicationGetSlaveOffset() : server.master_repl_offset;
|
||||
} else {
|
||||
node_offset = node->repl_offset;
|
||||
}
|
||||
|
||||
addReplyBulkCString(c, "role");
|
||||
addReplyBulkCString(c, nodeIsSlave(node) ? "replica" : "master");
|
||||
reply_count++;
|
||||
|
||||
addReplyBulkCString(c, "replication-offset");
|
||||
addReplyLongLong(c, node_offset);
|
||||
reply_count++;
|
||||
|
||||
addReplyBulkCString(c, "health");
|
||||
const char *health_msg = NULL;
|
||||
if (nodeFailed(node)) {
|
||||
health_msg = "fail";
|
||||
} else if (nodeIsSlave(node) && node_offset == 0) {
|
||||
health_msg = "loading";
|
||||
} else {
|
||||
health_msg = "online";
|
||||
}
|
||||
addReplyBulkCString(c, health_msg);
|
||||
reply_count++;
|
||||
|
||||
setDeferredMapLen(c, node_replylen, reply_count);
|
||||
int clusterGetShardCount(void) {
|
||||
return dictSize(server.cluster->shards);
|
||||
}
|
||||
|
||||
/* Add the shard reply of a single shard based off the given primary node. */
|
||||
void addShardReplyForClusterShards(client *c, list *nodes) {
|
||||
serverAssert(listLength(nodes) > 0);
|
||||
clusterNode *n = listNodeValue(listFirst(nodes));
|
||||
addReplyMapLen(c, 2);
|
||||
addReplyBulkCString(c, "slots");
|
||||
|
||||
/* Use slot_info_pairs from the primary only */
|
||||
n = clusterNodeGetMaster(n);
|
||||
|
||||
if (n->slot_info_pairs != NULL) {
|
||||
serverAssert((n->slot_info_pairs_count % 2) == 0);
|
||||
addReplyArrayLen(c, n->slot_info_pairs_count);
|
||||
for (int i = 0; i < n->slot_info_pairs_count; i++)
|
||||
addReplyLongLong(c, (unsigned long)n->slot_info_pairs[i]);
|
||||
} else {
|
||||
/* If no slot info pair is provided, the node owns no slots */
|
||||
addReplyArrayLen(c, 0);
|
||||
}
|
||||
|
||||
addReplyBulkCString(c, "nodes");
|
||||
addReplyArrayLen(c, listLength(nodes));
|
||||
listIter li;
|
||||
listRewind(nodes, &li);
|
||||
for (listNode *ln = listNext(&li); ln != NULL; ln = listNext(&li)) {
|
||||
clusterNode *n = listNodeValue(ln);
|
||||
addNodeDetailsToShardReply(c, n);
|
||||
clusterFreeNodesSlotsInfo(n);
|
||||
}
|
||||
void *clusterGetShardIterator(void) {
|
||||
return dictGetSafeIterator(server.cluster->shards);
|
||||
}
|
||||
|
||||
/* Add to the output buffer of the given client, an array of slot (start, end)
|
||||
* pair owned by the shard, also the primary and set of replica(s) along with
|
||||
* information about each node. */
|
||||
void clusterCommandShards(client *c) {
|
||||
addReplyArrayLen(c, dictSize(server.cluster->shards));
|
||||
/* This call will add slot_info_pairs to all nodes */
|
||||
clusterGenNodesSlotsInfo(0);
|
||||
dictIterator *di = dictGetSafeIterator(server.cluster->shards);
|
||||
for(dictEntry *de = dictNext(di); de != NULL; de = dictNext(di)) {
|
||||
addShardReplyForClusterShards(c, dictGetVal(de));
|
||||
}
|
||||
dictReleaseIterator(di);
|
||||
void *clusterNextShardHandle(void *shard_iterator) {
|
||||
dictEntry *de = dictNext(shard_iterator);
|
||||
if(de == NULL) return NULL;
|
||||
return dictGetVal(de);
|
||||
}
|
||||
|
||||
void clusterFreeShardIterator(void *shard_iterator) {
|
||||
dictReleaseIterator(shard_iterator);
|
||||
}
|
||||
|
||||
int clusterNodeHasSlotInfo(clusterNode *n) {
|
||||
return n->slot_info_pairs != NULL;
|
||||
}
|
||||
|
||||
int clusterNodeSlotInfoCount(clusterNode *n) {
|
||||
return n->slot_info_pairs_count;
|
||||
}
|
||||
|
||||
uint16_t clusterNodeSlotInfoEntry(clusterNode *n, int idx) {
|
||||
return n->slot_info_pairs[idx];
|
||||
}
|
||||
|
||||
int clusterGetShardNodeCount(void *shard) {
|
||||
return listLength((list*)shard);
|
||||
}
|
||||
|
||||
void *clusterShardHandleGetNodeIterator(void *shard) {
|
||||
listIter *li = zmalloc(sizeof(listIter));
|
||||
listRewind((list*)shard, li);
|
||||
return li;
|
||||
}
|
||||
|
||||
void clusterShardNodeIteratorFree(void *node_iterator) {
|
||||
zfree(node_iterator);
|
||||
}
|
||||
|
||||
clusterNode *clusterShardNodeIteratorNext(void *node_iterator) {
|
||||
listNode *item = listNext((listIter*)node_iterator);
|
||||
if (item == NULL) return NULL;
|
||||
return listNodeValue(item);
|
||||
}
|
||||
|
||||
clusterNode *clusterShardNodeFirst(void *shard) {
|
||||
listNode *item = listFirst((list*)shard);
|
||||
if (item == NULL) return NULL;
|
||||
return listNodeValue(item);
|
||||
}
|
||||
|
||||
int clusterNodeTcpPort(clusterNode *node) {
|
||||
return node->tcp_port;
|
||||
}
|
||||
|
||||
int clusterNodeTlsPort(clusterNode *node) {
|
||||
return node->tls_port;
|
||||
}
|
||||
|
||||
sds genClusterInfoString(void) {
|
||||
|
||||
@@ -1,3 +1,16 @@
|
||||
/*
|
||||
* Copyright (c) 2009-Present, Redis Ltd.
|
||||
* All rights reserved.
|
||||
*
|
||||
* Copyright (c) 2024-present, Valkey contributors.
|
||||
* All rights reserved.
|
||||
*
|
||||
* Licensed under your choice of the Redis Source Available License 2.0
|
||||
* (RSALv2) or the Server Side Public License v1 (SSPLv1).
|
||||
*
|
||||
* Portions of this file are available under BSD3 terms; see REDISCONTRIBUTIONS for more information.
|
||||
*/
|
||||
|
||||
#ifndef CLUSTER_LEGACY_H
|
||||
#define CLUSTER_LEGACY_H
|
||||
|
||||
@@ -51,6 +64,7 @@ typedef struct clusterLink {
|
||||
#define CLUSTER_NODE_MEET 128 /* Send a MEET message to this node */
|
||||
#define CLUSTER_NODE_MIGRATE_TO 256 /* Master eligible for replica migration. */
|
||||
#define CLUSTER_NODE_NOFAILOVER 512 /* Slave will not try to failover. */
|
||||
#define CLUSTER_NODE_EXTENSIONS_SUPPORTED 1024 /* This node supports extensions. */
|
||||
#define CLUSTER_NODE_NULL_NAME "\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000\000"
|
||||
|
||||
#define nodeIsSlave(n) ((n)->flags & CLUSTER_NODE_SLAVE)
|
||||
@@ -59,6 +73,7 @@ typedef struct clusterLink {
|
||||
#define nodeTimedOut(n) ((n)->flags & CLUSTER_NODE_PFAIL)
|
||||
#define nodeFailed(n) ((n)->flags & CLUSTER_NODE_FAIL)
|
||||
#define nodeCantFailover(n) ((n)->flags & CLUSTER_NODE_NOFAILOVER)
|
||||
#define nodeSupportsExtensions(n) ((n)->flags & CLUSTER_NODE_EXTENSIONS_SUPPORTED)
|
||||
|
||||
/* This structure represent elements of node->fail_reports. */
|
||||
typedef struct clusterNodeFailReport {
|
||||
@@ -133,10 +148,12 @@ typedef enum {
|
||||
CLUSTERMSG_EXT_TYPE_HUMAN_NODENAME,
|
||||
CLUSTERMSG_EXT_TYPE_FORGOTTEN_NODE,
|
||||
CLUSTERMSG_EXT_TYPE_SHARDID,
|
||||
CLUSTERMSG_EXT_TYPE_INTERNALSECRET,
|
||||
} clusterMsgPingtypes;
|
||||
|
||||
/* Helper function for making sure extensions are eight byte aligned. */
|
||||
#define EIGHT_BYTE_ALIGN(size) ((((size) + 7) / 8) * 8)
|
||||
#define CLUSTER_INTERNALSECRETLEN 40 /* sha1 hex length */
|
||||
|
||||
typedef struct {
|
||||
char hostname[1]; /* The announced hostname, ends with \0. */
|
||||
@@ -157,6 +174,10 @@ typedef struct {
|
||||
char shard_id[CLUSTER_NAMELEN]; /* The shard_id, 40 bytes fixed. */
|
||||
} clusterMsgPingExtShardId;
|
||||
|
||||
typedef struct {
|
||||
char internal_secret[CLUSTER_INTERNALSECRETLEN]; /* Current shard internal secret */
|
||||
} clusterMsgPingExtInternalSecret;
|
||||
|
||||
typedef struct {
|
||||
uint32_t length; /* Total length of this extension message (including this header) */
|
||||
uint16_t type; /* Type of this extension message (see clusterMsgPingExtTypes) */
|
||||
@@ -166,6 +187,7 @@ typedef struct {
|
||||
clusterMsgPingExtHumanNodename human_nodename;
|
||||
clusterMsgPingExtForgottenNode forgotten_node;
|
||||
clusterMsgPingExtShardId shard_id;
|
||||
clusterMsgPingExtInternalSecret internal_secret;
|
||||
} ext[]; /* Actual extension information, formatted so that the data is 8
|
||||
* byte aligned, regardless of its content. */
|
||||
} clusterMsgPingExt;
|
||||
@@ -318,6 +340,7 @@ struct clusterState {
|
||||
clusterNode *migrating_slots_to[CLUSTER_SLOTS];
|
||||
clusterNode *importing_slots_from[CLUSTER_SLOTS];
|
||||
clusterNode *slots[CLUSTER_SLOTS];
|
||||
char internal_secret[CLUSTER_INTERNALSECRETLEN];
|
||||
/* The following fields are used to take the slave state on elections. */
|
||||
mstime_t failover_auth_time; /* Time of previous or next election. */
|
||||
int failover_auth_count; /* Number of votes received so far. */
|
||||
|
||||
+146
-10
@@ -1239,6 +1239,9 @@ commandHistory CLIENT_LIST_History[] = {
|
||||
{"6.2.0","Added `argv-mem`, `tot-mem`, `laddr` and `redir` fields and the optional `ID` filter."},
|
||||
{"7.0.0","Added `resp`, `multi-mem`, `rbs` and `rbp` fields."},
|
||||
{"7.0.3","Added `ssub` field."},
|
||||
{"7.2.0","Added `lib-name` and `lib-ver` fields."},
|
||||
{"7.4.0","Added `watch` field."},
|
||||
{"8.0.0","Added `io-thread` field."},
|
||||
};
|
||||
#endif
|
||||
|
||||
@@ -1546,7 +1549,7 @@ struct COMMAND_STRUCT CLIENT_Subcommands[] = {
|
||||
{MAKE_CMD("id","Returns the unique client ID of the connection.","O(1)","5.0.0",CMD_DOC_NONE,NULL,NULL,"connection",COMMAND_GROUP_CONNECTION,CLIENT_ID_History,0,CLIENT_ID_Tips,0,clientCommand,2,CMD_NOSCRIPT|CMD_LOADING|CMD_STALE|CMD_SENTINEL,ACL_CATEGORY_CONNECTION,CLIENT_ID_Keyspecs,0,NULL,0)},
|
||||
{MAKE_CMD("info","Returns information about the connection.","O(1)","6.2.0",CMD_DOC_NONE,NULL,NULL,"connection",COMMAND_GROUP_CONNECTION,CLIENT_INFO_History,0,CLIENT_INFO_Tips,1,clientCommand,2,CMD_NOSCRIPT|CMD_LOADING|CMD_STALE|CMD_SENTINEL,ACL_CATEGORY_CONNECTION,CLIENT_INFO_Keyspecs,0,NULL,0)},
|
||||
{MAKE_CMD("kill","Terminates open connections.","O(N) where N is the number of client connections","2.4.0",CMD_DOC_NONE,NULL,NULL,"connection",COMMAND_GROUP_CONNECTION,CLIENT_KILL_History,6,CLIENT_KILL_Tips,0,clientCommand,-3,CMD_ADMIN|CMD_NOSCRIPT|CMD_LOADING|CMD_STALE|CMD_SENTINEL,ACL_CATEGORY_CONNECTION,CLIENT_KILL_Keyspecs,0,NULL,1),.args=CLIENT_KILL_Args},
|
||||
{MAKE_CMD("list","Lists open connections.","O(N) where N is the number of client connections","2.4.0",CMD_DOC_NONE,NULL,NULL,"connection",COMMAND_GROUP_CONNECTION,CLIENT_LIST_History,6,CLIENT_LIST_Tips,1,clientCommand,-2,CMD_ADMIN|CMD_NOSCRIPT|CMD_LOADING|CMD_STALE|CMD_SENTINEL,ACL_CATEGORY_CONNECTION,CLIENT_LIST_Keyspecs,0,NULL,2),.args=CLIENT_LIST_Args},
|
||||
{MAKE_CMD("list","Lists open connections.","O(N) where N is the number of client connections","2.4.0",CMD_DOC_NONE,NULL,NULL,"connection",COMMAND_GROUP_CONNECTION,CLIENT_LIST_History,9,CLIENT_LIST_Tips,1,clientCommand,-2,CMD_ADMIN|CMD_NOSCRIPT|CMD_LOADING|CMD_STALE|CMD_SENTINEL,ACL_CATEGORY_CONNECTION,CLIENT_LIST_Keyspecs,0,NULL,2),.args=CLIENT_LIST_Args},
|
||||
{MAKE_CMD("no-evict","Sets the client eviction mode of the connection.","O(1)","7.0.0",CMD_DOC_NONE,NULL,NULL,"connection",COMMAND_GROUP_CONNECTION,CLIENT_NO_EVICT_History,0,CLIENT_NO_EVICT_Tips,0,clientCommand,3,CMD_ADMIN|CMD_NOSCRIPT|CMD_LOADING|CMD_STALE|CMD_SENTINEL,ACL_CATEGORY_CONNECTION,CLIENT_NO_EVICT_Keyspecs,0,NULL,1),.args=CLIENT_NO_EVICT_Args},
|
||||
{MAKE_CMD("no-touch","Controls whether commands sent by the client affect the LRU/LFU of accessed keys.","O(1)","7.2.0",CMD_DOC_NONE,NULL,NULL,"connection",COMMAND_GROUP_CONNECTION,CLIENT_NO_TOUCH_History,0,CLIENT_NO_TOUCH_Tips,0,clientCommand,3,CMD_NOSCRIPT|CMD_LOADING|CMD_STALE,ACL_CATEGORY_CONNECTION,CLIENT_NO_TOUCH_Keyspecs,0,NULL,1),.args=CLIENT_NO_TOUCH_Args},
|
||||
{MAKE_CMD("pause","Suspends commands processing.","O(1)","3.0.0",CMD_DOC_NONE,NULL,NULL,"connection",COMMAND_GROUP_CONNECTION,CLIENT_PAUSE_History,1,CLIENT_PAUSE_Tips,0,clientCommand,-3,CMD_ADMIN|CMD_NOSCRIPT|CMD_LOADING|CMD_STALE|CMD_SENTINEL,ACL_CATEGORY_CONNECTION,CLIENT_PAUSE_Keyspecs,0,NULL,2),.args=CLIENT_PAUSE_Args},
|
||||
@@ -3467,6 +3470,78 @@ struct COMMAND_ARG HGETALL_Args[] = {
|
||||
{MAKE_ARG("key",ARG_TYPE_KEY,0,NULL,NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
};
|
||||
|
||||
/********** HGETDEL ********************/
|
||||
|
||||
#ifndef SKIP_CMD_HISTORY_TABLE
|
||||
/* HGETDEL history */
|
||||
#define HGETDEL_History NULL
|
||||
#endif
|
||||
|
||||
#ifndef SKIP_CMD_TIPS_TABLE
|
||||
/* HGETDEL tips */
|
||||
#define HGETDEL_Tips NULL
|
||||
#endif
|
||||
|
||||
#ifndef SKIP_CMD_KEY_SPECS_TABLE
|
||||
/* HGETDEL key specs */
|
||||
keySpec HGETDEL_Keyspecs[1] = {
|
||||
{NULL,CMD_KEY_RW|CMD_KEY_ACCESS|CMD_KEY_DELETE,KSPEC_BS_INDEX,.bs.index={1},KSPEC_FK_RANGE,.fk.range={0,1,0}}
|
||||
};
|
||||
#endif
|
||||
|
||||
/* HGETDEL fields argument table */
|
||||
struct COMMAND_ARG HGETDEL_fields_Subargs[] = {
|
||||
{MAKE_ARG("numfields",ARG_TYPE_INTEGER,-1,NULL,NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("field",ARG_TYPE_STRING,-1,NULL,NULL,NULL,CMD_ARG_MULTIPLE,0,NULL)},
|
||||
};
|
||||
|
||||
/* HGETDEL argument table */
|
||||
struct COMMAND_ARG HGETDEL_Args[] = {
|
||||
{MAKE_ARG("key",ARG_TYPE_KEY,0,NULL,NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("fields",ARG_TYPE_BLOCK,-1,"FIELDS",NULL,NULL,CMD_ARG_NONE,2,NULL),.subargs=HGETDEL_fields_Subargs},
|
||||
};
|
||||
|
||||
/********** HGETEX ********************/
|
||||
|
||||
#ifndef SKIP_CMD_HISTORY_TABLE
|
||||
/* HGETEX history */
|
||||
#define HGETEX_History NULL
|
||||
#endif
|
||||
|
||||
#ifndef SKIP_CMD_TIPS_TABLE
|
||||
/* HGETEX tips */
|
||||
#define HGETEX_Tips NULL
|
||||
#endif
|
||||
|
||||
#ifndef SKIP_CMD_KEY_SPECS_TABLE
|
||||
/* HGETEX key specs */
|
||||
keySpec HGETEX_Keyspecs[1] = {
|
||||
{"RW and UPDATE because it changes the TTL",CMD_KEY_RW|CMD_KEY_ACCESS|CMD_KEY_UPDATE,KSPEC_BS_INDEX,.bs.index={1},KSPEC_FK_RANGE,.fk.range={0,1,0}}
|
||||
};
|
||||
#endif
|
||||
|
||||
/* HGETEX expiration argument table */
|
||||
struct COMMAND_ARG HGETEX_expiration_Subargs[] = {
|
||||
{MAKE_ARG("seconds",ARG_TYPE_INTEGER,-1,"EX",NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("milliseconds",ARG_TYPE_INTEGER,-1,"PX",NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("unix-time-seconds",ARG_TYPE_UNIX_TIME,-1,"EXAT",NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("unix-time-milliseconds",ARG_TYPE_UNIX_TIME,-1,"PXAT",NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("persist",ARG_TYPE_PURE_TOKEN,-1,"PERSIST",NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
};
|
||||
|
||||
/* HGETEX fields argument table */
|
||||
struct COMMAND_ARG HGETEX_fields_Subargs[] = {
|
||||
{MAKE_ARG("numfields",ARG_TYPE_INTEGER,-1,NULL,NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("field",ARG_TYPE_STRING,-1,NULL,NULL,NULL,CMD_ARG_MULTIPLE,0,NULL)},
|
||||
};
|
||||
|
||||
/* HGETEX argument table */
|
||||
struct COMMAND_ARG HGETEX_Args[] = {
|
||||
{MAKE_ARG("key",ARG_TYPE_KEY,0,NULL,NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("expiration",ARG_TYPE_ONEOF,-1,NULL,NULL,NULL,CMD_ARG_OPTIONAL,5,NULL),.subargs=HGETEX_expiration_Subargs},
|
||||
{MAKE_ARG("fields",ARG_TYPE_BLOCK,-1,"FIELDS",NULL,NULL,CMD_ARG_NONE,2,NULL),.subargs=HGETEX_fields_Subargs},
|
||||
};
|
||||
|
||||
/********** HINCRBY ********************/
|
||||
|
||||
#ifndef SKIP_CMD_HISTORY_TABLE
|
||||
@@ -3778,7 +3853,9 @@ struct COMMAND_ARG HPEXPIRETIME_Args[] = {
|
||||
|
||||
#ifndef SKIP_CMD_TIPS_TABLE
|
||||
/* HPTTL tips */
|
||||
#define HPTTL_Tips NULL
|
||||
const char *HPTTL_Tips[] = {
|
||||
"nondeterministic_output",
|
||||
};
|
||||
#endif
|
||||
|
||||
#ifndef SKIP_CMD_KEY_SPECS_TABLE
|
||||
@@ -3896,6 +3973,60 @@ struct COMMAND_ARG HSET_Args[] = {
|
||||
{MAKE_ARG("data",ARG_TYPE_BLOCK,-1,NULL,NULL,NULL,CMD_ARG_MULTIPLE,2,NULL),.subargs=HSET_data_Subargs},
|
||||
};
|
||||
|
||||
/********** HSETEX ********************/
|
||||
|
||||
#ifndef SKIP_CMD_HISTORY_TABLE
|
||||
/* HSETEX history */
|
||||
#define HSETEX_History NULL
|
||||
#endif
|
||||
|
||||
#ifndef SKIP_CMD_TIPS_TABLE
|
||||
/* HSETEX tips */
|
||||
#define HSETEX_Tips NULL
|
||||
#endif
|
||||
|
||||
#ifndef SKIP_CMD_KEY_SPECS_TABLE
|
||||
/* HSETEX key specs */
|
||||
keySpec HSETEX_Keyspecs[1] = {
|
||||
{NULL,CMD_KEY_RW|CMD_KEY_UPDATE,KSPEC_BS_INDEX,.bs.index={1},KSPEC_FK_RANGE,.fk.range={0,1,0}}
|
||||
};
|
||||
#endif
|
||||
|
||||
/* HSETEX condition argument table */
|
||||
struct COMMAND_ARG HSETEX_condition_Subargs[] = {
|
||||
{MAKE_ARG("fnx",ARG_TYPE_PURE_TOKEN,-1,"FNX",NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("fxx",ARG_TYPE_PURE_TOKEN,-1,"FXX",NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
};
|
||||
|
||||
/* HSETEX expiration argument table */
|
||||
struct COMMAND_ARG HSETEX_expiration_Subargs[] = {
|
||||
{MAKE_ARG("seconds",ARG_TYPE_INTEGER,-1,"EX",NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("milliseconds",ARG_TYPE_INTEGER,-1,"PX",NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("unix-time-seconds",ARG_TYPE_UNIX_TIME,-1,"EXAT",NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("unix-time-milliseconds",ARG_TYPE_UNIX_TIME,-1,"PXAT",NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("keepttl",ARG_TYPE_PURE_TOKEN,-1,"KEEPTTL",NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
};
|
||||
|
||||
/* HSETEX fields data argument table */
|
||||
struct COMMAND_ARG HSETEX_fields_data_Subargs[] = {
|
||||
{MAKE_ARG("field",ARG_TYPE_STRING,-1,NULL,NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("value",ARG_TYPE_STRING,-1,NULL,NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
};
|
||||
|
||||
/* HSETEX fields argument table */
|
||||
struct COMMAND_ARG HSETEX_fields_Subargs[] = {
|
||||
{MAKE_ARG("numfields",ARG_TYPE_INTEGER,-1,NULL,NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("data",ARG_TYPE_BLOCK,-1,NULL,NULL,NULL,CMD_ARG_MULTIPLE,2,NULL),.subargs=HSETEX_fields_data_Subargs},
|
||||
};
|
||||
|
||||
/* HSETEX argument table */
|
||||
struct COMMAND_ARG HSETEX_Args[] = {
|
||||
{MAKE_ARG("key",ARG_TYPE_KEY,0,NULL,NULL,NULL,CMD_ARG_NONE,0,NULL)},
|
||||
{MAKE_ARG("condition",ARG_TYPE_ONEOF,-1,NULL,NULL,NULL,CMD_ARG_OPTIONAL,2,NULL),.subargs=HSETEX_condition_Subargs},
|
||||
{MAKE_ARG("expiration",ARG_TYPE_ONEOF,-1,NULL,NULL,NULL,CMD_ARG_OPTIONAL,5,NULL),.subargs=HSETEX_expiration_Subargs},
|
||||
{MAKE_ARG("fields",ARG_TYPE_BLOCK,-1,"FIELDS",NULL,NULL,CMD_ARG_NONE,2,NULL),.subargs=HSETEX_fields_Subargs},
|
||||
};
|
||||
|
||||
/********** HSETNX ********************/
|
||||
|
||||
#ifndef SKIP_CMD_HISTORY_TABLE
|
||||
@@ -3956,7 +4087,9 @@ struct COMMAND_ARG HSTRLEN_Args[] = {
|
||||
|
||||
#ifndef SKIP_CMD_TIPS_TABLE
|
||||
/* HTTL tips */
|
||||
#define HTTL_Tips NULL
|
||||
const char *HTTL_Tips[] = {
|
||||
"nondeterministic_output",
|
||||
};
|
||||
#endif
|
||||
|
||||
#ifndef SKIP_CMD_KEY_SPECS_TABLE
|
||||
@@ -11029,11 +11162,13 @@ struct COMMAND_STRUCT redisCommandTable[] = {
|
||||
/* hash */
|
||||
{MAKE_CMD("hdel","Deletes one or more fields and their values from a hash. Deletes the hash if no fields remain.","O(N) where N is the number of fields to be removed.","2.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HDEL_History,1,HDEL_Tips,0,hdelCommand,-3,CMD_WRITE|CMD_FAST,ACL_CATEGORY_HASH,HDEL_Keyspecs,1,NULL,2),.args=HDEL_Args},
|
||||
{MAKE_CMD("hexists","Determines whether a field exists in a hash.","O(1)","2.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HEXISTS_History,0,HEXISTS_Tips,0,hexistsCommand,3,CMD_READONLY|CMD_FAST,ACL_CATEGORY_HASH,HEXISTS_Keyspecs,1,NULL,2),.args=HEXISTS_Args},
|
||||
{MAKE_CMD("hexpire","Set expiry for hash field using relative time to expire (seconds)","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HEXPIRE_History,0,HEXPIRE_Tips,0,hexpireCommand,-6,CMD_WRITE|CMD_DENYOOM|CMD_FAST,ACL_CATEGORY_HASH,HEXPIRE_Keyspecs,1,NULL,4),.args=HEXPIRE_Args},
|
||||
{MAKE_CMD("hexpireat","Set expiry for hash field using an absolute Unix timestamp (seconds)","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HEXPIREAT_History,0,HEXPIREAT_Tips,0,hexpireatCommand,-6,CMD_WRITE|CMD_DENYOOM|CMD_FAST,ACL_CATEGORY_HASH,HEXPIREAT_Keyspecs,1,NULL,4),.args=HEXPIREAT_Args},
|
||||
{MAKE_CMD("hexpire","Set expiry for hash field using relative time to expire (seconds)","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HEXPIRE_History,0,HEXPIRE_Tips,0,hexpireCommand,-6,CMD_WRITE|CMD_FAST,ACL_CATEGORY_HASH,HEXPIRE_Keyspecs,1,NULL,4),.args=HEXPIRE_Args},
|
||||
{MAKE_CMD("hexpireat","Set expiry for hash field using an absolute Unix timestamp (seconds)","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HEXPIREAT_History,0,HEXPIREAT_Tips,0,hexpireatCommand,-6,CMD_WRITE|CMD_FAST,ACL_CATEGORY_HASH,HEXPIREAT_Keyspecs,1,NULL,4),.args=HEXPIREAT_Args},
|
||||
{MAKE_CMD("hexpiretime","Returns the expiration time of a hash field as a Unix timestamp, in seconds.","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HEXPIRETIME_History,0,HEXPIRETIME_Tips,0,hexpiretimeCommand,-5,CMD_READONLY|CMD_FAST,ACL_CATEGORY_HASH,HEXPIRETIME_Keyspecs,1,NULL,2),.args=HEXPIRETIME_Args},
|
||||
{MAKE_CMD("hget","Returns the value of a field in a hash.","O(1)","2.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HGET_History,0,HGET_Tips,0,hgetCommand,3,CMD_READONLY|CMD_FAST,ACL_CATEGORY_HASH,HGET_Keyspecs,1,NULL,2),.args=HGET_Args},
|
||||
{MAKE_CMD("hgetall","Returns all fields and values in a hash.","O(N) where N is the size of the hash.","2.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HGETALL_History,0,HGETALL_Tips,1,hgetallCommand,2,CMD_READONLY,ACL_CATEGORY_HASH,HGETALL_Keyspecs,1,NULL,1),.args=HGETALL_Args},
|
||||
{MAKE_CMD("hgetdel","Returns the value of a field and deletes it from the hash.","O(N) where N is the number of specified fields","8.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HGETDEL_History,0,HGETDEL_Tips,0,hgetdelCommand,-5,CMD_WRITE|CMD_FAST,ACL_CATEGORY_HASH,HGETDEL_Keyspecs,1,NULL,2),.args=HGETDEL_Args},
|
||||
{MAKE_CMD("hgetex","Get the value of one or more fields of a given hash key, and optionally set their expiration.","O(N) where N is the number of specified fields","8.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HGETEX_History,0,HGETEX_Tips,0,hgetexCommand,-5,CMD_WRITE|CMD_FAST,ACL_CATEGORY_HASH,HGETEX_Keyspecs,1,NULL,3),.args=HGETEX_Args},
|
||||
{MAKE_CMD("hincrby","Increments the integer value of a field in a hash by a number. Uses 0 as initial value if the field doesn't exist.","O(1)","2.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HINCRBY_History,0,HINCRBY_Tips,0,hincrbyCommand,4,CMD_WRITE|CMD_DENYOOM|CMD_FAST,ACL_CATEGORY_HASH,HINCRBY_Keyspecs,1,NULL,3),.args=HINCRBY_Args},
|
||||
{MAKE_CMD("hincrbyfloat","Increments the floating point value of a field by a number. Uses 0 as initial value if the field doesn't exist.","O(1)","2.6.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HINCRBYFLOAT_History,0,HINCRBYFLOAT_Tips,0,hincrbyfloatCommand,4,CMD_WRITE|CMD_DENYOOM|CMD_FAST,ACL_CATEGORY_HASH,HINCRBYFLOAT_Keyspecs,1,NULL,3),.args=HINCRBYFLOAT_Args},
|
||||
{MAKE_CMD("hkeys","Returns all fields in a hash.","O(N) where N is the size of the hash.","2.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HKEYS_History,0,HKEYS_Tips,1,hkeysCommand,2,CMD_READONLY,ACL_CATEGORY_HASH,HKEYS_Keyspecs,1,NULL,1),.args=HKEYS_Args},
|
||||
@@ -11041,16 +11176,17 @@ struct COMMAND_STRUCT redisCommandTable[] = {
|
||||
{MAKE_CMD("hmget","Returns the values of all fields in a hash.","O(N) where N is the number of fields being requested.","2.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HMGET_History,0,HMGET_Tips,0,hmgetCommand,-3,CMD_READONLY|CMD_FAST,ACL_CATEGORY_HASH,HMGET_Keyspecs,1,NULL,2),.args=HMGET_Args},
|
||||
{MAKE_CMD("hmset","Sets the values of multiple fields.","O(N) where N is the number of fields being set.","2.0.0",CMD_DOC_DEPRECATED,"`HSET` with multiple field-value pairs","4.0.0","hash",COMMAND_GROUP_HASH,HMSET_History,0,HMSET_Tips,0,hsetCommand,-4,CMD_WRITE|CMD_DENYOOM|CMD_FAST,ACL_CATEGORY_HASH,HMSET_Keyspecs,1,NULL,2),.args=HMSET_Args},
|
||||
{MAKE_CMD("hpersist","Removes the expiration time for each specified field","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HPERSIST_History,0,HPERSIST_Tips,0,hpersistCommand,-5,CMD_WRITE|CMD_FAST,ACL_CATEGORY_HASH,HPERSIST_Keyspecs,1,NULL,2),.args=HPERSIST_Args},
|
||||
{MAKE_CMD("hpexpire","Set expiry for hash field using relative time to expire (milliseconds)","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HPEXPIRE_History,0,HPEXPIRE_Tips,0,hpexpireCommand,-6,CMD_WRITE|CMD_DENYOOM|CMD_FAST,ACL_CATEGORY_HASH,HPEXPIRE_Keyspecs,1,NULL,4),.args=HPEXPIRE_Args},
|
||||
{MAKE_CMD("hpexpireat","Set expiry for hash field using an absolute Unix timestamp (milliseconds)","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HPEXPIREAT_History,0,HPEXPIREAT_Tips,0,hpexpireatCommand,-6,CMD_WRITE|CMD_DENYOOM|CMD_FAST,ACL_CATEGORY_HASH,HPEXPIREAT_Keyspecs,1,NULL,4),.args=HPEXPIREAT_Args},
|
||||
{MAKE_CMD("hpexpire","Set expiry for hash field using relative time to expire (milliseconds)","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HPEXPIRE_History,0,HPEXPIRE_Tips,0,hpexpireCommand,-6,CMD_WRITE|CMD_FAST,ACL_CATEGORY_HASH,HPEXPIRE_Keyspecs,1,NULL,4),.args=HPEXPIRE_Args},
|
||||
{MAKE_CMD("hpexpireat","Set expiry for hash field using an absolute Unix timestamp (milliseconds)","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HPEXPIREAT_History,0,HPEXPIREAT_Tips,0,hpexpireatCommand,-6,CMD_WRITE|CMD_FAST,ACL_CATEGORY_HASH,HPEXPIREAT_Keyspecs,1,NULL,4),.args=HPEXPIREAT_Args},
|
||||
{MAKE_CMD("hpexpiretime","Returns the expiration time of a hash field as a Unix timestamp, in msec.","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HPEXPIRETIME_History,0,HPEXPIRETIME_Tips,0,hpexpiretimeCommand,-5,CMD_READONLY|CMD_FAST,ACL_CATEGORY_HASH,HPEXPIRETIME_Keyspecs,1,NULL,2),.args=HPEXPIRETIME_Args},
|
||||
{MAKE_CMD("hpttl","Returns the TTL in milliseconds of a hash field.","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HPTTL_History,0,HPTTL_Tips,0,hpttlCommand,-5,CMD_READONLY|CMD_FAST,ACL_CATEGORY_HASH,HPTTL_Keyspecs,1,NULL,2),.args=HPTTL_Args},
|
||||
{MAKE_CMD("hpttl","Returns the TTL in milliseconds of a hash field.","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HPTTL_History,0,HPTTL_Tips,1,hpttlCommand,-5,CMD_READONLY|CMD_FAST,ACL_CATEGORY_HASH,HPTTL_Keyspecs,1,NULL,2),.args=HPTTL_Args},
|
||||
{MAKE_CMD("hrandfield","Returns one or more random fields from a hash.","O(N) where N is the number of fields returned","6.2.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HRANDFIELD_History,0,HRANDFIELD_Tips,1,hrandfieldCommand,-2,CMD_READONLY,ACL_CATEGORY_HASH,HRANDFIELD_Keyspecs,1,NULL,2),.args=HRANDFIELD_Args},
|
||||
{MAKE_CMD("hscan","Iterates over fields and values of a hash.","O(1) for every call. O(N) for a complete iteration, including enough command calls for the cursor to return back to 0. N is the number of elements inside the collection.","2.8.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HSCAN_History,0,HSCAN_Tips,1,hscanCommand,-3,CMD_READONLY,ACL_CATEGORY_HASH,HSCAN_Keyspecs,1,NULL,5),.args=HSCAN_Args},
|
||||
{MAKE_CMD("hset","Creates or modifies the value of a field in a hash.","O(1) for each field/value pair added, so O(N) to add N field/value pairs when the command is called with multiple field/value pairs.","2.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HSET_History,1,HSET_Tips,0,hsetCommand,-4,CMD_WRITE|CMD_DENYOOM|CMD_FAST,ACL_CATEGORY_HASH,HSET_Keyspecs,1,NULL,2),.args=HSET_Args},
|
||||
{MAKE_CMD("hsetex","Set the value of one or more fields of a given hash key, and optionally set their expiration.","O(N) where N is the number of fields being set.","8.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HSETEX_History,0,HSETEX_Tips,0,hsetexCommand,-6,CMD_WRITE|CMD_DENYOOM|CMD_FAST,ACL_CATEGORY_HASH,HSETEX_Keyspecs,1,NULL,4),.args=HSETEX_Args},
|
||||
{MAKE_CMD("hsetnx","Sets the value of a field in a hash only when the field doesn't exist.","O(1)","2.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HSETNX_History,0,HSETNX_Tips,0,hsetnxCommand,4,CMD_WRITE|CMD_DENYOOM|CMD_FAST,ACL_CATEGORY_HASH,HSETNX_Keyspecs,1,NULL,3),.args=HSETNX_Args},
|
||||
{MAKE_CMD("hstrlen","Returns the length of the value of a field.","O(1)","3.2.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HSTRLEN_History,0,HSTRLEN_Tips,0,hstrlenCommand,3,CMD_READONLY|CMD_FAST,ACL_CATEGORY_HASH,HSTRLEN_Keyspecs,1,NULL,2),.args=HSTRLEN_Args},
|
||||
{MAKE_CMD("httl","Returns the TTL in seconds of a hash field.","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HTTL_History,0,HTTL_Tips,0,httlCommand,-5,CMD_READONLY|CMD_FAST,ACL_CATEGORY_HASH,HTTL_Keyspecs,1,NULL,2),.args=HTTL_Args},
|
||||
{MAKE_CMD("httl","Returns the TTL in seconds of a hash field.","O(N) where N is the number of specified fields","7.4.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HTTL_History,0,HTTL_Tips,1,httlCommand,-5,CMD_READONLY|CMD_FAST,ACL_CATEGORY_HASH,HTTL_Keyspecs,1,NULL,2),.args=HTTL_Args},
|
||||
{MAKE_CMD("hvals","Returns all values in a hash.","O(N) where N is the size of the hash.","2.0.0",CMD_DOC_NONE,NULL,NULL,"hash",COMMAND_GROUP_HASH,HVALS_History,0,HVALS_Tips,1,hvalsCommand,2,CMD_READONLY,ACL_CATEGORY_HASH,HVALS_Keyspecs,1,NULL,1),.args=HVALS_Args},
|
||||
/* hyperloglog */
|
||||
{MAKE_CMD("pfadd","Adds elements to a HyperLogLog key. Creates the key if it doesn't exist.","O(1) to add every element.","2.8.9",CMD_DOC_NONE,NULL,NULL,"hyperloglog",COMMAND_GROUP_HYPERLOGLOG,PFADD_History,0,PFADD_Tips,0,pfaddCommand,-2,CMD_WRITE|CMD_DENYOOM|CMD_FAST,ACL_CATEGORY_HYPERLOGLOG,PFADD_Keyspecs,1,NULL,2),.args=PFADD_Args},
|
||||
@@ -11141,7 +11277,7 @@ struct COMMAND_STRUCT redisCommandTable[] = {
|
||||
{MAKE_CMD("sintercard","Returns the number of members of the intersect of multiple sets.","O(N*M) worst case where N is the cardinality of the smallest set and M is the number of sets.","7.0.0",CMD_DOC_NONE,NULL,NULL,"set",COMMAND_GROUP_SET,SINTERCARD_History,0,SINTERCARD_Tips,0,sinterCardCommand,-3,CMD_READONLY,ACL_CATEGORY_SET,SINTERCARD_Keyspecs,1,sintercardGetKeys,3),.args=SINTERCARD_Args},
|
||||
{MAKE_CMD("sinterstore","Stores the intersect of multiple sets in a key.","O(N*M) worst case where N is the cardinality of the smallest set and M is the number of sets.","1.0.0",CMD_DOC_NONE,NULL,NULL,"set",COMMAND_GROUP_SET,SINTERSTORE_History,0,SINTERSTORE_Tips,0,sinterstoreCommand,-3,CMD_WRITE|CMD_DENYOOM,ACL_CATEGORY_SET,SINTERSTORE_Keyspecs,2,NULL,2),.args=SINTERSTORE_Args},
|
||||
{MAKE_CMD("sismember","Determines whether a member belongs to a set.","O(1)","1.0.0",CMD_DOC_NONE,NULL,NULL,"set",COMMAND_GROUP_SET,SISMEMBER_History,0,SISMEMBER_Tips,0,sismemberCommand,3,CMD_READONLY|CMD_FAST,ACL_CATEGORY_SET,SISMEMBER_Keyspecs,1,NULL,2),.args=SISMEMBER_Args},
|
||||
{MAKE_CMD("smembers","Returns all members of a set.","O(N) where N is the set cardinality.","1.0.0",CMD_DOC_NONE,NULL,NULL,"set",COMMAND_GROUP_SET,SMEMBERS_History,0,SMEMBERS_Tips,1,sinterCommand,2,CMD_READONLY,ACL_CATEGORY_SET,SMEMBERS_Keyspecs,1,NULL,1),.args=SMEMBERS_Args},
|
||||
{MAKE_CMD("smembers","Returns all members of a set.","O(N) where N is the set cardinality.","1.0.0",CMD_DOC_NONE,NULL,NULL,"set",COMMAND_GROUP_SET,SMEMBERS_History,0,SMEMBERS_Tips,1,smembersCommand,2,CMD_READONLY,ACL_CATEGORY_SET,SMEMBERS_Keyspecs,1,NULL,1),.args=SMEMBERS_Args},
|
||||
{MAKE_CMD("smismember","Determines whether multiple members belong to a set.","O(N) where N is the number of elements being checked for membership","6.2.0",CMD_DOC_NONE,NULL,NULL,"set",COMMAND_GROUP_SET,SMISMEMBER_History,0,SMISMEMBER_Tips,0,smismemberCommand,-3,CMD_READONLY|CMD_FAST,ACL_CATEGORY_SET,SMISMEMBER_Keyspecs,1,NULL,2),.args=SMISMEMBER_Args},
|
||||
{MAKE_CMD("smove","Moves a member from one set to another.","O(1)","1.0.0",CMD_DOC_NONE,NULL,NULL,"set",COMMAND_GROUP_SET,SMOVE_History,0,SMOVE_Tips,0,smoveCommand,4,CMD_WRITE|CMD_FAST,ACL_CATEGORY_SET,SMOVE_Keyspecs,2,NULL,3),.args=SMOVE_Args},
|
||||
{MAKE_CMD("spop","Returns one or more random members from a set after removing them. Deletes the set if the last member was popped.","Without the count argument O(1), otherwise O(N) where N is the value of the passed count.","1.0.0",CMD_DOC_NONE,NULL,NULL,"set",COMMAND_GROUP_SET,SPOP_History,1,SPOP_Tips,1,spopCommand,-2,CMD_WRITE|CMD_FAST,ACL_CATEGORY_SET,SPOP_Keyspecs,1,NULL,2),.args=SPOP_Args},
|
||||
|
||||
@@ -31,6 +31,18 @@
|
||||
[
|
||||
"7.0.3",
|
||||
"Added `ssub` field."
|
||||
],
|
||||
[
|
||||
"7.2.0",
|
||||
"Added `lib-name` and `lib-ver` fields."
|
||||
],
|
||||
[
|
||||
"7.4.0",
|
||||
"Added `watch` field."
|
||||
],
|
||||
[
|
||||
"8.0.0",
|
||||
"Added `io-thread` field."
|
||||
]
|
||||
],
|
||||
"command_flags": [
|
||||
|
||||
@@ -9,7 +9,6 @@
|
||||
"history": [],
|
||||
"command_flags": [
|
||||
"WRITE",
|
||||
"DENYOOM",
|
||||
"FAST"
|
||||
],
|
||||
"acl_categories": [
|
||||
|
||||
@@ -9,7 +9,6 @@
|
||||
"history": [],
|
||||
"command_flags": [
|
||||
"WRITE",
|
||||
"DENYOOM",
|
||||
"FAST"
|
||||
],
|
||||
"acl_categories": [
|
||||
|
||||
@@ -0,0 +1,78 @@
|
||||
{
|
||||
"HGETDEL": {
|
||||
"summary": "Returns the value of a field and deletes it from the hash.",
|
||||
"complexity": "O(N) where N is the number of specified fields",
|
||||
"group": "hash",
|
||||
"since": "8.0.0",
|
||||
"arity": -5,
|
||||
"function": "hgetdelCommand",
|
||||
"history": [],
|
||||
"command_flags": [
|
||||
"WRITE",
|
||||
"FAST"
|
||||
],
|
||||
"acl_categories": [
|
||||
"HASH"
|
||||
],
|
||||
"key_specs": [
|
||||
{
|
||||
"flags": [
|
||||
"RW",
|
||||
"ACCESS",
|
||||
"DELETE"
|
||||
],
|
||||
"begin_search": {
|
||||
"index": {
|
||||
"pos": 1
|
||||
}
|
||||
},
|
||||
"find_keys": {
|
||||
"range": {
|
||||
"lastkey": 0,
|
||||
"step": 1,
|
||||
"limit": 0
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"reply_schema": {
|
||||
"description": "List of values associated with the given fields, in the same order as they are requested.",
|
||||
"type": "array",
|
||||
"minItems": 1,
|
||||
"items": {
|
||||
"oneOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
]
|
||||
}
|
||||
},
|
||||
"arguments": [
|
||||
{
|
||||
"name": "key",
|
||||
"type": "key",
|
||||
"key_spec_index": 0
|
||||
},
|
||||
{
|
||||
"name": "fields",
|
||||
"token": "FIELDS",
|
||||
"type": "block",
|
||||
"arguments": [
|
||||
{
|
||||
"name": "numfields",
|
||||
"type": "integer"
|
||||
},
|
||||
{
|
||||
"name": "field",
|
||||
"type": "string",
|
||||
"multiple": true
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,111 @@
|
||||
{
|
||||
"HGETEX": {
|
||||
"summary": "Get the value of one or more fields of a given hash key, and optionally set their expiration.",
|
||||
"complexity": "O(N) where N is the number of specified fields",
|
||||
"group": "hash",
|
||||
"since": "8.0.0",
|
||||
"arity": -5,
|
||||
"function": "hgetexCommand",
|
||||
"history": [],
|
||||
"command_flags": [
|
||||
"WRITE",
|
||||
"FAST"
|
||||
],
|
||||
"acl_categories": [
|
||||
"HASH"
|
||||
],
|
||||
"key_specs": [
|
||||
{
|
||||
"notes": "RW and UPDATE because it changes the TTL",
|
||||
"flags": [
|
||||
"RW",
|
||||
"ACCESS",
|
||||
"UPDATE"
|
||||
],
|
||||
"begin_search": {
|
||||
"index": {
|
||||
"pos": 1
|
||||
}
|
||||
},
|
||||
"find_keys": {
|
||||
"range": {
|
||||
"lastkey": 0,
|
||||
"step": 1,
|
||||
"limit": 0
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"reply_schema": {
|
||||
"description": "List of values associated with the given fields, in the same order as they are requested.",
|
||||
"type": "array",
|
||||
"minItems": 1,
|
||||
"items": {
|
||||
"oneOf": [
|
||||
{
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"type": "null"
|
||||
}
|
||||
]
|
||||
}
|
||||
},
|
||||
"arguments": [
|
||||
{
|
||||
"name": "key",
|
||||
"type": "key",
|
||||
"key_spec_index": 0
|
||||
},
|
||||
{
|
||||
"name": "expiration",
|
||||
"type": "oneof",
|
||||
"optional": true,
|
||||
"arguments": [
|
||||
{
|
||||
"name": "seconds",
|
||||
"type": "integer",
|
||||
"token": "EX"
|
||||
},
|
||||
{
|
||||
"name": "milliseconds",
|
||||
"type": "integer",
|
||||
"token": "PX"
|
||||
},
|
||||
{
|
||||
"name": "unix-time-seconds",
|
||||
"type": "unix-time",
|
||||
"token": "EXAT"
|
||||
},
|
||||
{
|
||||
"name": "unix-time-milliseconds",
|
||||
"type": "unix-time",
|
||||
"token": "PXAT"
|
||||
},
|
||||
{
|
||||
"name": "persist",
|
||||
"type": "pure-token",
|
||||
"token": "PERSIST"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "fields",
|
||||
"token": "FIELDS",
|
||||
"type": "block",
|
||||
"arguments": [
|
||||
{
|
||||
"name": "numfields",
|
||||
"type": "integer"
|
||||
},
|
||||
{
|
||||
"name": "field",
|
||||
"type": "string",
|
||||
"multiple": true
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,7 +9,6 @@
|
||||
"history": [],
|
||||
"command_flags": [
|
||||
"WRITE",
|
||||
"DENYOOM",
|
||||
"FAST"
|
||||
],
|
||||
"acl_categories": [
|
||||
|
||||
@@ -9,7 +9,6 @@
|
||||
"history": [],
|
||||
"command_flags": [
|
||||
"WRITE",
|
||||
"DENYOOM",
|
||||
"FAST"
|
||||
],
|
||||
"acl_categories": [
|
||||
|
||||
@@ -14,6 +14,9 @@
|
||||
"acl_categories": [
|
||||
"HASH"
|
||||
],
|
||||
"command_tips": [
|
||||
"NONDETERMINISTIC_OUTPUT"
|
||||
],
|
||||
"key_specs": [
|
||||
{
|
||||
"flags": [
|
||||
|
||||
@@ -0,0 +1,132 @@
|
||||
{
|
||||
"HSETEX": {
|
||||
"summary": "Set the value of one or more fields of a given hash key, and optionally set their expiration.",
|
||||
"complexity": "O(N) where N is the number of fields being set.",
|
||||
"group": "hash",
|
||||
"since": "8.0.0",
|
||||
"arity": -6,
|
||||
"function": "hsetexCommand",
|
||||
"command_flags": [
|
||||
"WRITE",
|
||||
"DENYOOM",
|
||||
"FAST"
|
||||
],
|
||||
"acl_categories": [
|
||||
"HASH"
|
||||
],
|
||||
"key_specs": [
|
||||
{
|
||||
"flags": [
|
||||
"RW",
|
||||
"UPDATE"
|
||||
],
|
||||
"begin_search": {
|
||||
"index": {
|
||||
"pos": 1
|
||||
}
|
||||
},
|
||||
"find_keys": {
|
||||
"range": {
|
||||
"lastkey": 0,
|
||||
"step": 1,
|
||||
"limit": 0
|
||||
}
|
||||
}
|
||||
}
|
||||
],
|
||||
"reply_schema": {
|
||||
"oneOf": [
|
||||
{
|
||||
"description": "No field was set (due to FXX or FNX flags).",
|
||||
"const": 0
|
||||
},
|
||||
{
|
||||
"description": "All the fields were set.",
|
||||
"const": 1
|
||||
}
|
||||
]
|
||||
},
|
||||
"arguments": [
|
||||
{
|
||||
"name": "key",
|
||||
"type": "key",
|
||||
"key_spec_index": 0
|
||||
},
|
||||
{
|
||||
"name": "condition",
|
||||
"type": "oneof",
|
||||
"optional": true,
|
||||
"arguments": [
|
||||
{
|
||||
"name": "fnx",
|
||||
"type": "pure-token",
|
||||
"token": "FNX"
|
||||
},
|
||||
{
|
||||
"name": "fxx",
|
||||
"type": "pure-token",
|
||||
"token": "FXX"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "expiration",
|
||||
"type": "oneof",
|
||||
"optional": true,
|
||||
"arguments": [
|
||||
{
|
||||
"name": "seconds",
|
||||
"type": "integer",
|
||||
"token": "EX"
|
||||
},
|
||||
{
|
||||
"name": "milliseconds",
|
||||
"type": "integer",
|
||||
"token": "PX"
|
||||
},
|
||||
{
|
||||
"name": "unix-time-seconds",
|
||||
"type": "unix-time",
|
||||
"token": "EXAT"
|
||||
},
|
||||
{
|
||||
"name": "unix-time-milliseconds",
|
||||
"type": "unix-time",
|
||||
"token": "PXAT"
|
||||
},
|
||||
{
|
||||
"name": "keepttl",
|
||||
"type": "pure-token",
|
||||
"token": "KEEPTTL"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "fields",
|
||||
"token": "FIELDS",
|
||||
"type": "block",
|
||||
"arguments": [
|
||||
{
|
||||
"name": "numfields",
|
||||
"type": "integer"
|
||||
},
|
||||
{
|
||||
"name": "data",
|
||||
"type": "block",
|
||||
"multiple": true,
|
||||
"arguments": [
|
||||
{
|
||||
"name": "field",
|
||||
"type": "string"
|
||||
},
|
||||
{
|
||||
"name": "value",
|
||||
"type": "string"
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -14,6 +14,9 @@
|
||||
"acl_categories": [
|
||||
"HASH"
|
||||
],
|
||||
"command_tips": [
|
||||
"NONDETERMINISTIC_OUTPUT"
|
||||
],
|
||||
"key_specs": [
|
||||
{
|
||||
"flags": [
|
||||
|
||||
@@ -29,6 +29,9 @@
|
||||
"replication.backlog": {
|
||||
"type": "integer"
|
||||
},
|
||||
"replica.fullsync.buffer": {
|
||||
"type": "integer"
|
||||
},
|
||||
"clients.slaves": {
|
||||
"type": "integer"
|
||||
},
|
||||
@@ -44,6 +47,9 @@
|
||||
"lua.caches": {
|
||||
"type": "integer"
|
||||
},
|
||||
"script.VMs": {
|
||||
"type": "integer"
|
||||
},
|
||||
"functions.caches": {
|
||||
"type": "integer"
|
||||
},
|
||||
|
||||
@@ -0,0 +1,75 @@
|
||||
{
|
||||
"SFLUSH": {
|
||||
"summary": "Remove all keys from selected range of slots.",
|
||||
"complexity": "O(N)+O(k) where N is the number of keys and k is the number of slots.",
|
||||
"group": "server",
|
||||
"since": "8.0.0",
|
||||
"arity": -3,
|
||||
"function": "sflushCommand",
|
||||
"command_flags": [
|
||||
"WRITE",
|
||||
"EXPERIMENTAL"
|
||||
],
|
||||
"acl_categories": [
|
||||
"KEYSPACE",
|
||||
"DANGEROUS"
|
||||
],
|
||||
"command_tips": [
|
||||
],
|
||||
"reply_schema": {
|
||||
"description": "List of slot ranges",
|
||||
"type": "array",
|
||||
"minItems": 0,
|
||||
"maxItems": 4294967295,
|
||||
"items": {
|
||||
"type": "array",
|
||||
"minItems": 2,
|
||||
"maxItems": 2,
|
||||
"items": [
|
||||
{
|
||||
"description": "start slot number",
|
||||
"type": "integer"
|
||||
},
|
||||
{
|
||||
"description": "end slot number",
|
||||
"type": "integer"
|
||||
}
|
||||
]
|
||||
}
|
||||
},
|
||||
"arguments": [
|
||||
{
|
||||
"name": "data",
|
||||
"type": "block",
|
||||
"multiple": true,
|
||||
"arguments": [
|
||||
{
|
||||
"name": "slot-start",
|
||||
"type": "integer"
|
||||
},
|
||||
{
|
||||
"name": "slot-last",
|
||||
"type": "integer"
|
||||
}
|
||||
]
|
||||
},
|
||||
{
|
||||
"name": "flush-type",
|
||||
"type": "oneof",
|
||||
"optional": true,
|
||||
"arguments": [
|
||||
{
|
||||
"name": "async",
|
||||
"type": "pure-token",
|
||||
"token": "ASYNC"
|
||||
},
|
||||
{
|
||||
"name": "sync",
|
||||
"type": "pure-token",
|
||||
"token": "SYNC"
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -5,7 +5,7 @@
|
||||
"group": "set",
|
||||
"since": "1.0.0",
|
||||
"arity": 2,
|
||||
"function": "sinterCommand",
|
||||
"function": "smembersCommand",
|
||||
"command_flags": [
|
||||
"READONLY"
|
||||
],
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user