From 6fb7f0ad72238a7a401d6cbea204f5b1da9040c8 Mon Sep 17 00:00:00 2001 From: Vadim Yalovets Date: Thu, 25 Jun 2026 11:28:18 +0300 Subject: [PATCH 01/32] PS-10983 Packaging tasks for release - PS 9.7.1-1 --- build-ps/debian/not-installed | 1 - 1 file changed, 1 deletion(-) diff --git a/build-ps/debian/not-installed b/build-ps/debian/not-installed index 1a228b4c3108..8f7692a0e41d 100644 --- a/build-ps/debian/not-installed +++ b/build-ps/debian/not-installed @@ -45,4 +45,3 @@ usr/lib/mysql/plugin/debug/component_validate_password.so usr/lib/mysql/plugin/debug/authentication_ldap_simple.so # Router plugins usr/lib/mysqlrouter/plugin/mysql_rest_service.so -usr/lib/mysqlrouter/private/libprotobuf.so.* From 861edce9b943902b808e3acfa237ebc545a757cb Mon Sep 17 00:00:00 2001 From: Vadim Yalovets Date: Sat, 27 Jun 2026 13:48:57 +0300 Subject: [PATCH 02/32] PS-10983 Packaging tasks for release - PS 9.7.1-1 (#6046) --- build-ps/debian/percona-mysql-router.install | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/build-ps/debian/percona-mysql-router.install b/build-ps/debian/percona-mysql-router.install index c3937952f979..9f75ec9ab01c 100644 --- a/build-ps/debian/percona-mysql-router.install +++ b/build-ps/debian/percona-mysql-router.install @@ -43,7 +43,7 @@ usr/lib/mysqlrouter/plugin/rest_router.so usr/lib/mysqlrouter/plugin/rest_routing.so usr/lib/mysqlrouter/plugin/rest_metadata_cache.so # private shared libraries -#usr/lib/mysqlrouter/private/libprotobuf-lite.so.* +usr/lib/mysqlrouter/private/libprotobuf.so.* usr/lib/mysqlrouter/private/libabsl_*.so usr/lib/mysqlrouter/private/libmysqlharness.so.1 usr/lib/mysqlrouter/private/libmysqlharness_stdx.so.1 From bbd1a749af1cec3bcc97403fbd6ce32f2444595a Mon Sep 17 00:00:00 2001 From: Vadim Yalovets Date: Mon, 29 Jun 2026 15:34:33 +0300 Subject: [PATCH 03/32] PS-10983 Packaging tasks for release - PS 9.7.1-1 (#6047) --- build-ps/debian/percona-mysql-router.install | 2 ++ router/doc/sample_mysqlrouter.conf | 10 ++++++++++ 2 files changed, 12 insertions(+) diff --git a/build-ps/debian/percona-mysql-router.install b/build-ps/debian/percona-mysql-router.install index 9f75ec9ab01c..605e49f3d3c6 100644 --- a/build-ps/debian/percona-mysql-router.install +++ b/build-ps/debian/percona-mysql-router.install @@ -35,6 +35,8 @@ usr/lib/mysqlrouter/plugin/io.so usr/lib/mysqlrouter/plugin/keepalive.so usr/lib/mysqlrouter/plugin/routing.so usr/lib/mysqlrouter/plugin/router_openssl.so +usr/lib/mysqlrouter/plugin/host_cache.so +usr/lib/mysqlrouter/plugin/rest_host_cache.so usr/lib/mysqlrouter/plugin/router_protobuf.so usr/lib/mysqlrouter/plugin/metadata_cache.so usr/lib/mysqlrouter/plugin/rest_api.so diff --git a/router/doc/sample_mysqlrouter.conf b/router/doc/sample_mysqlrouter.conf index d3a5b3ab44ad..471974ab3543 100644 --- a/router/doc/sample_mysqlrouter.conf +++ b/router/doc/sample_mysqlrouter.conf @@ -55,6 +55,16 @@ #routing_strategy = first-available #destinations = mysql-server1:3306,mysql-server2 +# DNS resolver cache used by the routing plugin. All options have defaults +# and the section can be omitted, but it must be present in the configuration +# when the routing plugin is loaded. +[host_cache] +#enabled = true +#ttl_success_seconds = 60 +#ttl_negative_seconds = 10 +#ttl_jitter_ratio = 0.2 +#max_entries = 250 + # If no plugin is configured which starts a service, keepalive # will make sure MySQL Router will not immediately exit. It is # safe to remove once Router is configured. From b4ca3ed9238184d6e66cf892059f92b443f3ac73 Mon Sep 17 00:00:00 2001 From: Vadim Yalovets Date: Tue, 30 Jun 2026 20:30:33 +0300 Subject: [PATCH 04/32] PS-10983 Packaging tasks for release - PS 9.7.1-1 --- .../percona-mysql-router.mysqlrouter.tmpfile | 2 +- build-ps/percona-server-9.0_builder.sh | 37 ++++++++++++++++++- build-ps/percona-server.spec.in | 7 ++++ 3 files changed, 44 insertions(+), 2 deletions(-) diff --git a/build-ps/debian/percona-mysql-router.mysqlrouter.tmpfile b/build-ps/debian/percona-mysql-router.mysqlrouter.tmpfile index 9d7467065849..99fbff17d548 100644 --- a/build-ps/debian/percona-mysql-router.mysqlrouter.tmpfile +++ b/build-ps/debian/percona-mysql-router.mysqlrouter.tmpfile @@ -20,4 +20,4 @@ # along with this program; if not, write to the Free Software # Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA -d run 0755 mysqlrouter mysqlrouter - +d /run/mysqlrouter 0755 mysqlrouter mysqlrouter - diff --git a/build-ps/percona-server-9.0_builder.sh b/build-ps/percona-server-9.0_builder.sh index e7e4c68ec45c..19633df9e5b2 100644 --- a/build-ps/percona-server-9.0_builder.sh +++ b/build-ps/percona-server-9.0_builder.sh @@ -704,10 +704,29 @@ build_srpm(){ sed -i "/^%changelog/a * $(date "+%a") $(date "+%b") $(date "+%d") $(date "+%Y") Percona Development Team - ${VERSION}-${RELEASE}" percona-server.spec # cd ${WORKDIR}/rpmbuild/SOURCES + CALLHOME_SHA="0e3a2ed40336c70727f9aad8402a8a820ebc8db0" + CALLHOME_SHA256="3497f6631e71799bed9dedb1d72350bf1f0565d93578955234ac30cf2fb6eba4" + wget -q "https://raw.githubusercontent.com/percona/telemetry-agent/${CALLHOME_SHA}/call-home.sh" + echo "${CALLHOME_SHA256} call-home.sh" | sha256sum -c - || { echo "ERROR: call-home.sh checksum mismatch"; exit 1; } tar vxzf ${WORKDIR}/${TARFILE} --wildcards '*/build-ps/rpm/*.patch' --strip=3 tar vxzf ${WORKDIR}/${TARFILE} --wildcards '*/build-ps/rpm/mysql_config.sh' --strip=3 tar vxzf ${WORKDIR}/${TARFILE} --wildcards '*/build-ps/rpm/percona-telemetry-setup.sh' --strip=3 tar vxzf ${WORKDIR}/${TARFILE} --wildcards '*/build-ps/rpm/percona-telemetry-cleanup.sh' --strip=3 + # + cd ${WORKDIR}/rpmbuild/SPECS + ls -la + cat percona-server.spec + grep -n SOURCE999 percona-server.spec + line_number=$(grep -n SOURCE999 percona-server.spec | awk -F ':' '{print $1}') + cp ../SOURCES/call-home.sh ./ + awk -v n=$line_number 'NR <= n {print > "part1.txt"} NR > n {print > "part2.txt"}' percona-server.spec + head -n -1 part1.txt > temp && mv temp part1.txt + echo "cat <<'CALLHOME' > \$tfn" >> part1.txt + cat call-home.sh >> part1.txt + echo "CALLHOME" >> part1.txt + cat part2.txt >> part1.txt + rm -f call-home.sh part2.txt + mv part1.txt percona-server.spec cd ${WORKDIR} # mv -fv ${TARFILE} ${WORKDIR}/rpmbuild/SOURCES @@ -954,7 +973,23 @@ build_deb(){ cd ${DIRNAME} dch -b -m -D "$DEBIAN_VERSION" --force-distribution -v "${VERSION}-${RELEASE}-${DEB_RELEASE}.${DEBIAN_VERSION}" 'Update distribution' - # Telemetry is now handled by percona-telemetry-setup.sh installed via .install file + cd debian/ + CALLHOME_SHA="0e3a2ed40336c70727f9aad8402a8a820ebc8db0" + CALLHOME_SHA256="3497f6631e71799bed9dedb1d72350bf1f0565d93578955234ac30cf2fb6eba4" + wget -q "https://raw.githubusercontent.com/percona/telemetry-agent/${CALLHOME_SHA}/call-home.sh" + echo "${CALLHOME_SHA256} call-home.sh" | sha256sum -c - || { echo "ERROR: call-home.sh checksum mismatch"; exit 1; } + sed -i 's:exit 0::' percona-server-server.postinst + echo "tfn=\$(/usr/bin/mktemp -p \$(/usr/bin/mktemp -d /tmp/XXXXXXXX) call-home.XXXXXX.sh)" >> percona-server-server.postinst + echo "cat <<'CALLHOME' > \$tfn" >> percona-server-server.postinst + cat call-home.sh >> percona-server-server.postinst + echo "CALLHOME" >> percona-server-server.postinst + echo "bash +x \$tfn -f \"PRODUCT_FAMILY_PS\" -v \"${VERSION}-${RELEASE}-${DEB_RELEASE}\" -d \"PACKAGE\" &>/dev/null || :" >> percona-server-server.postinst + echo "chgrp percona-telemetry /usr/local/percona/telemetry_uuid &>/dev/null || :" >> percona-server-server.postinst + echo "chmod 664 /usr/local/percona/telemetry_uuid &>/dev/null || :" >> percona-server-server.postinst + echo "rm -rf \$tfn" >> percona-server-server.postinst + echo "exit 0" >> percona-server-server.postinst + rm -f call-home.sh + cd ../ # NOTE: a legacy block here used to force CC=gcc-4.7 / gcc-4.8 for any # DEBIAN_VERSION not present in a hardcoded allowlist (trusty, xenial, diff --git a/build-ps/percona-server.spec.in b/build-ps/percona-server.spec.in index 04bf3c5b3d8d..ca696eca13ea 100644 --- a/build-ps/percona-server.spec.in +++ b/build-ps/percona-server.spec.in @@ -824,6 +824,13 @@ if [ -d /etc/percona-server.conf.d ]; then fi fi +tfn=$(/usr/bin/mktemp -p "$(/usr/bin/mktemp -d /tmp/XXXXXXXX)" call-home.XXXXXX.sh) +cp %SOURCE999 /tmp/ 2>/dev/null || +bash $tfn -f "PRODUCT_FAMILY_PS" -v %{mysql_version}-%{percona_server_version}-%{rpm_release} -d "PACKAGE" &>/dev/null || : +chgrp percona-telemetry /usr/local/percona/telemetry_uuid &>/dev/null || : +chmod 664 /usr/local/percona/telemetry_uuid &>/dev/null || : +rm -f $tfn + echo "Percona Server is distributed with several useful UDF (User Defined Function) from Percona Toolkit." echo "Run the following command to install these functions (fnv_64, fnv1a_64, murmur_hash):" echo "mysql -e \"INSTALL COMPONENT 'file://component_percona_udf'\"" From 6fa1af352c904a27df658a063f575391e37966d1 Mon Sep 17 00:00:00 2001 From: Vadim Yalovets Date: Tue, 14 Jul 2026 00:05:56 +0300 Subject: [PATCH 05/32] PS-11238 debian package postinst inconsistency --- .../debian/percona-server-server.postinst | 43 ++++++++++++++----- build-ps/percona-server.spec.in | 31 +++++++------ 2 files changed, 50 insertions(+), 24 deletions(-) diff --git a/build-ps/debian/percona-server-server.postinst b/build-ps/debian/percona-server-server.postinst index dac7f8b18378..4d9f33b339d8 100755 --- a/build-ps/debian/percona-server-server.postinst +++ b/build-ps/debian/percona-server-server.postinst @@ -61,6 +61,33 @@ check_apparmor_files() { fi } +# Decide what to do about an existing /etc/mysql/my.cnf and, if needed, +# ask the admin via debconf (template "$1/existing_config_file"). +resolve_cnf_action() { + TEMPLATE_PREFIX="$1" + CNF_ACTION="Use NEW my.cnf" + if [ -L "/etc/mysql/my.cnf" ]; then + CNF_TARGET=$(readlink -f /etc/mysql/my.cnf 2>/dev/null || true) + if [ "${CNF_TARGET}" = "/etc/mysql/mysql.cnf" ]; then + # Already our alternative, nothing to ask/do. + return 1 + fi + # A symlink managed by update-alternatives, but pointing elsewhere: + # ask before taking it over. + db_input high "${TEMPLATE_PREFIX}/existing_config_file" || true + db_go + db_get "${TEMPLATE_PREFIX}/existing_config_file" && CNF_ACTION=${RET} + elif [ -e "/etc/mysql/my.cnf" ]; then + # A plain file: back it up, then ask. + cp /etc/mysql/my.cnf /etc/mysql/my.cnf.bak + echo "NOTE: /etc/mysql/my.cnf backed up to /etc/mysql/my.cnf.bak" + db_input high "${TEMPLATE_PREFIX}/existing_config_file" || true + db_go + db_get "${TEMPLATE_PREFIX}/existing_config_file" && CNF_ACTION=${RET} + fi + [ "${CNF_ACTION}" = "Use NEW my.cnf" ] +} + MY_BASEDIR_VERSION=$(my_print_defaults --loose-verbose mysqld server | grep basedir | awk -F'=' '{print $2}') TOKUDB=$(dpkg -l | grep -c 'percona-server-tokudb') if [ $TOKUDB = 1 ] @@ -114,13 +141,8 @@ case "$1" in fi fi - CNF_ACTION="Use NEW my.cnf" - # If the existing config file is a proper file, we back it up - if [ -f "/etc/mysql/my.cnf" ] && [ ! -L "/etc/mysql/my.cnf" ]; then - cp /etc/mysql/my.cnf /etc/mysql/my.cnf.bak - db_input high percona-server-server/existing_config_file || true - db_go - db_get percona-server-server/existing_config_file && CNF_ACTION=${RET} + if resolve_cnf_action percona-server-server; then + update-alternatives --force --install /etc/mysql/my.cnf my.cnf "/etc/mysql/mysql.cnf" 300 fi if [ -d /etc/mysql/percona-server.conf.d ]; then CONF_EXISTS=$(grep "percona-server.conf.d" /etc/mysql/mysql.cnf | wc -l) @@ -128,10 +150,6 @@ case "$1" in echo "!includedir /etc/mysql/percona-server.conf.d/" >> /etc/mysql/mysql.cnf fi fi - if [ "${CNF_ACTION}" = "Use NEW my.cnf" ]; then - update-alternatives --force --install /etc/mysql/my.cnf my.cnf "/etc/mysql/mysql.cnf" 300 - update-alternatives --set my.cnf /etc/mysql/mysql.cnf - fi PROFILE_ACTION="Use NEW AppArmor profile" # If the existing AppArmor module/local profile is the proper file, we back it up @@ -181,6 +199,9 @@ EOF fi set +e else + if resolve_cnf_action percona-server-server; then + update-alternatives --force --install /etc/mysql/my.cnf my.cnf "/etc/mysql/mysql.cnf" 300 + fi if [ -f "/etc/apparmor.d/usr.sbin.mysqld" ]; then check_apparmor_files fi diff --git a/build-ps/percona-server.spec.in b/build-ps/percona-server.spec.in index ca696eca13ea..2e5b97f77903 100644 --- a/build-ps/percona-server.spec.in +++ b/build-ps/percona-server.spec.in @@ -260,19 +260,24 @@ Requires: percona-telemetry-agent %endif Obsoletes: community-mysql-bench Obsoletes: mysql-bench -Obsoletes: mariadb-connector-c-config -Obsoletes: mariadb-backup -Obsoletes: mariadb-bench -Obsoletes: mariadb-server -Obsoletes: mariadb-server-galera -Obsoletes: mariadb-server-utils -Obsoletes: mariadb-galera-server -Obsoletes: mariadb-gssapi-server -Obsoletes: mariadb-oqgraph-engine -Provides: MySQL-server%{?_isa} = %{version}-%{release} -Provides: mysql-server = %{version}-%{release} -Provides: mysql-server%{?_isa} = %{version}-%{release} -Conflicts: Percona-SQL-server-50 Percona-Server-server-51 Percona-Server-server-55 Percona-Server-server-56 Percona-Server-server-57 +Obsoletes: mariadb-connector-c-config mariadb11.8-connector-c-config +Obsoletes: mariadb-backup mariadb11.8-backup +Obsoletes: mariadb-bench mariadb11.8-bench +Obsoletes: mariadb-server mariadb11.8-server +Obsoletes: mariadb-server-galera mariadb11.8-server-galera +Obsoletes: mariadb-server-utils mariadb11.8-server-utils +Obsoletes: mariadb-galera-server mariadb11.8-galera-server +Obsoletes: mariadb-gssapi-server mariadb11.8-gssapi-server +Obsoletes: mariadb-oqgraph-engine mariadb11.8-oqgraph-engine +Obsoletes: mariadb-client-utils mariadb11.8-client-utils +Obsoletes: mysql8.4-server < 99 +Obsoletes: mysql8.4 < 99 +Obsoletes: mysql8.4-common < 99 +Obsoletes: mysql8.4-errmsg < 99 +Provides: MySQL-server%{?_isa} = %{version}-%{release} +Provides: mysql-server = %{version}-%{release} +Provides: mysql-server%{?_isa} = %{version}-%{release} +Conflicts: Percona-SQL-server-50 Percona-Server-server-51 Percona-Server-server-55 Percona-Server-server-56 Percona-Server-server-57 Requires(post): systemd Requires(preun): systemd From f8995bbd27ea9495ee035af5bfbeb7d0eed1a0d7 Mon Sep 17 00:00:00 2001 From: Jaideep Karande Date: Thu, 25 Jun 2026 12:14:02 +0530 Subject: [PATCH 06/32] PS-11216: Timestamps needed in the GCS_DEBUG_TRACE file Every line written to GCS_DEBUG_TRACE now carries an ISO 8601 UTC timestamp with microsecond precision at the front of the line: [YYYY-MM-DDTHH:MM:SS.uuuuuuZ] [MYSQL_GCS_DEBUG] [GCS] --- .../r/gr_gcs_debug_trace_timestamps.result | 28 ++++++ .../t/gr_gcs_debug_trace_timestamps.test | 95 +++++++++++++++++++ .../include/mysql/gcs/gcs_logging_system.h | 8 +- .../src/interface/gcs_logging_system.cc | 39 ++++++++ 4 files changed, 164 insertions(+), 6 deletions(-) create mode 100644 mysql-test/suite/group_replication/r/gr_gcs_debug_trace_timestamps.result create mode 100644 mysql-test/suite/group_replication/t/gr_gcs_debug_trace_timestamps.test diff --git a/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_timestamps.result b/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_timestamps.result new file mode 100644 index 000000000000..70f779f02a4d --- /dev/null +++ b/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_timestamps.result @@ -0,0 +1,28 @@ +include/group_replication.inc +Warnings: +Note #### Sending passwords in plain text without SSL/TLS is extremely insecure. +Note #### Storing MySQL user name or password information in the connection metadata repository is not secure and is therefore not recommended. Please consider using the USER and PASSWORD connection options for START REPLICA; see the 'START REPLICA Syntax' in the MySQL Manual for more information. +[connection server1] + +# Step 1: Start Group Replication — GCS_DEBUG_TRACE must be created. +include/start_and_bootstrap_group_replication.inc + +# Step 2: Enable GCS_DEBUG_BASIC tracing +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_BASIC"; +include/assert.inc [Debug option is GCS_DEBUG_BASIC] + +# Step 3: verify ISO 8601 UTC timestamps(microsecond) in GCS_DEBUG_TRACE +include/wait_for_pattern_in_file.inc [\[\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{6}Z\] \[MYSQL_GCS_DEBUG\] \[GCS\]] + +# Step 4: disable then re-enable tracing; timestamps must persist +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_ALL"; +include/assert_grep.inc [Timestamps persist after tracing re-enable] + +# Step 5: Verify Group Replication is still ONLINE. +include/assert.inc [Member is ONLINE after timestamp test] + +# Step 6: Cleanup. +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +include/stop_group_replication.inc +include/group_replication_end.inc diff --git a/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_timestamps.test b/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_timestamps.test new file mode 100644 index 000000000000..e1d40ef7eb41 --- /dev/null +++ b/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_timestamps.test @@ -0,0 +1,95 @@ +################################################################################ +# PS-11216: Timestamps needed in the GCS_DEBUG_TRACE file +# +# This test case verify that every log written to GCS_DEBUG_TRACE carries a +# ISO 8601 UTC timestamp with microsecond precision, leading the line: +# [YYYY-MM-DDTHH:MM:SS.uuuuuuZ] [MYSQL_GCS_DEBUG] [GCS] +# +# Test steps: +# 1. Start Group Replication — GCS_DEBUG_TRACE must be created. +# 2. Enable tracing; let the consumer write a few lines. +# 3. Verify ISO 8601 UTC timestamps with microsecond precision. +# 4. Disable then re-enable; verify timestamps continue in new entries. +# 5. Verify Group Replication is still ONLINE. +# 6. Cleanup. +################################################################################ + +--source include/have_group_replication_plugin.inc +--let $rpl_skip_group_replication_start = 1 +--source include/group_replication.inc + +############################################################################## +# Step 1: Start Group Replication — GCS_DEBUG_TRACE is created. +############################################################################## + +--echo +--echo # Step 1: Start Group Replication — GCS_DEBUG_TRACE must be created. +--source include/start_and_bootstrap_group_replication.inc +--file_exists $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE + +############################################################################## +# Step 2: Enable tracing; let the consumer write a few lines. +############################################################################## +--echo +--echo # Step 2: Enable GCS_DEBUG_BASIC tracing +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_BASIC"; + +--let $assert_text = Debug option is GCS_DEBUG_BASIC +--let $assert_cond = "[SELECT @@GLOBAL.group_replication_communication_debug_options]" = "GCS_DEBUG_BASIC" +--source include/assert.inc + +############################################################################## +# Step 3: Verify ISO 8601 UTC timestamps with microsecond precision. +############################################################################## +--echo +--echo # Step 3: verify ISO 8601 UTC timestamps(microsecond) in GCS_DEBUG_TRACE + +# Wait until at least one timestamped line appears in the trace file. +--let $grep_pattern = \[\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{6}Z\] \[MYSQL_GCS_DEBUG\] \[GCS\] +--let $grep_file = $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE +--source include/wait_for_pattern_in_file.inc + +############################################################################## +# Step 4: Disable then re-enable; verify timestamps continue in new entries. +############################################################################## +--echo +--echo # Step 4: disable then re-enable tracing; timestamps must persist + +# Record file size before the cycle so we can prove new entries were appended. +--let $size_before = `SELECT LENGTH(LOAD_FILE(CONCAT(@@datadir, 'GCS_DEBUG_TRACE')))` + +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_ALL"; + +# Wait until the file grows beyond the size recorded before the cycle. +--let $wait_condition = SELECT LENGTH(LOAD_FILE(CONCAT(@@datadir, 'GCS_DEBUG_TRACE'))) > $size_before +--source include/wait_condition.inc + +--let $assert_text = Timestamps persist after tracing re-enable +--let $assert_file = $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE +--let $assert_select = \[\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{6}Z\] \[MYSQL_GCS_DEBUG\] \[GCS\] +--let $assert_count_condition = >= 1 +--source include/assert_grep.inc + +############################################################################## +# Step 5: Verify Group Replication is still ONLINE. +############################################################################## +--echo +--echo # Step 5: Verify Group Replication is still ONLINE. + +--let $assert_text = Member is ONLINE after timestamp test +--let $assert_cond = COUNT(*) = 1 FROM performance_schema.replication_group_members WHERE MEMBER_STATE = "ONLINE" +--source include/assert.inc + +############################################################################## +# Step 6: Cleanup. +############################################################################## +--echo +--echo # Step 6: Cleanup. + +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +--source include/stop_group_replication.inc + +--remove_file $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE + +--source include/group_replication_end.inc diff --git a/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h b/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h index 5954c3c8144f..c7dfd65efba4 100644 --- a/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h +++ b/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h @@ -659,6 +659,7 @@ class Gcs_default_debugger { /** Add extra information as a message prefix. + Format: [YYYY-MM-DDTHH:MM:SS.uuuuuuZ] [MYSQL_GCS_DEBUG] [GCS] We assume that there is room to accommodate it. Before changing this method, make sure the maximum buffer size will always have room to @@ -666,12 +667,7 @@ class Gcs_default_debugger { @return Return the size of appended information */ - inline size_t append_prefix(char *buffer) { - strcpy(buffer, GCS_DEBUG_PREFIX); - strcpy(buffer + GCS_DEBUG_PREFIX_SIZE, GCS_PREFIX); - - return GCS_DEBUG_PREFIX_SIZE + GCS_PREFIX_SIZE; - } + size_t append_prefix(char *buffer); /** Append information into a message such as end of line. diff --git a/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc b/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc index 94b86404aa9d..ed813b775936 100644 --- a/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc +++ b/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc @@ -37,6 +37,7 @@ #include "my_dir.h" #include "my_io.h" #include "my_sys.h" +#include "my_systime.h" #endif /* XCOM_STANDALONE */ #include "plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h" @@ -337,6 +338,44 @@ enum_gcs_error Gcs_default_debugger::initialize() { enum_gcs_error Gcs_default_debugger::finalize() { return m_sink->finalize(); } +/* + Returns 0 for XCOM_STANDALONE because XCOM_STANDALONE excludes the MySQL + portability headers (my_systime.h, my_sys.h) needed by my_micro_time(). + + XCOM_STANDALONE is defined when XCom is built as a standalone library + for unit-testing the consensus protocol in isolation. +*/ +static size_t gcs_format_timestamp(char *buf [[maybe_unused]]) { +#ifdef XCOM_STANDALONE + return 0; +#else + const unsigned long long us = my_micro_time(); + const time_t sec = static_cast(us / 1000000); + const long usec = static_cast(us % 1000000); + struct tm tm_info; + if (gmtime_r(&sec, &tm_info) == nullptr) return 0; + int len = sprintf(buf, "[%04d-%02d-%02dT%02d:%02d:%02d.%06ldZ] ", + tm_info.tm_year + 1900, tm_info.tm_mon + 1, tm_info.tm_mday, + tm_info.tm_hour, tm_info.tm_min, tm_info.tm_sec, usec); + return (len > 0) ? static_cast(len) : 0; +#endif +} + +size_t Gcs_default_debugger::append_prefix(char *buffer) { + size_t base = 0; + + /* Timestamp leads the prefix so the final format is: + [YYYY-MM-DDTHH:MM:SS.uuuuuuZ] [MYSQL_GCS_DEBUG] [GCS] */ + size_t ts_len = gcs_format_timestamp(buffer); + base += ts_len; + + strcpy(buffer + base, GCS_DEBUG_PREFIX); + strcpy(buffer + base + GCS_DEBUG_PREFIX_SIZE, GCS_PREFIX); + base += GCS_DEBUG_PREFIX_SIZE + GCS_PREFIX_SIZE; + + return base; +} + /** Reference to the default debugger which is used internally by GCS and XCOM. */ From 88252ad0345bffc747d02ed12a9c2227149f5f98 Mon Sep 17 00:00:00 2001 From: Jaideep Karande Date: Thu, 25 Jun 2026 12:14:02 +0530 Subject: [PATCH 07/32] PS-11216: Timestamps needed in the GCS_DEBUG_TRACE file Every line written to GCS_DEBUG_TRACE now carries an ISO 8601 UTC timestamp with microsecond precision at the front of the line: [YYYY-MM-DDTHH:MM:SS.uuuuuuZ] [MYSQL_GCS_DEBUG] [GCS] --- .../r/gr_gcs_debug_trace_timestamps.result | 28 ++++++ .../t/gr_gcs_debug_trace_timestamps.test | 95 +++++++++++++++++++ .../include/mysql/gcs/gcs_logging_system.h | 8 +- .../src/interface/gcs_logging_system.cc | 39 ++++++++ 4 files changed, 164 insertions(+), 6 deletions(-) create mode 100644 mysql-test/suite/group_replication/r/gr_gcs_debug_trace_timestamps.result create mode 100644 mysql-test/suite/group_replication/t/gr_gcs_debug_trace_timestamps.test diff --git a/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_timestamps.result b/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_timestamps.result new file mode 100644 index 000000000000..70f779f02a4d --- /dev/null +++ b/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_timestamps.result @@ -0,0 +1,28 @@ +include/group_replication.inc +Warnings: +Note #### Sending passwords in plain text without SSL/TLS is extremely insecure. +Note #### Storing MySQL user name or password information in the connection metadata repository is not secure and is therefore not recommended. Please consider using the USER and PASSWORD connection options for START REPLICA; see the 'START REPLICA Syntax' in the MySQL Manual for more information. +[connection server1] + +# Step 1: Start Group Replication — GCS_DEBUG_TRACE must be created. +include/start_and_bootstrap_group_replication.inc + +# Step 2: Enable GCS_DEBUG_BASIC tracing +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_BASIC"; +include/assert.inc [Debug option is GCS_DEBUG_BASIC] + +# Step 3: verify ISO 8601 UTC timestamps(microsecond) in GCS_DEBUG_TRACE +include/wait_for_pattern_in_file.inc [\[\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{6}Z\] \[MYSQL_GCS_DEBUG\] \[GCS\]] + +# Step 4: disable then re-enable tracing; timestamps must persist +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_ALL"; +include/assert_grep.inc [Timestamps persist after tracing re-enable] + +# Step 5: Verify Group Replication is still ONLINE. +include/assert.inc [Member is ONLINE after timestamp test] + +# Step 6: Cleanup. +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +include/stop_group_replication.inc +include/group_replication_end.inc diff --git a/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_timestamps.test b/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_timestamps.test new file mode 100644 index 000000000000..e1d40ef7eb41 --- /dev/null +++ b/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_timestamps.test @@ -0,0 +1,95 @@ +################################################################################ +# PS-11216: Timestamps needed in the GCS_DEBUG_TRACE file +# +# This test case verify that every log written to GCS_DEBUG_TRACE carries a +# ISO 8601 UTC timestamp with microsecond precision, leading the line: +# [YYYY-MM-DDTHH:MM:SS.uuuuuuZ] [MYSQL_GCS_DEBUG] [GCS] +# +# Test steps: +# 1. Start Group Replication — GCS_DEBUG_TRACE must be created. +# 2. Enable tracing; let the consumer write a few lines. +# 3. Verify ISO 8601 UTC timestamps with microsecond precision. +# 4. Disable then re-enable; verify timestamps continue in new entries. +# 5. Verify Group Replication is still ONLINE. +# 6. Cleanup. +################################################################################ + +--source include/have_group_replication_plugin.inc +--let $rpl_skip_group_replication_start = 1 +--source include/group_replication.inc + +############################################################################## +# Step 1: Start Group Replication — GCS_DEBUG_TRACE is created. +############################################################################## + +--echo +--echo # Step 1: Start Group Replication — GCS_DEBUG_TRACE must be created. +--source include/start_and_bootstrap_group_replication.inc +--file_exists $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE + +############################################################################## +# Step 2: Enable tracing; let the consumer write a few lines. +############################################################################## +--echo +--echo # Step 2: Enable GCS_DEBUG_BASIC tracing +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_BASIC"; + +--let $assert_text = Debug option is GCS_DEBUG_BASIC +--let $assert_cond = "[SELECT @@GLOBAL.group_replication_communication_debug_options]" = "GCS_DEBUG_BASIC" +--source include/assert.inc + +############################################################################## +# Step 3: Verify ISO 8601 UTC timestamps with microsecond precision. +############################################################################## +--echo +--echo # Step 3: verify ISO 8601 UTC timestamps(microsecond) in GCS_DEBUG_TRACE + +# Wait until at least one timestamped line appears in the trace file. +--let $grep_pattern = \[\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{6}Z\] \[MYSQL_GCS_DEBUG\] \[GCS\] +--let $grep_file = $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE +--source include/wait_for_pattern_in_file.inc + +############################################################################## +# Step 4: Disable then re-enable; verify timestamps continue in new entries. +############################################################################## +--echo +--echo # Step 4: disable then re-enable tracing; timestamps must persist + +# Record file size before the cycle so we can prove new entries were appended. +--let $size_before = `SELECT LENGTH(LOAD_FILE(CONCAT(@@datadir, 'GCS_DEBUG_TRACE')))` + +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_ALL"; + +# Wait until the file grows beyond the size recorded before the cycle. +--let $wait_condition = SELECT LENGTH(LOAD_FILE(CONCAT(@@datadir, 'GCS_DEBUG_TRACE'))) > $size_before +--source include/wait_condition.inc + +--let $assert_text = Timestamps persist after tracing re-enable +--let $assert_file = $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE +--let $assert_select = \[\d{4}-\d{2}-\d{2}T\d{2}:\d{2}:\d{2}\.\d{6}Z\] \[MYSQL_GCS_DEBUG\] \[GCS\] +--let $assert_count_condition = >= 1 +--source include/assert_grep.inc + +############################################################################## +# Step 5: Verify Group Replication is still ONLINE. +############################################################################## +--echo +--echo # Step 5: Verify Group Replication is still ONLINE. + +--let $assert_text = Member is ONLINE after timestamp test +--let $assert_cond = COUNT(*) = 1 FROM performance_schema.replication_group_members WHERE MEMBER_STATE = "ONLINE" +--source include/assert.inc + +############################################################################## +# Step 6: Cleanup. +############################################################################## +--echo +--echo # Step 6: Cleanup. + +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +--source include/stop_group_replication.inc + +--remove_file $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE + +--source include/group_replication_end.inc diff --git a/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h b/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h index c08c1430ecf3..1a464f8fd951 100644 --- a/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h +++ b/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h @@ -679,6 +679,7 @@ class Gcs_default_debugger { /** Add extra information as a message prefix. + Format: [YYYY-MM-DDTHH:MM:SS.uuuuuuZ] [MYSQL_GCS_DEBUG] [GCS] We assume that there is room to accommodate it. Before changing this method, make sure the maximum buffer size will always have room to @@ -686,12 +687,7 @@ class Gcs_default_debugger { @return Return the size of appended information */ - inline size_t append_prefix(char *buffer) { - strcpy(buffer, GCS_DEBUG_PREFIX); - strcpy(buffer + GCS_DEBUG_PREFIX_SIZE, GCS_PREFIX); - - return GCS_DEBUG_PREFIX_SIZE + GCS_PREFIX_SIZE; - } + size_t append_prefix(char *buffer); /** Append information into a message such as end of line. diff --git a/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc b/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc index 389f052d720d..4b1e91e7f356 100644 --- a/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc +++ b/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc @@ -40,6 +40,7 @@ #include "my_dir.h" #include "my_io.h" #include "my_sys.h" +#include "my_systime.h" #endif /* XCOM_STANDALONE */ #include "plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h" @@ -342,6 +343,44 @@ enum_gcs_error Gcs_default_debugger::initialize() { enum_gcs_error Gcs_default_debugger::finalize() { return m_sink->finalize(); } +/* + Returns 0 for XCOM_STANDALONE because XCOM_STANDALONE excludes the MySQL + portability headers (my_systime.h, my_sys.h) needed by my_micro_time(). + + XCOM_STANDALONE is defined when XCom is built as a standalone library + for unit-testing the consensus protocol in isolation. +*/ +static size_t gcs_format_timestamp(char *buf [[maybe_unused]]) { +#ifdef XCOM_STANDALONE + return 0; +#else + const unsigned long long us = my_micro_time(); + const time_t sec = static_cast(us / 1000000); + const long usec = static_cast(us % 1000000); + struct tm tm_info; + if (gmtime_r(&sec, &tm_info) == nullptr) return 0; + int len = sprintf(buf, "[%04d-%02d-%02dT%02d:%02d:%02d.%06ldZ] ", + tm_info.tm_year + 1900, tm_info.tm_mon + 1, tm_info.tm_mday, + tm_info.tm_hour, tm_info.tm_min, tm_info.tm_sec, usec); + return (len > 0) ? static_cast(len) : 0; +#endif +} + +size_t Gcs_default_debugger::append_prefix(char *buffer) { + size_t base = 0; + + /* Timestamp leads the prefix so the final format is: + [YYYY-MM-DDTHH:MM:SS.uuuuuuZ] [MYSQL_GCS_DEBUG] [GCS] */ + size_t ts_len = gcs_format_timestamp(buffer); + base += ts_len; + + strcpy(buffer + base, GCS_DEBUG_PREFIX); + strcpy(buffer + base + GCS_DEBUG_PREFIX_SIZE, GCS_PREFIX); + base += GCS_DEBUG_PREFIX_SIZE + GCS_PREFIX_SIZE; + + return base; +} + /** Reference to the default debugger which is used internally by GCS and XCOM. */ From 4e3455e398b20609d45928177760b88b79eb07c4 Mon Sep 17 00:00:00 2001 From: Jaideep Karande Date: Thu, 2 Jul 2026 01:18:44 +0530 Subject: [PATCH 08/32] PS-11217: GCS_DEBUG_TRACE log not possible to rotate/archive without restarting replication https://perconadev.atlassian.net/browse/PS-11217 Problem 1: Group Replication's GCS_DEBUG_TRACE file descriptor was opened once and never revisited. So, if an operator externally moved or removed the file output kept going to the stale/detached file descriptor with no error reported. The only fix was a full Group Replication restart. Problem 2: GCS_DEBUG_TRACE are always written to a same file, which mean file can grow huge and logs of all MySQL sessions are written in a same file, no automatic rotation of logs is available. Resolution: If the file is gone (moved/deleted) reopen the file. Log Rotation: Size-based: new sysvar group_replication_communication_debug_max_file_size (bytes, default 0 = disabled) caps the trace file size, like max_binlog_size. Additinally timestamp is added to the file name that can help in opening the right log file if huge logs are generated. Similarly during every START GROUP_REPLICATION logs will be written to a new trace file. --- .../r/gr_gcs_debug_trace_auto_rotation.result | 34 +++ .../r/gr_gcs_debug_trace_reopen.result | 23 ++ .../r/gr_persist_only_variables.result | 10 +- .../r/gr_persist_variables.result | 10 +- .../r/gr_set_option_during_stop.result | 3 + ...r_show_global_and_session_variables.result | 4 +- .../r/gr_variables_default_values.result | 5 +- .../r/gr_variables_privileges.result | 3 + .../t/gr_gcs_debug_trace_auto_rotation.test | 168 +++++++++++++ .../t/gr_gcs_debug_trace_reopen.test | 63 +++++ .../t/gr_set_option_during_stop.test | 1 + .../gr_show_global_and_session_variables.test | 2 +- .../t/gr_variables_default_values.test | 7 +- .../include/plugin_variables.h | 5 + .../include/mysql/gcs/gcs_logging_system.h | 79 ++++++ .../src/bindings/xcom/gcs_xcom_interface.cc | 26 ++ .../src/interface/gcs_logging_system.cc | 238 +++++++++++++++++- plugin/group_replication/src/plugin.cc | 27 ++ 18 files changed, 689 insertions(+), 19 deletions(-) create mode 100644 mysql-test/suite/group_replication/r/gr_gcs_debug_trace_auto_rotation.result create mode 100644 mysql-test/suite/group_replication/r/gr_gcs_debug_trace_reopen.result create mode 100644 mysql-test/suite/group_replication/t/gr_gcs_debug_trace_auto_rotation.test create mode 100644 mysql-test/suite/group_replication/t/gr_gcs_debug_trace_reopen.test diff --git a/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_auto_rotation.result b/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_auto_rotation.result new file mode 100644 index 000000000000..d334985d3859 --- /dev/null +++ b/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_auto_rotation.result @@ -0,0 +1,34 @@ +include/group_replication.inc +Warnings: +Note #### Sending passwords in plain text without SSL/TLS is extremely insecure. +Note #### Storing MySQL user name or password information in the connection metadata repository is not secure and is therefore not recommended. Please consider using the USER and PASSWORD connection options for START REPLICA; see the 'START REPLICA Syntax' in the MySQL Manual for more information. +[connection server1] + +# 1. Set communication_debug_max_file_size and start GR. +# Trace file with timestamp are created. +SET GLOBAL group_replication_communication_debug_max_file_size = 200; +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_ALL"; +include/start_and_bootstrap_group_replication.inc +Initial trace file is timestamped, as expected +No plain GCS_DEBUG_TRACE created, as expected + +# 2. Size-based rotation: old file stays, new timestamped file created. +Size-based rotation created a new timestamped file, as expected +Original file still exists after rotation (no archiving), as expected +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; + +# 3. Restart Group Replication -- a new timestamped file is created on start. +Recorded file count before restart +include/stop_group_replication.inc +include/start_and_bootstrap_group_replication.inc +New timestamped file created on GR restart, as expected + +# 4. Group Replication is ONLINE. +include/assert.inc [Member is ONLINE after automatic rotation] + +# 5. Cleanup. +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +SET GLOBAL group_replication_communication_debug_max_file_size = 0; +include/stop_group_replication.inc +Cleanup complete +include/group_replication_end.inc diff --git a/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_reopen.result b/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_reopen.result new file mode 100644 index 000000000000..ff747493fc4b --- /dev/null +++ b/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_reopen.result @@ -0,0 +1,23 @@ +include/group_replication.inc +Warnings: +Note #### Sending passwords in plain text without SSL/TLS is extremely insecure. +Note #### Storing MySQL user name or password information in the connection metadata repository is not secure and is therefore not recommended. Please consider using the USER and PASSWORD connection options for START REPLICA; see the 'START REPLICA Syntax' in the MySQL Manual for more information. +[connection server1] + +# 1. Start Group Replication. +include/start_and_bootstrap_group_replication.inc + +# 2. Enable GCS_DEBUG_ALL tracing. +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_ALL"; + +# 3. Remove GCS_DEBUG_TRACE. + +# 4. Wait for stat auto-detection -- GCS_DEBUG_TRACE must reappear + +# 5. Group Replication is ONLINE. +include/assert.inc [Member is ONLINE after stale-fd detection and reopen] + +# 6. cleanup +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +include/stop_group_replication.inc +include/group_replication_end.inc diff --git a/mysql-test/suite/group_replication/r/gr_persist_only_variables.result b/mysql-test/suite/group_replication/r/gr_persist_only_variables.result index 84b51ce8187b..0109e899b555 100644 --- a/mysql-test/suite/group_replication/r/gr_persist_only_variables.result +++ b/mysql-test/suite/group_replication/r/gr_persist_only_variables.result @@ -30,6 +30,7 @@ SET PERSIST_ONLY group_replication_bootstrap_group = @@GLOBAL.group_replication_ SET PERSIST_ONLY group_replication_certification_loop_chunk_size = @@GLOBAL.group_replication_certification_loop_chunk_size; SET PERSIST_ONLY group_replication_certification_loop_sleep_time = @@GLOBAL.group_replication_certification_loop_sleep_time; SET PERSIST_ONLY group_replication_clone_threshold = @@GLOBAL.group_replication_clone_threshold; +SET PERSIST_ONLY group_replication_communication_debug_max_file_size = @@GLOBAL.group_replication_communication_debug_max_file_size; SET PERSIST_ONLY group_replication_communication_debug_options = @@GLOBAL.group_replication_communication_debug_options; SET PERSIST_ONLY group_replication_communication_max_message_size = @@GLOBAL.group_replication_communication_max_message_size; SET PERSIST_ONLY group_replication_communication_stack = @@GLOBAL.group_replication_communication_stack; @@ -87,7 +88,7 @@ SET PERSIST_ONLY group_replication_view_change_uuid = @@GLOBAL.group_replication SET PERSIST_ONLY group_replication_xcom_ssl_accept_retries = @@GLOBAL.group_replication_xcom_ssl_accept_retries; SET PERSIST_ONLY group_replication_xcom_ssl_socket_timeout = @@GLOBAL.group_replication_xcom_ssl_socket_timeout; -include/assert.inc ['Expect 65 persisted variables.'] +include/assert.inc ['Expect 66 persisted variables.'] ############################################################ # 2. Restart server, it must bootstrap the group and preserve @@ -96,9 +97,9 @@ include/assert.inc ['Expect 65 persisted variables.'] include/rpl/reconnect.inc include/gr_wait_for_member_state.inc -include/assert.inc ['Expect 65 persisted variables in persisted_variables table.'] -include/assert.inc ['Expect 64 variables which last value was set through SET PERSIST.'] -include/assert.inc ['Expect 64 persisted variables with matching persisted and global values.'] +include/assert.inc ['Expect 66 persisted variables in persisted_variables table.'] +include/assert.inc ['Expect 65 variables which last value was set through SET PERSIST.'] +include/assert.inc ['Expect 65 persisted variables with matching persisted and global values.'] ############################################################ # 3. Test RESET PERSIST IF EXISTS. @@ -112,6 +113,7 @@ RESET PERSIST IF EXISTS group_replication_bootstrap_group; RESET PERSIST IF EXISTS group_replication_certification_loop_chunk_size; RESET PERSIST IF EXISTS group_replication_certification_loop_sleep_time; RESET PERSIST IF EXISTS group_replication_clone_threshold; +RESET PERSIST IF EXISTS group_replication_communication_debug_max_file_size; RESET PERSIST IF EXISTS group_replication_communication_debug_options; RESET PERSIST IF EXISTS group_replication_communication_max_message_size; RESET PERSIST IF EXISTS group_replication_communication_stack; diff --git a/mysql-test/suite/group_replication/r/gr_persist_variables.result b/mysql-test/suite/group_replication/r/gr_persist_variables.result index 177d2c41ecde..8aa637026699 100644 --- a/mysql-test/suite/group_replication/r/gr_persist_variables.result +++ b/mysql-test/suite/group_replication/r/gr_persist_variables.result @@ -32,6 +32,7 @@ SET PERSIST group_replication_bootstrap_group = @@GLOBAL.group_replication_boots SET PERSIST group_replication_certification_loop_chunk_size = @@GLOBAL.group_replication_certification_loop_chunk_size; SET PERSIST group_replication_certification_loop_sleep_time = @@GLOBAL.group_replication_certification_loop_sleep_time; SET PERSIST group_replication_clone_threshold = @@GLOBAL.group_replication_clone_threshold; +SET PERSIST group_replication_communication_debug_max_file_size = @@GLOBAL.group_replication_communication_debug_max_file_size; SET PERSIST group_replication_communication_debug_options = @@GLOBAL.group_replication_communication_debug_options; SET PERSIST group_replication_communication_max_message_size = @@GLOBAL.group_replication_communication_max_message_size; SET PERSIST group_replication_communication_stack = @@GLOBAL.group_replication_communication_stack; @@ -91,7 +92,7 @@ Warning 1681 'group_replication_view_change_uuid' is deprecated and will be remo SET PERSIST group_replication_xcom_ssl_accept_retries = @@GLOBAL.group_replication_xcom_ssl_accept_retries; SET PERSIST group_replication_xcom_ssl_socket_timeout = @@GLOBAL.group_replication_xcom_ssl_socket_timeout; -include/assert.inc ['Expect 65 persisted variables.'] +include/assert.inc ['Expect 66 persisted variables.'] ############################################################ # 2. Restart server, it must bootstrap the group and preserve @@ -100,9 +101,9 @@ include/assert.inc ['Expect 65 persisted variables.'] include/rpl/reconnect.inc include/gr_wait_for_member_state.inc -include/assert.inc ['Expect 65 persisted variables in persisted_variables table.'] -include/assert.inc ['Expect 64 variables which last value was set through SET PERSIST.'] -include/assert.inc ['Expect 64 variables which last value was set through SET PERSIST is equal to its global value.'] +include/assert.inc ['Expect 66 persisted variables in persisted_variables table.'] +include/assert.inc ['Expect 65 variables which last value was set through SET PERSIST.'] +include/assert.inc ['Expect 65 variables which last value was set through SET PERSIST is equal to its global value.'] ############################################################ # 3. Test RESET PERSIST. @@ -116,6 +117,7 @@ RESET PERSIST group_replication_bootstrap_group; RESET PERSIST group_replication_certification_loop_chunk_size; RESET PERSIST group_replication_certification_loop_sleep_time; RESET PERSIST group_replication_clone_threshold; +RESET PERSIST group_replication_communication_debug_max_file_size; RESET PERSIST group_replication_communication_debug_options; RESET PERSIST group_replication_communication_max_message_size; RESET PERSIST group_replication_communication_stack; diff --git a/mysql-test/suite/group_replication/r/gr_set_option_during_stop.result b/mysql-test/suite/group_replication/r/gr_set_option_during_stop.result index 26b5d0270db5..a3e290d583fc 100644 --- a/mysql-test/suite/group_replication/r/gr_set_option_during_stop.result +++ b/mysql-test/suite/group_replication/r/gr_set_option_during_stop.result @@ -30,6 +30,7 @@ SELECT VARIABLE_NAME FROM performance_schema.global_variables WHERE VARIABLE_NAME LIKE 'group_replication_%' AND VARIABLE_NAME != 'group_replication_allow_local_lower_version_join' AND VARIABLE_NAME != 'group_replication_bootstrap_group' + AND VARIABLE_NAME != 'group_replication_communication_debug_max_file_size' AND VARIABLE_NAME != 'group_replication_communication_stack' AND VARIABLE_NAME != 'group_replication_consistency' AND VARIABLE_NAME != 'group_replication_exit_state_action' @@ -187,6 +188,8 @@ SET @value= @@GLOBAL.group_replication_certification_loop_chunk_size; SET @@GLOBAL.group_replication_certification_loop_chunk_size= @value; SET @value= @@GLOBAL.group_replication_certification_loop_sleep_time; SET @@GLOBAL.group_replication_certification_loop_sleep_time= @value; +SET @value= @@GLOBAL.group_replication_communication_debug_max_file_size; +SET @@GLOBAL.group_replication_communication_debug_max_file_size= @value; SET @value= @@GLOBAL.group_replication_communication_stack; SET @@GLOBAL.group_replication_communication_stack= @value; SET @value= @@GLOBAL.group_replication_consistency; diff --git a/mysql-test/suite/group_replication/r/gr_show_global_and_session_variables.result b/mysql-test/suite/group_replication/r/gr_show_global_and_session_variables.result index 06e7e112a8d7..91afba3dabb8 100644 --- a/mysql-test/suite/group_replication/r/gr_show_global_and_session_variables.result +++ b/mysql-test/suite/group_replication/r/gr_show_global_and_session_variables.result @@ -7,8 +7,8 @@ Note #### Storing MySQL user name or password information in the connection meta include/start_and_bootstrap_group_replication.inc include/stop_group_replication.inc -# Test#1: Basic check that there are 66 GR variables. -include/assert.inc [There are 66 GR variables at present.] +# Test#1: Basic check that there are 67 GR variables. +include/assert.inc [There are 67 GR variables at present.] # Test#2: Verify group replication related variables at GLOBAL scope. SET @@SESSION.group_replication_allow_local_lower_version_join= 1; diff --git a/mysql-test/suite/group_replication/r/gr_variables_default_values.result b/mysql-test/suite/group_replication/r/gr_variables_default_values.result index d6cf596cbdae..f1a97c57db5e 100644 --- a/mysql-test/suite/group_replication/r/gr_variables_default_values.result +++ b/mysql-test/suite/group_replication/r/gr_variables_default_values.result @@ -28,9 +28,9 @@ include/stop_group_replication.inc # # Test Unit#1 # Set global/session group replication variables to default. -# Curently there are 66 group replication variables. +# Curently there are 67 group replication variables. # -include/assert.inc [There are 66 GR variables at present.] +include/assert.inc [There are 67 GR variables at present.] SET @@GLOBAL.group_replication_auto_increment_increment= default; ERROR 42000: Variable 'group_replication_auto_increment_increment' can't be set to the value of 'DEFAULT' SET @@GLOBAL.group_replication_compression_threshold= default; @@ -118,6 +118,7 @@ include/assert.inc [Default group_replication_ssl_mode is DISABLED] include/assert.inc [Default group_replication_start_on_boot is ON/1] include/assert.inc [Default group_replication_transaction_size_limit is 150000000] include/assert.inc [Default group_replication_communication_debug_options is "GCS_DEBUG_NONE"] +include/assert.inc [Default group_replication_communication_debug_max_file_size is 0] include/assert.inc [Default group_replication_unreachable_majority_timeout is 0] include/assert.inc [Default group_replication_member_weight is 50] include/assert.inc [Default group_replication_recovery_public_key_path is ""(EMPTY)] diff --git a/mysql-test/suite/group_replication/r/gr_variables_privileges.result b/mysql-test/suite/group_replication/r/gr_variables_privileges.result index a14721acc69f..08efac7144a9 100644 --- a/mysql-test/suite/group_replication/r/gr_variables_privileges.result +++ b/mysql-test/suite/group_replication/r/gr_variables_privileges.result @@ -43,6 +43,8 @@ SET GLOBAL group_replication_certification_loop_sleep_time = @@GLOBAL.group_repl ERROR 42000: Access denied; you need (at least one of) the SUPER or SYSTEM_VARIABLES_ADMIN privilege(s) for this operation SET GLOBAL group_replication_clone_threshold = @@GLOBAL.group_replication_clone_threshold; ERROR 42000: Access denied; you need (at least one of) the SUPER or SYSTEM_VARIABLES_ADMIN privilege(s) for this operation +SET GLOBAL group_replication_communication_debug_max_file_size = @@GLOBAL.group_replication_communication_debug_max_file_size; +ERROR 42000: Access denied; you need (at least one of) the SUPER or SYSTEM_VARIABLES_ADMIN privilege(s) for this operation SET GLOBAL group_replication_communication_debug_options = @@GLOBAL.group_replication_communication_debug_options; ERROR 42000: Access denied; you need (at least one of) the SUPER or SYSTEM_VARIABLES_ADMIN privilege(s) for this operation SET GLOBAL group_replication_communication_max_message_size = @@GLOBAL.group_replication_communication_max_message_size; @@ -182,6 +184,7 @@ SET GLOBAL group_replication_bootstrap_group = @@GLOBAL.group_replication_bootst SET GLOBAL group_replication_certification_loop_chunk_size = @@GLOBAL.group_replication_certification_loop_chunk_size; SET GLOBAL group_replication_certification_loop_sleep_time = @@GLOBAL.group_replication_certification_loop_sleep_time; SET GLOBAL group_replication_clone_threshold = @@GLOBAL.group_replication_clone_threshold; +SET GLOBAL group_replication_communication_debug_max_file_size = @@GLOBAL.group_replication_communication_debug_max_file_size; SET GLOBAL group_replication_communication_debug_options = @@GLOBAL.group_replication_communication_debug_options; SET GLOBAL group_replication_communication_max_message_size = @@GLOBAL.group_replication_communication_max_message_size; SET GLOBAL group_replication_communication_stack = @@GLOBAL.group_replication_communication_stack; diff --git a/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_auto_rotation.test b/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_auto_rotation.test new file mode 100644 index 000000000000..5b7226831906 --- /dev/null +++ b/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_auto_rotation.test @@ -0,0 +1,168 @@ +################################################################################ +# +# This test confirms AUTOMATIC rotation of the GCS_DEBUG_TRACE file. +# +# When group_replication_communication_debug_max_file_size > 0: +# - Every file (including the first) gets a timestamp in its name: +# GCS_DEBUG_TRACE.YYYYMMDDTHHMMSS (UTC) +# - Rotation creates a new timestamped file; the old file is left in place +# (no archiving/renaming -- same model as binary logs). +# - Each START GROUP_REPLICATION also opens a fresh timestamped file. +# +# Test steps: +# 1. Set communication_debug_max_file_size and start GR. +# Trace file with timestamp are created. +# 2. Size-based rotation: old file stays, new timestamped file appears. +# 3. Restart Group Replication: another new timestamped file is created. +# 4. Group Replication is ONLINE. +# 5. Cleanup. +################################################################################ + +--source include/have_group_replication_plugin.inc +--let $rpl_skip_group_replication_start= 1 +--source include/group_replication.inc + +--connection server1 + +--echo +--echo # 1. Set communication_debug_max_file_size and start GR. +--echo # Trace file with timestamp are created. + +--error 0,1 +--remove_file $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE +SET GLOBAL group_replication_communication_debug_max_file_size = 200; +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_ALL"; +--source include/start_and_bootstrap_group_replication.inc + +# Confirm file with timestamp was created. +perl; + use strict; + use File::Glob ':glob'; + my $dir = "$ENV{MYSQLTEST_VARDIR}/mysqld.1/data"; + my $deadline = time() + 30; + my @files; + while (time() < $deadline) { + @files = bsd_glob("$dir/GCS_DEBUG_TRACE.[0-9]*"); + last if @files; + select(undef, undef, undef, 0.5); + } + die "ERROR: no timestamped GCS_DEBUG_TRACE.* file appeared within 30s\n" + unless @files; + print "Initial trace file is timestamped, as expected\n"; + + die "ERROR: plain GCS_DEBUG_TRACE must not be created when max_file_size > 0\n" + if -f "$dir/GCS_DEBUG_TRACE"; + print "No plain GCS_DEBUG_TRACE created, as expected\n"; + + # Save first file path for step 2 + my $out = "$ENV{MYSQLTEST_VARDIR}/tmp/gcs_first_file.txt"; + open(my $fh, '>', $out) or die "Cannot write $out: $!\n"; + print $fh $files[0] . "\n"; + close $fh; +EOF + +--echo +--echo # 2. Size-based rotation: old file stays, new timestamped file created. + +# Confirm some new trace file was created due to rotation. +perl; + use strict; + use File::Glob ':glob'; + my $dir = "$ENV{MYSQLTEST_VARDIR}/mysqld.1/data"; + + # Retrieve first file saved in step 1 + my $in = "$ENV{MYSQLTEST_VARDIR}/tmp/gcs_first_file.txt"; + open(my $fh, '<', $in) or die "Cannot open $in: $!\n"; + my $first_file = <$fh>; + chomp $first_file; + close $fh; + + # Wait for a second timestamped file (rotation at 200-byte threshold) + my $deadline = time() + 30; + my @files; + while (time() < $deadline) { + @files = bsd_glob("$dir/GCS_DEBUG_TRACE.[0-9]*"); + last if scalar(@files) >= 2; + select(undef, undef, undef, 0.5); + } + die "ERROR: size-based rotation did not produce a second file within 30s\n" + unless scalar(@files) >= 2; + print "Size-based rotation created a new timestamped file, as expected\n"; + + # Old file must still be present -- no archiving or renaming + die "ERROR: original file '$first_file' was removed after rotation " . + "(archiving must not happen)\n" + unless -f $first_file; + print "Original file still exists after rotation (no archiving), as expected\n"; +EOF + +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; + +--echo +--echo # 3. Restart Group Replication -- a new timestamped file is created on start. + +# Not number of files before restart +perl; + use strict; + use File::Glob ':glob'; + my $dir = "$ENV{MYSQLTEST_VARDIR}/mysqld.1/data"; + my @files = bsd_glob("$dir/GCS_DEBUG_TRACE.[0-9]*"); + my $out = "$ENV{MYSQLTEST_VARDIR}/tmp/gcs_count_before_restart.txt"; + open(my $fh, '>', $out) or die "Cannot write $out: $!\n"; + print $fh scalar(@files) . "\n"; + close $fh; + print "Recorded file count before restart\n"; +EOF + +--source include/stop_group_replication.inc +--source include/start_and_bootstrap_group_replication.inc + +# Confirm more GCS_DEBUG_TRACE.* file exists post restart +perl; + use strict; + use File::Glob ':glob'; + my $in = "$ENV{MYSQLTEST_VARDIR}/tmp/gcs_count_before_restart.txt"; + open(my $fh, '<', $in) or die "Cannot open $in: $!\n"; + my $before = int(<$fh>); + close $fh; + + my $dir = "$ENV{MYSQLTEST_VARDIR}/mysqld.1/data"; + my $deadline = time() + 30; + my @files; + while (time() < $deadline) { + @files = bsd_glob("$dir/GCS_DEBUG_TRACE.[0-9]*"); + last if scalar(@files) > $before; + select(undef, undef, undef, 0.5); + } + die "ERROR: expected a new timestamped file on GR restart " . + "(before=$before, after=" . scalar(@files) . ")\n" + unless scalar(@files) > $before; + print "New timestamped file created on GR restart, as expected\n"; +EOF + +--echo +--echo # 4. Group Replication is ONLINE. + +--let $assert_text= Member is ONLINE after automatic rotation +--let $assert_cond= COUNT(*) = 1 FROM performance_schema.replication_group_members WHERE MEMBER_STATE = "ONLINE" +--source include/assert.inc + +--echo +--echo # 5. Cleanup. + +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +SET GLOBAL group_replication_communication_debug_max_file_size = 0; +--source include/stop_group_replication.inc +perl; + use strict; + use File::Glob ':glob'; + my $dir = "$ENV{MYSQLTEST_VARDIR}/mysqld.1/data"; + for my $f (bsd_glob("$dir/GCS_DEBUG_TRACE*")) { + unlink $f or warn "Could not remove $f: $!\n"; + } + unlink "$ENV{MYSQLTEST_VARDIR}/tmp/gcs_first_file.txt"; + unlink "$ENV{MYSQLTEST_VARDIR}/tmp/gcs_count_before_restart.txt"; + print "Cleanup complete\n"; +EOF + +--source include/group_replication_end.inc diff --git a/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_reopen.test b/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_reopen.test new file mode 100644 index 000000000000..3a97b606c999 --- /dev/null +++ b/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_reopen.test @@ -0,0 +1,63 @@ +################################################################################ +# +# This test case proves that when GCS_DEBUG_TRACE is moved or deleted while the +# fd is open, subsequent writes go to the newly created file. +# +# Test steps: +# 1. Start Group Replication. +# 2. Enable GCS_DEBUG_ALL tracing. +# 3. Remove GCS_DEBUG_TRACE. +# 4. Wait for stat auto-detection -- GCS_DEBUG_TRACE must reappear +# 5. Group Replication is ONLINE. +# 6. Cleanup. +################################################################################ + +--source include/have_group_replication_plugin.inc +--let $rpl_skip_group_replication_start= 1 +--source include/group_replication.inc + +--connection server1 + +--echo +--echo # 1. Start Group Replication. + +--error 0,1 +--remove_file $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE + +--source include/start_and_bootstrap_group_replication.inc + +--echo +--echo # 2. Enable GCS_DEBUG_ALL tracing. +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_ALL"; +--let $wait_condition = SELECT LENGTH(LOAD_FILE(CONCAT(@@datadir, 'GCS_DEBUG_TRACE'))) > 0 +--source include/wait_condition.inc + +--echo +--echo # 3. Remove GCS_DEBUG_TRACE. +--remove_file $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE + +--echo +--echo # 4. Wait for stat auto-detection -- GCS_DEBUG_TRACE must reappear + +# Wait for GCS_DEBUG_TRACE re-creation and logs to be written. +--let $wait_condition = SELECT LENGTH(LOAD_FILE(CONCAT(@@datadir, 'GCS_DEBUG_TRACE'))) > 0 +--source include/wait_condition.inc + +--file_exists $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE + +--echo +--echo # 5. Group Replication is ONLINE. + +--let $assert_text= Member is ONLINE after stale-fd detection and reopen +--let $assert_cond= COUNT(*) = 1 FROM performance_schema.replication_group_members WHERE MEMBER_STATE = "ONLINE" +--source include/assert.inc + +--echo +--echo # 6. cleanup + +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +--source include/stop_group_replication.inc + +--remove_file $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE + +--source include/group_replication_end.inc diff --git a/mysql-test/suite/group_replication/t/gr_set_option_during_stop.test b/mysql-test/suite/group_replication/t/gr_set_option_during_stop.test index 21613d9876ed..4b871b7e9f1c 100644 --- a/mysql-test/suite/group_replication/t/gr_set_option_during_stop.test +++ b/mysql-test/suite/group_replication/t/gr_set_option_during_stop.test @@ -63,6 +63,7 @@ INSERT INTO gr_options_that_cannot_be_change (name) WHERE VARIABLE_NAME LIKE 'group_replication_%' AND VARIABLE_NAME != 'group_replication_allow_local_lower_version_join' AND VARIABLE_NAME != 'group_replication_bootstrap_group' + AND VARIABLE_NAME != 'group_replication_communication_debug_max_file_size' AND VARIABLE_NAME != 'group_replication_communication_stack' AND VARIABLE_NAME != 'group_replication_consistency' AND VARIABLE_NAME != 'group_replication_exit_state_action' diff --git a/mysql-test/suite/group_replication/t/gr_show_global_and_session_variables.test b/mysql-test/suite/group_replication/t/gr_show_global_and_session_variables.test index bc7846541e09..e5054b9cba9b 100644 --- a/mysql-test/suite/group_replication/t/gr_show_global_and_session_variables.test +++ b/mysql-test/suite/group_replication/t/gr_show_global_and_session_variables.test @@ -30,7 +30,7 @@ --source include/start_and_bootstrap_group_replication.inc --source include/stop_group_replication.inc ---let $gr_var_count= 66 +--let $gr_var_count= 67 --echo --echo # Test#1: Basic check that there are $gr_var_count GR variables. diff --git a/mysql-test/suite/group_replication/t/gr_variables_default_values.test b/mysql-test/suite/group_replication/t/gr_variables_default_values.test index a4acac113b25..4d52ffc2778e 100644 --- a/mysql-test/suite/group_replication/t/gr_variables_default_values.test +++ b/mysql-test/suite/group_replication/t/gr_variables_default_values.test @@ -96,7 +96,7 @@ SET @@GLOBAL.group_replication_preemptive_garbage_collection = default; --let $saved_gr_xcom_ssl_accept_retries = `SELECT @@GLOBAL.group_replication_xcom_ssl_accept_retries;` # Total number of GR variables. ---let $total_gr_vars= 66 +--let $total_gr_vars= 67 --echo # --echo # Test Unit#1 @@ -303,6 +303,11 @@ SET @@SESSION.group_replication_consistency= default; --let $assert_cond= "[SELECT @@GLOBAL.group_replication_communication_debug_options]" = "GCS_DEBUG_NONE" --source include/assert.inc +# group_replication_communication_debug_max_file_size +--let $assert_text= Default group_replication_communication_debug_max_file_size is 0 +--let $assert_cond= "[SELECT @@GLOBAL.group_replication_communication_debug_max_file_size]" = 0 +--source include/assert.inc + # group_replication_unreachable_majority_timeout --let $assert_text= Default group_replication_unreachable_majority_timeout is 0 --let $assert_cond= "[SELECT @@GLOBAL.group_replication_unreachable_majority_timeout]" = 0 diff --git a/plugin/group_replication/include/plugin_variables.h b/plugin/group_replication/include/plugin_variables.h index e40d354254fe..b590a96d3006 100644 --- a/plugin/group_replication/include/plugin_variables.h +++ b/plugin/group_replication/include/plugin_variables.h @@ -261,6 +261,11 @@ struct plugin_options_variables { char *communication_debug_options_var; +#define DEFAULT_COMMUNICATION_DEBUG_MAX_FILE_SIZE 0UL +#define MIN_COMMUNICATION_DEBUG_MAX_FILE_SIZE 0UL +#define MAX_COMMUNICATION_DEBUG_MAX_FILE_SIZE ~0UL + ulong communication_debug_max_file_size_var; + const char *exit_state_actions[4] = {"READ_ONLY", "ABORT_SERVER", "OFFLINE_MODE", (char *)nullptr}; TYPELIB exit_state_actions_typelib_t = {3, "exit_state_actions_typelib_t", diff --git a/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h b/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h index 5954c3c8144f..fefd4e5b806f 100644 --- a/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h +++ b/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h @@ -835,6 +835,85 @@ class Gcs_file_sink : public Sink_interface { */ Gcs_file_sink(Gcs_file_sink &d); Gcs_file_sink &operator=(const Gcs_file_sink &d); + + /* + Full path of the currently open trace file. Updated on every + initialize/rotate/reopen so stale-fd detection checks the right path. + */ + std::string m_current_path; + /* + Open a new timestamped file (e.g. gcs_debug_trace.log.20260706T165057) + and switch m_fd to it. Needed for size-based rotation. + The old file is not deleted similar to binlog. + + @note Do not call during fd error like disk full etc. Existing file + descriptors are not closed if reopen fails, so user can continue to + write to existing fd. If reopen is successful existing fd are closed. + @note Manual deletion of logs is needed. + + @retval GCS_OK new file is open; m_current_path updated. + @retval GCS_NOK could not open the new file; old fd remains valid. + */ + enum_gcs_error rotate(); + + /* + Same as rotate(): open a new timestamped file and switch m_fd to it. + Called when the current file was moved or deleted externally. + + @refer rotate(), notes have been added in rotate, same applicable here + + @retval GCS_OK new file is open; m_current_path updated. + @retval GCS_NOK could not open/create the file; old fd remains valid. + */ + enum_gcs_error reopen(); + + /* + Maximum size, in bytes, of the file before it is automatically rotated. + 0 disables size-based rotation. + Static to avoid major changes to class like constructor param, + initialization etc. + */ + static size_t m_max_file_size; + + /* + Bytes written to the current file since it was last opened or rotated. + */ + size_t m_current_file_size{0}; + + /* + Timestamp, in microseconds, of the last stale-file check. + */ + ulonglong m_last_stat_check_time{0}; + + /* + Minimum time, in microseconds, between two stale-file checks. The check + costs two syscalls (my_stat on the path + my_fstat on the fd), so running + it on every trace line is wasteful. + */ + static constexpr ulonglong STAT_CHECK_INTERVAL_US = 100ULL * 1000ULL; + + /* Returns true if enough time elapsed to run the stale-file check again. */ + bool should_check_file(); + + /* + Timestamp, in microseconds, when the most recent rotation error was written + to the error log. A value of zero means that no error has been logged yet. + */ + ulonglong m_last_logged_error_time{0}; + + /* + Minimum time, in microseconds, between rotation error messages. + */ + static constexpr ulonglong ERROR_LOG_INTERVAL_US = 1ULL * 1000ULL * 1000ULL; + + /* Returns true if time has elapsed to log another error. */ + bool should_log_error(); + + public: + /* @refer m_max_file_size above */ + static void set_max_debug_file_size(size_t file_size) { + m_max_file_size = file_size; + } }; #endif /* XCOM_STANDALONE */ diff --git a/plugin/group_replication/libmysqlgcs/src/bindings/xcom/gcs_xcom_interface.cc b/plugin/group_replication/libmysqlgcs/src/bindings/xcom/gcs_xcom_interface.cc index 7817da988c4b..0b245f92a692 100644 --- a/plugin/group_replication/libmysqlgcs/src/bindings/xcom/gcs_xcom_interface.cc +++ b/plugin/group_replication/libmysqlgcs/src/bindings/xcom/gcs_xcom_interface.cc @@ -331,6 +331,32 @@ enum_gcs_error Gcs_xcom_interface::initialize( m_wait_for_ssl_init_cond.init( key_GCS_COND_Gcs_xcom_interface_m_wait_for_ssl_init_cond); +#ifndef XCOM_STANDALONE + { + /* + Either we can modify the constructors of Gcs_file_sink and pass the + parameters initialize_logging or we need to init set_max_debug_file_size + first before the log file is created. + Otherwise first log file will always be GCS_DEBUG_TRACE (without + timestamp) since set_max_debug_file_size has not been called yet, which + is not a big issue, but we can initialize + communication_debug_max_file_size first. + */ + const std::string *debug_max_file_size = + interface_params.get_parameter("communication_debug_max_file_size"); + if (debug_max_file_size != nullptr && !debug_max_file_size->empty()) { + try { + // This should not fail since debug_max_file_size is originally long + ulong max_file_size = std::stoul(*debug_max_file_size); + Gcs_file_sink::set_max_debug_file_size(max_file_size); + } catch (const std::exception &) { + MYSQL_GCS_LOG_ERROR( + "Failed to initialize GCS_DEBUG_TRACE log rotation. Could not read " + "parameter communication_debug_max_file_size."); + } + } + } +#endif /* Initialize logging sub-systems. */ diff --git a/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc b/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc index 94b86404aa9d..8c0549b0accd 100644 --- a/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc +++ b/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc @@ -37,10 +37,13 @@ #include "my_dir.h" #include "my_io.h" #include "my_sys.h" +#include "my_systime.h" #endif /* XCOM_STANDALONE */ #include "plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h" +size_t Gcs_file_sink::m_max_file_size = 0; + void *consumer_function(void *ptr); Gcs_async_buffer::Gcs_async_buffer(Sink_interface *sink, int buffer_size) @@ -343,6 +346,65 @@ enum_gcs_error Gcs_default_debugger::finalize() { return m_sink->finalize(); } Gcs_default_debugger *Gcs_debug_manager::m_debugger = nullptr; #ifndef XCOM_STANDALONE + +/* + Generate a conflict-safe rotation name for the active trace file. + Form: .YYYYMMDDTHHMMSS (UTC). + If the timestamped file already exists: + .YYYYMMDDTHHMMSS.1, .YYYYMMDDTHHMMSS.2, ... + up to .1000 + + If timestamp generation fails: + .1, .2, ... + up to .1000 + + @note Entire usage of gcs_generate_rotated_name is inside XCOM_STANDALONE. + + @returns + 0: on success and sets out_path; + 1: if all 1000 names are taken. +*/ +static int gcs_generate_rotated_name(const std::string &base_name, + std::string &out_path) { + char ts_buf[32]; + std::string rotation_base = base_name; + MY_STAT st_buf; + + const unsigned long long us = my_micro_time(); + const time_t sec = static_cast(us / 1000000); + + struct tm tm_info; + if (gmtime_r(&sec, &tm_info) != nullptr) { + const int len = + snprintf(ts_buf, sizeof(ts_buf), "%04d%02d%02dT%02d%02d%02d", + tm_info.tm_year + 1900, tm_info.tm_mon + 1, tm_info.tm_mday, + tm_info.tm_hour, tm_info.tm_min, tm_info.tm_sec); + + if (len >= 0 && static_cast(len) < sizeof(ts_buf)) { + rotation_base += "."; + rotation_base += ts_buf; + } + } + if (my_stat(rotation_base.c_str(), &st_buf, MYF(0)) == nullptr) { + out_path = rotation_base; + return 0; + } + + for (int n = 1; n <= 1000; ++n) { + std::string candidate = rotation_base + "." + std::to_string(n); + + if (my_stat(candidate.c_str(), &st_buf, MYF(0)) == nullptr) { + out_path = candidate; + return 0; + } + } + + MYSQL_GCS_LOG_ERROR("GCS debug trace: all rotation names for '" + << rotation_base.c_str() + << "' are taken. Cannot rotate."); + return 1; +} + Gcs_file_sink::Gcs_file_sink(const std::string &file_name, const std::string &dir_name) : m_fd(0), @@ -402,6 +464,18 @@ enum_gcs_error Gcs_file_sink::initialize() { return GCS_NOK; } } + /* + If size-based rotation is enabled, open a new timestamped file so each + GR run is cleanly separated. + */ + if (m_max_file_size > 0) { + std::string new_name; + if (gcs_generate_rotated_name(file_name_buffer, new_name) == 0) { + strncpy(file_name_buffer, new_name.c_str(), FN_REFLEN - 1); + file_name_buffer[FN_REFLEN - 1] = '\0'; + } + /* If name generation fails fall through and open canonical (appends). */ + } if ((m_fd = my_create(file_name_buffer, 0640, O_CREAT | O_WRONLY | O_APPEND, MYF(0))) < 0) { @@ -416,6 +490,8 @@ enum_gcs_error Gcs_file_sink::initialize() { return GCS_NOK; } + m_current_path = file_name_buffer; + m_current_file_size = 0; m_initialized = true; return GCS_OK; @@ -432,6 +508,131 @@ enum_gcs_error Gcs_file_sink::finalize() { return GCS_OK; } +bool Gcs_file_sink::should_log_error() { + const ulonglong now = my_micro_time(); + if (now - m_last_logged_error_time < ERROR_LOG_INTERVAL_US) { + return false; + } + m_last_logged_error_time = now; + return true; +} + +bool Gcs_file_sink::should_check_file() { + const ulonglong now = my_micro_time(); + if (now - m_last_stat_check_time < STAT_CHECK_INTERVAL_US) { + return false; + } + m_last_stat_check_time = now; + return true; +} + +/* + Since reopen() is called only when file is renamed/deleted, there is no need + to suppress logging to avoid flooding since stats itself is controlled by + should_check_file(). +*/ +enum_gcs_error Gcs_file_sink::reopen() { + if (!m_initialized) return GCS_OK; + char file_name_buffer[FN_REFLEN]; + + if (get_file_name(file_name_buffer)) { + MYSQL_GCS_LOG_ERROR("GCS debug trace reopen: error validating file name '" + << m_file_name << "'."); + return GCS_NOK; + } + + std::string new_name = file_name_buffer; + if (m_max_file_size > 0) { + if (gcs_generate_rotated_name(file_name_buffer, new_name) != 0) { + MYSQL_GCS_LOG_ERROR( + "GCS debug trace reopen: could not generate new file " + "name for '" + << file_name_buffer << "'."); + return GCS_NOK; + } + } + + File new_fd = + my_create(new_name.c_str(), 0640, O_CREAT | O_WRONLY | O_APPEND, MYF(0)); + /* + We have not closed existing file descriptors, so even if my_create fails + call to existing fd are still safe. write continues to write to detached + file descriptor without any error. + We keep the old fd active in case of error. + */ + if (new_fd < 0) { + int errno_gcs = 0; +#if defined(_WIN32) + errno_gcs = WSAGetLastError(); +#else + errno_gcs = errno; +#endif + MYSQL_GCS_LOG_ERROR("GCS debug trace reopen: error opening file '" + << new_name.c_str() << "': " << strerror(errno_gcs) + << "."); + return GCS_NOK; + } + + /* ignore any error, we have new FD */ + my_sync(m_fd, MYF(0)); + my_close(m_fd, MYF(0)); + m_fd = new_fd; + m_current_path = new_name; + m_initialized = true; + m_current_file_size = 0; + + return GCS_OK; +} + +enum_gcs_error Gcs_file_sink::rotate() { + if (!m_initialized) return GCS_OK; + char file_name_buffer[FN_REFLEN]; + + if (get_file_name(file_name_buffer)) { + if (should_log_error()) { + MYSQL_GCS_LOG_ERROR("GCS debug trace rotate: error validating file name '" + << m_file_name << "'."); + } + return GCS_NOK; + } + + std::string new_name = file_name_buffer; + if (gcs_generate_rotated_name(file_name_buffer, new_name) != 0) { + if (should_log_error()) { + MYSQL_GCS_LOG_ERROR( + "GCS debug trace rotate: could not generate new file " + "name for '" + << file_name_buffer << "'."); + } + return GCS_NOK; + } + + File new_fd = + my_create(new_name.c_str(), 0640, O_CREAT | O_WRONLY | O_APPEND, MYF(0)); + if (new_fd < 0) { + int errno_gcs = 0; +#if defined(_WIN32) + errno_gcs = WSAGetLastError(); +#else + errno_gcs = errno; +#endif + if (should_log_error()) { + MYSQL_GCS_LOG_ERROR("GCS debug trace rotate: error opening new file '" + << new_name << "': " << strerror(errno_gcs) << "."); + } + return GCS_NOK; + } + + my_sync(m_fd, MYF(0)); + my_close(m_fd, MYF(0)); + m_fd = new_fd; + m_current_path = new_name; + m_initialized = true; + m_current_file_size = 0; + + return GCS_OK; +} + void Gcs_file_sink::log_event(const std::string &message) { log_event(message.c_str(), message.length()); } @@ -441,6 +642,30 @@ void Gcs_file_sink::log_event(const char *message, size_t message_size) { written = my_write(m_fd, (const uchar *)message, message_size, MYF(0)); + if (written != MY_FILE_ERROR) { + m_current_file_size += written; + if (m_max_file_size > 0 && m_current_file_size >= m_max_file_size) { + rotate(); + } else if (should_check_file()) { + /* + Check if the current trace file was moved or deleted. + It was noted when the trace file was moved/deleted, no debug trace + was later generated in the session. + + Note: the check is done after the write (and only once per + should_check_file() interval), so if the file is removed between our + write and this check, the line just written goes to the orphaned inode + and some logs are lost until reopen() creates a new file. + */ + MY_STAT path_st, fd_st; + bool path_gone = + (my_stat(m_current_path.c_str(), &path_st, MYF(0)) == nullptr); + bool ino_changed = !path_gone && my_fstat(m_fd, &fd_st) == 0 && + path_st.st_ino != fd_st.st_ino; + if (path_gone || ino_changed) reopen(); + } + } + if (written == MY_FILE_ERROR) { int errno_gcs = 0; #if defined(_WIN32) @@ -455,12 +680,15 @@ void Gcs_file_sink::log_event(const char *message, size_t message_size) { const std::string Gcs_file_sink::get_information() const { std::string invalid("invalid"); - char file_name_buffer[FN_REFLEN]; if (!m_initialized) return invalid; - - if (get_file_name(file_name_buffer)) return invalid; - - return std::string(file_name_buffer); + /* + Although m_current_path can now be mutated from a different thread by + rotate()/reopen() during logging, it is safe to read it here without a + mutex: get_information() is only called from the initialization path (see + Gcs_xcom_interface::initialize_logging), before any concurrent logging + thread can rotate/reopen the file. + */ + return std::string(m_current_path); } #endif /* XCOM_STANDALONE */ diff --git a/plugin/group_replication/src/plugin.cc b/plugin/group_replication/src/plugin.cc index f854189a69e6..40d979bc1449 100644 --- a/plugin/group_replication/src/plugin.cc +++ b/plugin/group_replication/src/plugin.cc @@ -2750,6 +2750,17 @@ int build_gcs_parameters(Gcs_interface_parameters &gcs_module_parameters) { gcs_module_parameters.add_parameter("communication_debug_path", mysql_real_data_home); + /* + Maximum size, in bytes, of the GCS debug trace file before it is + automatically rotated. 0 disables size-based rotation. Read once here, + at Group Replication start -- like communication_debug_file/_path above. + + @note changing this sysvar takes effect at the next Group Replication start + */ + gcs_module_parameters.add_parameter( + "communication_debug_max_file_size", + std::to_string(ov.communication_debug_max_file_size_var)); + sv.deinit(); return result; } @@ -4970,6 +4981,21 @@ static MYSQL_SYSVAR_STR( "GCS_DEBUG_NONE" /* default */ ); +static MYSQL_SYSVAR_ULONG( + communication_debug_max_file_size, /* name */ + ov.communication_debug_max_file_size_var, /* var */ + PLUGIN_VAR_OPCMDARG | PLUGIN_VAR_PERSIST_AS_READ_ONLY, /* optional var */ + "Maximum size, in bytes, of the GCS debug trace file (GCS_DEBUG_TRACE) " + "before it is automatically rotated. 0 disables size-based rotation. " + "Takes effect at the next Group Replication start.", + nullptr, /* check func. */ + nullptr, /* update func. */ + DEFAULT_COMMUNICATION_DEBUG_MAX_FILE_SIZE, /* default */ + MIN_COMMUNICATION_DEBUG_MAX_FILE_SIZE, /* min */ + MAX_COMMUNICATION_DEBUG_MAX_FILE_SIZE, /* max */ + 0 /* block */ +); + static MYSQL_SYSVAR_ENUM(exit_state_action, /* name */ ov.exit_state_action_var, /* var */ PLUGIN_VAR_OPCMDARG | @@ -5470,6 +5496,7 @@ static SYS_VAR *group_replication_system_vars[] = { MYSQL_SYSVAR(flow_control_applier_threshold), MYSQL_SYSVAR(transaction_size_limit), MYSQL_SYSVAR(communication_debug_options), + MYSQL_SYSVAR(communication_debug_max_file_size), MYSQL_SYSVAR(exit_state_action), MYSQL_SYSVAR(autorejoin_tries), MYSQL_SYSVAR(unreachable_majority_timeout), From d3999594a0ef25bb9edfc50853b868746dbd3459 Mon Sep 17 00:00:00 2001 From: Jaideep Karande Date: Thu, 2 Jul 2026 01:18:44 +0530 Subject: [PATCH 09/32] PS-11217: GCS_DEBUG_TRACE log not possible to rotate/archive without restarting replication https://perconadev.atlassian.net/browse/PS-11217 Problem 1: Group Replication's GCS_DEBUG_TRACE file descriptor was opened once and never revisited. So, if an operator externally moved or removed the file output kept going to the stale/detached file descriptor with no error reported. The only fix was a full Group Replication restart. Problem 2: GCS_DEBUG_TRACE are always written to a same file, which mean file can grow huge and logs of all MySQL sessions are written in a same file, no automatic rotation of logs is available. Resolution: If the file is gone (moved/deleted) reopen the file. Log Rotation: Size-based: new sysvar group_replication_communication_debug_max_file_size (bytes, default 0 = disabled) caps the trace file size, like max_binlog_size. Additinally timestamp is added to the file name that can help in opening the right log file if huge logs are generated. Similarly during every START GROUP_REPLICATION logs will be written to a new trace file. --- .../r/gr_gcs_debug_trace_auto_rotation.result | 34 +++ .../r/gr_gcs_debug_trace_reopen.result | 23 ++ .../r/gr_persist_only_variables.result | 10 +- .../r/gr_persist_variables.result | 10 +- .../r/gr_set_option_during_stop.result | 3 + ...r_show_global_and_session_variables.result | 4 +- .../r/gr_variables_default_values.result | 5 +- .../r/gr_variables_privileges.result | 3 + .../t/gr_gcs_debug_trace_auto_rotation.test | 168 +++++++++++++ .../t/gr_gcs_debug_trace_reopen.test | 63 +++++ .../t/gr_set_option_during_stop.test | 1 + .../gr_show_global_and_session_variables.test | 2 +- .../t/gr_variables_default_values.test | 7 +- .../include/plugin_variables.h | 5 + .../include/mysql/gcs/gcs_logging_system.h | 79 ++++++ .../src/bindings/xcom/gcs_xcom_interface.cc | 26 ++ .../src/interface/gcs_logging_system.cc | 238 +++++++++++++++++- plugin/group_replication/src/plugin.cc | 27 ++ 18 files changed, 689 insertions(+), 19 deletions(-) create mode 100644 mysql-test/suite/group_replication/r/gr_gcs_debug_trace_auto_rotation.result create mode 100644 mysql-test/suite/group_replication/r/gr_gcs_debug_trace_reopen.result create mode 100644 mysql-test/suite/group_replication/t/gr_gcs_debug_trace_auto_rotation.test create mode 100644 mysql-test/suite/group_replication/t/gr_gcs_debug_trace_reopen.test diff --git a/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_auto_rotation.result b/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_auto_rotation.result new file mode 100644 index 000000000000..d334985d3859 --- /dev/null +++ b/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_auto_rotation.result @@ -0,0 +1,34 @@ +include/group_replication.inc +Warnings: +Note #### Sending passwords in plain text without SSL/TLS is extremely insecure. +Note #### Storing MySQL user name or password information in the connection metadata repository is not secure and is therefore not recommended. Please consider using the USER and PASSWORD connection options for START REPLICA; see the 'START REPLICA Syntax' in the MySQL Manual for more information. +[connection server1] + +# 1. Set communication_debug_max_file_size and start GR. +# Trace file with timestamp are created. +SET GLOBAL group_replication_communication_debug_max_file_size = 200; +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_ALL"; +include/start_and_bootstrap_group_replication.inc +Initial trace file is timestamped, as expected +No plain GCS_DEBUG_TRACE created, as expected + +# 2. Size-based rotation: old file stays, new timestamped file created. +Size-based rotation created a new timestamped file, as expected +Original file still exists after rotation (no archiving), as expected +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; + +# 3. Restart Group Replication -- a new timestamped file is created on start. +Recorded file count before restart +include/stop_group_replication.inc +include/start_and_bootstrap_group_replication.inc +New timestamped file created on GR restart, as expected + +# 4. Group Replication is ONLINE. +include/assert.inc [Member is ONLINE after automatic rotation] + +# 5. Cleanup. +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +SET GLOBAL group_replication_communication_debug_max_file_size = 0; +include/stop_group_replication.inc +Cleanup complete +include/group_replication_end.inc diff --git a/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_reopen.result b/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_reopen.result new file mode 100644 index 000000000000..ff747493fc4b --- /dev/null +++ b/mysql-test/suite/group_replication/r/gr_gcs_debug_trace_reopen.result @@ -0,0 +1,23 @@ +include/group_replication.inc +Warnings: +Note #### Sending passwords in plain text without SSL/TLS is extremely insecure. +Note #### Storing MySQL user name or password information in the connection metadata repository is not secure and is therefore not recommended. Please consider using the USER and PASSWORD connection options for START REPLICA; see the 'START REPLICA Syntax' in the MySQL Manual for more information. +[connection server1] + +# 1. Start Group Replication. +include/start_and_bootstrap_group_replication.inc + +# 2. Enable GCS_DEBUG_ALL tracing. +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_ALL"; + +# 3. Remove GCS_DEBUG_TRACE. + +# 4. Wait for stat auto-detection -- GCS_DEBUG_TRACE must reappear + +# 5. Group Replication is ONLINE. +include/assert.inc [Member is ONLINE after stale-fd detection and reopen] + +# 6. cleanup +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +include/stop_group_replication.inc +include/group_replication_end.inc diff --git a/mysql-test/suite/group_replication/r/gr_persist_only_variables.result b/mysql-test/suite/group_replication/r/gr_persist_only_variables.result index e164382f9978..440002b7ffe7 100644 --- a/mysql-test/suite/group_replication/r/gr_persist_only_variables.result +++ b/mysql-test/suite/group_replication/r/gr_persist_only_variables.result @@ -29,6 +29,7 @@ SET PERSIST_ONLY group_replication_bootstrap_group = @@GLOBAL.group_replication_ SET PERSIST_ONLY group_replication_certification_loop_chunk_size = @@GLOBAL.group_replication_certification_loop_chunk_size; SET PERSIST_ONLY group_replication_certification_loop_sleep_time = @@GLOBAL.group_replication_certification_loop_sleep_time; SET PERSIST_ONLY group_replication_clone_threshold = @@GLOBAL.group_replication_clone_threshold; +SET PERSIST_ONLY group_replication_communication_debug_max_file_size = @@GLOBAL.group_replication_communication_debug_max_file_size; SET PERSIST_ONLY group_replication_communication_debug_options = @@GLOBAL.group_replication_communication_debug_options; SET PERSIST_ONLY group_replication_communication_max_message_size = @@GLOBAL.group_replication_communication_max_message_size; SET PERSIST_ONLY group_replication_communication_stack = @@GLOBAL.group_replication_communication_stack; @@ -86,7 +87,7 @@ SET PERSIST_ONLY group_replication_view_change_uuid = @@GLOBAL.group_replication SET PERSIST_ONLY group_replication_xcom_ssl_accept_retries = @@GLOBAL.group_replication_xcom_ssl_accept_retries; SET PERSIST_ONLY group_replication_xcom_ssl_socket_timeout = @@GLOBAL.group_replication_xcom_ssl_socket_timeout; -include/assert.inc ['Expect 64 persisted variables.'] +include/assert.inc ['Expect 65 persisted variables.'] ############################################################ # 2. Restart server, it must bootstrap the group and preserve @@ -95,9 +96,9 @@ include/assert.inc ['Expect 64 persisted variables.'] include/rpl/reconnect.inc include/gr_wait_for_member_state.inc -include/assert.inc ['Expect 64 persisted variables in persisted_variables table.'] -include/assert.inc ['Expect 63 variables which last value was set through SET PERSIST.'] -include/assert.inc ['Expect 63 persisted variables with matching persisted and global values.'] +include/assert.inc ['Expect 65 persisted variables in persisted_variables table.'] +include/assert.inc ['Expect 64 variables which last value was set through SET PERSIST.'] +include/assert.inc ['Expect 64 persisted variables with matching persisted and global values.'] ############################################################ # 3. Test RESET PERSIST IF EXISTS. @@ -110,6 +111,7 @@ RESET PERSIST IF EXISTS group_replication_bootstrap_group; RESET PERSIST IF EXISTS group_replication_certification_loop_chunk_size; RESET PERSIST IF EXISTS group_replication_certification_loop_sleep_time; RESET PERSIST IF EXISTS group_replication_clone_threshold; +RESET PERSIST IF EXISTS group_replication_communication_debug_max_file_size; RESET PERSIST IF EXISTS group_replication_communication_debug_options; RESET PERSIST IF EXISTS group_replication_communication_max_message_size; RESET PERSIST IF EXISTS group_replication_communication_stack; diff --git a/mysql-test/suite/group_replication/r/gr_persist_variables.result b/mysql-test/suite/group_replication/r/gr_persist_variables.result index 56d096d823c1..de8ff976a54b 100644 --- a/mysql-test/suite/group_replication/r/gr_persist_variables.result +++ b/mysql-test/suite/group_replication/r/gr_persist_variables.result @@ -29,6 +29,7 @@ SET PERSIST group_replication_bootstrap_group = @@GLOBAL.group_replication_boots SET PERSIST group_replication_certification_loop_chunk_size = @@GLOBAL.group_replication_certification_loop_chunk_size; SET PERSIST group_replication_certification_loop_sleep_time = @@GLOBAL.group_replication_certification_loop_sleep_time; SET PERSIST group_replication_clone_threshold = @@GLOBAL.group_replication_clone_threshold; +SET PERSIST group_replication_communication_debug_max_file_size = @@GLOBAL.group_replication_communication_debug_max_file_size; SET PERSIST group_replication_communication_debug_options = @@GLOBAL.group_replication_communication_debug_options; SET PERSIST group_replication_communication_max_message_size = @@GLOBAL.group_replication_communication_max_message_size; SET PERSIST group_replication_communication_stack = @@GLOBAL.group_replication_communication_stack; @@ -88,7 +89,7 @@ Warning 1681 'group_replication_view_change_uuid' is deprecated and will be remo SET PERSIST group_replication_xcom_ssl_accept_retries = @@GLOBAL.group_replication_xcom_ssl_accept_retries; SET PERSIST group_replication_xcom_ssl_socket_timeout = @@GLOBAL.group_replication_xcom_ssl_socket_timeout; -include/assert.inc ['Expect 64 persisted variables.'] +include/assert.inc ['Expect 65 persisted variables.'] ############################################################ # 2. Restart server, it must bootstrap the group and preserve @@ -97,9 +98,9 @@ include/assert.inc ['Expect 64 persisted variables.'] include/rpl/reconnect.inc include/gr_wait_for_member_state.inc -include/assert.inc ['Expect 64 persisted variables in persisted_variables table.'] -include/assert.inc ['Expect 63 variables which last value was set through SET PERSIST.'] -include/assert.inc ['Expect 63 variables which last value was set through SET PERSIST is equal to its global value.'] +include/assert.inc ['Expect 65 persisted variables in persisted_variables table.'] +include/assert.inc ['Expect 64 variables which last value was set through SET PERSIST.'] +include/assert.inc ['Expect 64 variables which last value was set through SET PERSIST is equal to its global value.'] ############################################################ # 3. Test RESET PERSIST. @@ -112,6 +113,7 @@ RESET PERSIST group_replication_bootstrap_group; RESET PERSIST group_replication_certification_loop_chunk_size; RESET PERSIST group_replication_certification_loop_sleep_time; RESET PERSIST group_replication_clone_threshold; +RESET PERSIST group_replication_communication_debug_max_file_size; RESET PERSIST group_replication_communication_debug_options; RESET PERSIST group_replication_communication_max_message_size; RESET PERSIST group_replication_communication_stack; diff --git a/mysql-test/suite/group_replication/r/gr_set_option_during_stop.result b/mysql-test/suite/group_replication/r/gr_set_option_during_stop.result index 61d1470a1754..0dce33026aca 100644 --- a/mysql-test/suite/group_replication/r/gr_set_option_during_stop.result +++ b/mysql-test/suite/group_replication/r/gr_set_option_during_stop.result @@ -29,6 +29,7 @@ INSERT INTO gr_options_that_cannot_be_change (name) SELECT VARIABLE_NAME FROM performance_schema.global_variables WHERE VARIABLE_NAME LIKE 'group_replication_%' AND VARIABLE_NAME != 'group_replication_bootstrap_group' + AND VARIABLE_NAME != 'group_replication_communication_debug_max_file_size' AND VARIABLE_NAME != 'group_replication_communication_stack' AND VARIABLE_NAME != 'group_replication_consistency' AND VARIABLE_NAME != 'group_replication_exit_state_action' @@ -182,6 +183,8 @@ SET @value= @@GLOBAL.group_replication_certification_loop_chunk_size; SET @@GLOBAL.group_replication_certification_loop_chunk_size= @value; SET @value= @@GLOBAL.group_replication_certification_loop_sleep_time; SET @@GLOBAL.group_replication_certification_loop_sleep_time= @value; +SET @value= @@GLOBAL.group_replication_communication_debug_max_file_size; +SET @@GLOBAL.group_replication_communication_debug_max_file_size= @value; SET @value= @@GLOBAL.group_replication_communication_stack; SET @@GLOBAL.group_replication_communication_stack= @value; SET @value= @@GLOBAL.group_replication_consistency; diff --git a/mysql-test/suite/group_replication/r/gr_show_global_and_session_variables.result b/mysql-test/suite/group_replication/r/gr_show_global_and_session_variables.result index c2e586885e77..eb160202589b 100644 --- a/mysql-test/suite/group_replication/r/gr_show_global_and_session_variables.result +++ b/mysql-test/suite/group_replication/r/gr_show_global_and_session_variables.result @@ -7,8 +7,8 @@ Note #### Storing MySQL user name or password information in the connection meta include/start_and_bootstrap_group_replication.inc include/stop_group_replication.inc -# Test#1: Basic check that there are 65 GR variables. -include/assert.inc [There are 65 GR variables at present.] +# Test#1: Basic check that there are 66 GR variables. +include/assert.inc [There are 66 GR variables at present.] # Test#2: Verify group replication related variables at GLOBAL scope. SET @@SESSION.group_replication_auto_increment_increment= 7; diff --git a/mysql-test/suite/group_replication/r/gr_variables_default_values.result b/mysql-test/suite/group_replication/r/gr_variables_default_values.result index 10cddaff13ac..1bb1b064237d 100644 --- a/mysql-test/suite/group_replication/r/gr_variables_default_values.result +++ b/mysql-test/suite/group_replication/r/gr_variables_default_values.result @@ -28,9 +28,9 @@ include/stop_group_replication.inc # # Test Unit#1 # Set global/session group replication variables to default. -# Curently there are 65 group replication variables. +# Curently there are 66 group replication variables. # -include/assert.inc [There are 65 GR variables at present.] +include/assert.inc [There are 66 GR variables at present.] SET @@GLOBAL.group_replication_auto_increment_increment= default; ERROR 42000: Variable 'group_replication_auto_increment_increment' can't be set to the value of 'DEFAULT' SET @@GLOBAL.group_replication_compression_threshold= default; @@ -116,6 +116,7 @@ include/assert.inc [Default group_replication_ssl_mode is REQUIRED] include/assert.inc [Default group_replication_start_on_boot is ON/1] include/assert.inc [Default group_replication_transaction_size_limit is 150000000] include/assert.inc [Default group_replication_communication_debug_options is "GCS_DEBUG_NONE"] +include/assert.inc [Default group_replication_communication_debug_max_file_size is 0] include/assert.inc [Default group_replication_unreachable_majority_timeout is 0] include/assert.inc [Default group_replication_member_weight is 50] include/assert.inc [Default group_replication_recovery_public_key_path is ""(EMPTY)] diff --git a/mysql-test/suite/group_replication/r/gr_variables_privileges.result b/mysql-test/suite/group_replication/r/gr_variables_privileges.result index f91558223632..5d715491187e 100644 --- a/mysql-test/suite/group_replication/r/gr_variables_privileges.result +++ b/mysql-test/suite/group_replication/r/gr_variables_privileges.result @@ -41,6 +41,8 @@ SET GLOBAL group_replication_certification_loop_sleep_time = @@GLOBAL.group_repl ERROR 42000: Access denied; you need (at least one of) the SUPER or SYSTEM_VARIABLES_ADMIN privilege(s) for this operation SET GLOBAL group_replication_clone_threshold = @@GLOBAL.group_replication_clone_threshold; ERROR 42000: Access denied; you need (at least one of) the SUPER or SYSTEM_VARIABLES_ADMIN privilege(s) for this operation +SET GLOBAL group_replication_communication_debug_max_file_size = @@GLOBAL.group_replication_communication_debug_max_file_size; +ERROR 42000: Access denied; you need (at least one of) the SUPER or SYSTEM_VARIABLES_ADMIN privilege(s) for this operation SET GLOBAL group_replication_communication_debug_options = @@GLOBAL.group_replication_communication_debug_options; ERROR 42000: Access denied; you need (at least one of) the SUPER or SYSTEM_VARIABLES_ADMIN privilege(s) for this operation SET GLOBAL group_replication_communication_max_message_size = @@GLOBAL.group_replication_communication_max_message_size; @@ -177,6 +179,7 @@ SET GLOBAL group_replication_bootstrap_group = @@GLOBAL.group_replication_bootst SET GLOBAL group_replication_certification_loop_chunk_size = @@GLOBAL.group_replication_certification_loop_chunk_size; SET GLOBAL group_replication_certification_loop_sleep_time = @@GLOBAL.group_replication_certification_loop_sleep_time; SET GLOBAL group_replication_clone_threshold = @@GLOBAL.group_replication_clone_threshold; +SET GLOBAL group_replication_communication_debug_max_file_size = @@GLOBAL.group_replication_communication_debug_max_file_size; SET GLOBAL group_replication_communication_debug_options = @@GLOBAL.group_replication_communication_debug_options; SET GLOBAL group_replication_communication_max_message_size = @@GLOBAL.group_replication_communication_max_message_size; SET GLOBAL group_replication_communication_stack = @@GLOBAL.group_replication_communication_stack; diff --git a/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_auto_rotation.test b/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_auto_rotation.test new file mode 100644 index 000000000000..5b7226831906 --- /dev/null +++ b/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_auto_rotation.test @@ -0,0 +1,168 @@ +################################################################################ +# +# This test confirms AUTOMATIC rotation of the GCS_DEBUG_TRACE file. +# +# When group_replication_communication_debug_max_file_size > 0: +# - Every file (including the first) gets a timestamp in its name: +# GCS_DEBUG_TRACE.YYYYMMDDTHHMMSS (UTC) +# - Rotation creates a new timestamped file; the old file is left in place +# (no archiving/renaming -- same model as binary logs). +# - Each START GROUP_REPLICATION also opens a fresh timestamped file. +# +# Test steps: +# 1. Set communication_debug_max_file_size and start GR. +# Trace file with timestamp are created. +# 2. Size-based rotation: old file stays, new timestamped file appears. +# 3. Restart Group Replication: another new timestamped file is created. +# 4. Group Replication is ONLINE. +# 5. Cleanup. +################################################################################ + +--source include/have_group_replication_plugin.inc +--let $rpl_skip_group_replication_start= 1 +--source include/group_replication.inc + +--connection server1 + +--echo +--echo # 1. Set communication_debug_max_file_size and start GR. +--echo # Trace file with timestamp are created. + +--error 0,1 +--remove_file $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE +SET GLOBAL group_replication_communication_debug_max_file_size = 200; +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_ALL"; +--source include/start_and_bootstrap_group_replication.inc + +# Confirm file with timestamp was created. +perl; + use strict; + use File::Glob ':glob'; + my $dir = "$ENV{MYSQLTEST_VARDIR}/mysqld.1/data"; + my $deadline = time() + 30; + my @files; + while (time() < $deadline) { + @files = bsd_glob("$dir/GCS_DEBUG_TRACE.[0-9]*"); + last if @files; + select(undef, undef, undef, 0.5); + } + die "ERROR: no timestamped GCS_DEBUG_TRACE.* file appeared within 30s\n" + unless @files; + print "Initial trace file is timestamped, as expected\n"; + + die "ERROR: plain GCS_DEBUG_TRACE must not be created when max_file_size > 0\n" + if -f "$dir/GCS_DEBUG_TRACE"; + print "No plain GCS_DEBUG_TRACE created, as expected\n"; + + # Save first file path for step 2 + my $out = "$ENV{MYSQLTEST_VARDIR}/tmp/gcs_first_file.txt"; + open(my $fh, '>', $out) or die "Cannot write $out: $!\n"; + print $fh $files[0] . "\n"; + close $fh; +EOF + +--echo +--echo # 2. Size-based rotation: old file stays, new timestamped file created. + +# Confirm some new trace file was created due to rotation. +perl; + use strict; + use File::Glob ':glob'; + my $dir = "$ENV{MYSQLTEST_VARDIR}/mysqld.1/data"; + + # Retrieve first file saved in step 1 + my $in = "$ENV{MYSQLTEST_VARDIR}/tmp/gcs_first_file.txt"; + open(my $fh, '<', $in) or die "Cannot open $in: $!\n"; + my $first_file = <$fh>; + chomp $first_file; + close $fh; + + # Wait for a second timestamped file (rotation at 200-byte threshold) + my $deadline = time() + 30; + my @files; + while (time() < $deadline) { + @files = bsd_glob("$dir/GCS_DEBUG_TRACE.[0-9]*"); + last if scalar(@files) >= 2; + select(undef, undef, undef, 0.5); + } + die "ERROR: size-based rotation did not produce a second file within 30s\n" + unless scalar(@files) >= 2; + print "Size-based rotation created a new timestamped file, as expected\n"; + + # Old file must still be present -- no archiving or renaming + die "ERROR: original file '$first_file' was removed after rotation " . + "(archiving must not happen)\n" + unless -f $first_file; + print "Original file still exists after rotation (no archiving), as expected\n"; +EOF + +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; + +--echo +--echo # 3. Restart Group Replication -- a new timestamped file is created on start. + +# Not number of files before restart +perl; + use strict; + use File::Glob ':glob'; + my $dir = "$ENV{MYSQLTEST_VARDIR}/mysqld.1/data"; + my @files = bsd_glob("$dir/GCS_DEBUG_TRACE.[0-9]*"); + my $out = "$ENV{MYSQLTEST_VARDIR}/tmp/gcs_count_before_restart.txt"; + open(my $fh, '>', $out) or die "Cannot write $out: $!\n"; + print $fh scalar(@files) . "\n"; + close $fh; + print "Recorded file count before restart\n"; +EOF + +--source include/stop_group_replication.inc +--source include/start_and_bootstrap_group_replication.inc + +# Confirm more GCS_DEBUG_TRACE.* file exists post restart +perl; + use strict; + use File::Glob ':glob'; + my $in = "$ENV{MYSQLTEST_VARDIR}/tmp/gcs_count_before_restart.txt"; + open(my $fh, '<', $in) or die "Cannot open $in: $!\n"; + my $before = int(<$fh>); + close $fh; + + my $dir = "$ENV{MYSQLTEST_VARDIR}/mysqld.1/data"; + my $deadline = time() + 30; + my @files; + while (time() < $deadline) { + @files = bsd_glob("$dir/GCS_DEBUG_TRACE.[0-9]*"); + last if scalar(@files) > $before; + select(undef, undef, undef, 0.5); + } + die "ERROR: expected a new timestamped file on GR restart " . + "(before=$before, after=" . scalar(@files) . ")\n" + unless scalar(@files) > $before; + print "New timestamped file created on GR restart, as expected\n"; +EOF + +--echo +--echo # 4. Group Replication is ONLINE. + +--let $assert_text= Member is ONLINE after automatic rotation +--let $assert_cond= COUNT(*) = 1 FROM performance_schema.replication_group_members WHERE MEMBER_STATE = "ONLINE" +--source include/assert.inc + +--echo +--echo # 5. Cleanup. + +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +SET GLOBAL group_replication_communication_debug_max_file_size = 0; +--source include/stop_group_replication.inc +perl; + use strict; + use File::Glob ':glob'; + my $dir = "$ENV{MYSQLTEST_VARDIR}/mysqld.1/data"; + for my $f (bsd_glob("$dir/GCS_DEBUG_TRACE*")) { + unlink $f or warn "Could not remove $f: $!\n"; + } + unlink "$ENV{MYSQLTEST_VARDIR}/tmp/gcs_first_file.txt"; + unlink "$ENV{MYSQLTEST_VARDIR}/tmp/gcs_count_before_restart.txt"; + print "Cleanup complete\n"; +EOF + +--source include/group_replication_end.inc diff --git a/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_reopen.test b/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_reopen.test new file mode 100644 index 000000000000..3a97b606c999 --- /dev/null +++ b/mysql-test/suite/group_replication/t/gr_gcs_debug_trace_reopen.test @@ -0,0 +1,63 @@ +################################################################################ +# +# This test case proves that when GCS_DEBUG_TRACE is moved or deleted while the +# fd is open, subsequent writes go to the newly created file. +# +# Test steps: +# 1. Start Group Replication. +# 2. Enable GCS_DEBUG_ALL tracing. +# 3. Remove GCS_DEBUG_TRACE. +# 4. Wait for stat auto-detection -- GCS_DEBUG_TRACE must reappear +# 5. Group Replication is ONLINE. +# 6. Cleanup. +################################################################################ + +--source include/have_group_replication_plugin.inc +--let $rpl_skip_group_replication_start= 1 +--source include/group_replication.inc + +--connection server1 + +--echo +--echo # 1. Start Group Replication. + +--error 0,1 +--remove_file $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE + +--source include/start_and_bootstrap_group_replication.inc + +--echo +--echo # 2. Enable GCS_DEBUG_ALL tracing. +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_ALL"; +--let $wait_condition = SELECT LENGTH(LOAD_FILE(CONCAT(@@datadir, 'GCS_DEBUG_TRACE'))) > 0 +--source include/wait_condition.inc + +--echo +--echo # 3. Remove GCS_DEBUG_TRACE. +--remove_file $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE + +--echo +--echo # 4. Wait for stat auto-detection -- GCS_DEBUG_TRACE must reappear + +# Wait for GCS_DEBUG_TRACE re-creation and logs to be written. +--let $wait_condition = SELECT LENGTH(LOAD_FILE(CONCAT(@@datadir, 'GCS_DEBUG_TRACE'))) > 0 +--source include/wait_condition.inc + +--file_exists $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE + +--echo +--echo # 5. Group Replication is ONLINE. + +--let $assert_text= Member is ONLINE after stale-fd detection and reopen +--let $assert_cond= COUNT(*) = 1 FROM performance_schema.replication_group_members WHERE MEMBER_STATE = "ONLINE" +--source include/assert.inc + +--echo +--echo # 6. cleanup + +SET GLOBAL group_replication_communication_debug_options = "GCS_DEBUG_NONE"; +--source include/stop_group_replication.inc + +--remove_file $MYSQLTEST_VARDIR/mysqld.1/data/GCS_DEBUG_TRACE + +--source include/group_replication_end.inc diff --git a/mysql-test/suite/group_replication/t/gr_set_option_during_stop.test b/mysql-test/suite/group_replication/t/gr_set_option_during_stop.test index af177c86f619..c453f4d2b136 100644 --- a/mysql-test/suite/group_replication/t/gr_set_option_during_stop.test +++ b/mysql-test/suite/group_replication/t/gr_set_option_during_stop.test @@ -62,6 +62,7 @@ INSERT INTO gr_options_that_cannot_be_change (name) SELECT VARIABLE_NAME FROM performance_schema.global_variables WHERE VARIABLE_NAME LIKE 'group_replication_%' AND VARIABLE_NAME != 'group_replication_bootstrap_group' + AND VARIABLE_NAME != 'group_replication_communication_debug_max_file_size' AND VARIABLE_NAME != 'group_replication_communication_stack' AND VARIABLE_NAME != 'group_replication_consistency' AND VARIABLE_NAME != 'group_replication_exit_state_action' diff --git a/mysql-test/suite/group_replication/t/gr_show_global_and_session_variables.test b/mysql-test/suite/group_replication/t/gr_show_global_and_session_variables.test index 350924a6393a..d920eced791c 100644 --- a/mysql-test/suite/group_replication/t/gr_show_global_and_session_variables.test +++ b/mysql-test/suite/group_replication/t/gr_show_global_and_session_variables.test @@ -30,7 +30,7 @@ --source include/start_and_bootstrap_group_replication.inc --source include/stop_group_replication.inc ---let $gr_var_count= 65 +--let $gr_var_count= 66 --echo --echo # Test#1: Basic check that there are $gr_var_count GR variables. diff --git a/mysql-test/suite/group_replication/t/gr_variables_default_values.test b/mysql-test/suite/group_replication/t/gr_variables_default_values.test index 8aa60fec644a..dc700ef62af6 100644 --- a/mysql-test/suite/group_replication/t/gr_variables_default_values.test +++ b/mysql-test/suite/group_replication/t/gr_variables_default_values.test @@ -95,7 +95,7 @@ SET @@GLOBAL.group_replication_preemptive_garbage_collection = default; --let $saved_gr_xcom_ssl_accept_retries = `SELECT @@GLOBAL.group_replication_xcom_ssl_accept_retries;` # Total number of GR variables. ---let $total_gr_vars= 65 +--let $total_gr_vars= 66 --echo # --echo # Test Unit#1 @@ -296,6 +296,11 @@ SET @@SESSION.group_replication_consistency= default; --let $assert_cond= "[SELECT @@GLOBAL.group_replication_communication_debug_options]" = "GCS_DEBUG_NONE" --source include/assert.inc +# group_replication_communication_debug_max_file_size +--let $assert_text= Default group_replication_communication_debug_max_file_size is 0 +--let $assert_cond= "[SELECT @@GLOBAL.group_replication_communication_debug_max_file_size]" = 0 +--source include/assert.inc + # group_replication_unreachable_majority_timeout --let $assert_text= Default group_replication_unreachable_majority_timeout is 0 --let $assert_cond= "[SELECT @@GLOBAL.group_replication_unreachable_majority_timeout]" = 0 diff --git a/plugin/group_replication/include/plugin_variables.h b/plugin/group_replication/include/plugin_variables.h index c3e35238ef26..51c14e3f5af8 100644 --- a/plugin/group_replication/include/plugin_variables.h +++ b/plugin/group_replication/include/plugin_variables.h @@ -259,6 +259,11 @@ struct plugin_options_variables { char *communication_debug_options_var; +#define DEFAULT_COMMUNICATION_DEBUG_MAX_FILE_SIZE 0UL +#define MIN_COMMUNICATION_DEBUG_MAX_FILE_SIZE 0UL +#define MAX_COMMUNICATION_DEBUG_MAX_FILE_SIZE ~0UL + ulong communication_debug_max_file_size_var; + const char *exit_state_actions[4] = {"READ_ONLY", "ABORT_SERVER", "OFFLINE_MODE", (char *)nullptr}; TYPELIB exit_state_actions_typelib_t = {3, "exit_state_actions_typelib_t", diff --git a/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h b/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h index c08c1430ecf3..ed211c4699b9 100644 --- a/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h +++ b/plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h @@ -869,6 +869,85 @@ class Gcs_file_sink : public Sink_interface { */ Gcs_file_sink(Gcs_file_sink &d); Gcs_file_sink &operator=(const Gcs_file_sink &d); + + /* + Full path of the currently open trace file. Updated on every + initialize/rotate/reopen so stale-fd detection checks the right path. + */ + std::string m_current_path; + /* + Open a new timestamped file (e.g. gcs_debug_trace.log.20260706T165057) + and switch m_fd to it. Needed for size-based rotation. + The old file is not deleted similar to binlog. + + @note Do not call during fd error like disk full etc. Existing file + descriptors are not closed if reopen fails, so user can continue to + write to existing fd. If reopen is successful existing fd are closed. + @note Manual deletion of logs is needed. + + @retval GCS_OK new file is open; m_current_path updated. + @retval GCS_NOK could not open the new file; old fd remains valid. + */ + enum_gcs_error rotate(); + + /* + Same as rotate(): open a new timestamped file and switch m_fd to it. + Called when the current file was moved or deleted externally. + + @refer rotate(), notes have been added in rotate, same applicable here + + @retval GCS_OK new file is open; m_current_path updated. + @retval GCS_NOK could not open/create the file; old fd remains valid. + */ + enum_gcs_error reopen(); + + /* + Maximum size, in bytes, of the file before it is automatically rotated. + 0 disables size-based rotation. + Static to avoid major changes to class like constructor param, + initialization etc. + */ + static size_t m_max_file_size; + + /* + Bytes written to the current file since it was last opened or rotated. + */ + size_t m_current_file_size{0}; + + /* + Timestamp, in microseconds, of the last stale-file check. + */ + ulonglong m_last_stat_check_time{0}; + + /* + Minimum time, in microseconds, between two stale-file checks. The check + costs two syscalls (my_stat on the path + my_fstat on the fd), so running + it on every trace line is wasteful. + */ + static constexpr ulonglong STAT_CHECK_INTERVAL_US = 100ULL * 1000ULL; + + /* Returns true if enough time elapsed to run the stale-file check again. */ + bool should_check_file(); + + /* + Timestamp, in microseconds, when the most recent rotation error was written + to the error log. A value of zero means that no error has been logged yet. + */ + ulonglong m_last_logged_error_time{0}; + + /* + Minimum time, in microseconds, between rotation error messages. + */ + static constexpr ulonglong ERROR_LOG_INTERVAL_US = 1ULL * 1000ULL * 1000ULL; + + /* Returns true if time has elapsed to log another error. */ + bool should_log_error(); + + public: + /* @refer m_max_file_size above */ + static void set_max_debug_file_size(size_t file_size) { + m_max_file_size = file_size; + } }; #endif /* XCOM_STANDALONE */ diff --git a/plugin/group_replication/libmysqlgcs/src/bindings/xcom/gcs_xcom_interface.cc b/plugin/group_replication/libmysqlgcs/src/bindings/xcom/gcs_xcom_interface.cc index 4ab21f922bbd..91b5a5b12390 100644 --- a/plugin/group_replication/libmysqlgcs/src/bindings/xcom/gcs_xcom_interface.cc +++ b/plugin/group_replication/libmysqlgcs/src/bindings/xcom/gcs_xcom_interface.cc @@ -351,6 +351,32 @@ enum_gcs_error Gcs_xcom_interface::initialize( m_wait_for_ssl_init_cond.init( key_GCS_COND_Gcs_xcom_interface_m_wait_for_ssl_init_cond); +#ifndef XCOM_STANDALONE + { + /* + Either we can modify the constructors of Gcs_file_sink and pass the + parameters initialize_logging or we need to init set_max_debug_file_size + first before the log file is created. + Otherwise first log file will always be GCS_DEBUG_TRACE (without + timestamp) since set_max_debug_file_size has not been called yet, which + is not a big issue, but we can initialize + communication_debug_max_file_size first. + */ + const std::string *debug_max_file_size = + interface_params.get_parameter("communication_debug_max_file_size"); + if (debug_max_file_size != nullptr && !debug_max_file_size->empty()) { + try { + // This should not fail since debug_max_file_size is originally long + ulong max_file_size = std::stoul(*debug_max_file_size); + Gcs_file_sink::set_max_debug_file_size(max_file_size); + } catch (const std::exception &) { + MYSQL_GCS_LOG_ERROR( + "Failed to initialize GCS_DEBUG_TRACE log rotation. Could not read " + "parameter communication_debug_max_file_size."); + } + } + } +#endif /* Initialize logging sub-systems. */ diff --git a/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc b/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc index 389f052d720d..51e4ec7db44c 100644 --- a/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc +++ b/plugin/group_replication/libmysqlgcs/src/interface/gcs_logging_system.cc @@ -40,10 +40,13 @@ #include "my_dir.h" #include "my_io.h" #include "my_sys.h" +#include "my_systime.h" #endif /* XCOM_STANDALONE */ #include "plugin/group_replication/libmysqlgcs/include/mysql/gcs/gcs_logging_system.h" +size_t Gcs_file_sink::m_max_file_size = 0; + void *consumer_function(void *ptr); Gcs_async_buffer::Gcs_async_buffer(Sink_interface *sink, int buffer_size) @@ -348,6 +351,65 @@ enum_gcs_error Gcs_default_debugger::finalize() { return m_sink->finalize(); } Gcs_default_debugger *Gcs_debug_manager::m_debugger = nullptr; #ifndef XCOM_STANDALONE + +/* + Generate a conflict-safe rotation name for the active trace file. + Form: .YYYYMMDDTHHMMSS (UTC). + If the timestamped file already exists: + .YYYYMMDDTHHMMSS.1, .YYYYMMDDTHHMMSS.2, ... + up to .1000 + + If timestamp generation fails: + .1, .2, ... + up to .1000 + + @note Entire usage of gcs_generate_rotated_name is inside XCOM_STANDALONE. + + @returns + 0: on success and sets out_path; + 1: if all 1000 names are taken. +*/ +static int gcs_generate_rotated_name(const std::string &base_name, + std::string &out_path) { + char ts_buf[32]; + std::string rotation_base = base_name; + MY_STAT st_buf; + + const unsigned long long us = my_micro_time(); + const time_t sec = static_cast(us / 1000000); + + struct tm tm_info; + if (gmtime_r(&sec, &tm_info) != nullptr) { + const int len = + snprintf(ts_buf, sizeof(ts_buf), "%04d%02d%02dT%02d%02d%02d", + tm_info.tm_year + 1900, tm_info.tm_mon + 1, tm_info.tm_mday, + tm_info.tm_hour, tm_info.tm_min, tm_info.tm_sec); + + if (len >= 0 && static_cast(len) < sizeof(ts_buf)) { + rotation_base += "."; + rotation_base += ts_buf; + } + } + if (my_stat(rotation_base.c_str(), &st_buf, MYF(0)) == nullptr) { + out_path = rotation_base; + return 0; + } + + for (int n = 1; n <= 1000; ++n) { + std::string candidate = rotation_base + "." + std::to_string(n); + + if (my_stat(candidate.c_str(), &st_buf, MYF(0)) == nullptr) { + out_path = candidate; + return 0; + } + } + + MYSQL_GCS_LOG_ERROR("GCS debug trace: all rotation names for '" + << rotation_base.c_str() + << "' are taken. Cannot rotate."); + return 1; +} + Gcs_file_sink::Gcs_file_sink(const std::string &file_name, const std::string &dir_name) : m_fd(0), @@ -407,6 +469,18 @@ enum_gcs_error Gcs_file_sink::initialize() { return GCS_NOK; } } + /* + If size-based rotation is enabled, open a new timestamped file so each + GR run is cleanly separated. + */ + if (m_max_file_size > 0) { + std::string new_name; + if (gcs_generate_rotated_name(file_name_buffer, new_name) == 0) { + strncpy(file_name_buffer, new_name.c_str(), FN_REFLEN - 1); + file_name_buffer[FN_REFLEN - 1] = '\0'; + } + /* If name generation fails fall through and open canonical (appends). */ + } if ((m_fd = my_create(file_name_buffer, 0640, O_CREAT | O_WRONLY | O_APPEND, MYF(0))) < 0) { @@ -421,6 +495,8 @@ enum_gcs_error Gcs_file_sink::initialize() { return GCS_NOK; } + m_current_path = file_name_buffer; + m_current_file_size = 0; m_initialized = true; return GCS_OK; @@ -437,6 +513,131 @@ enum_gcs_error Gcs_file_sink::finalize() { return GCS_OK; } +bool Gcs_file_sink::should_log_error() { + const ulonglong now = my_micro_time(); + if (now - m_last_logged_error_time < ERROR_LOG_INTERVAL_US) { + return false; + } + m_last_logged_error_time = now; + return true; +} + +bool Gcs_file_sink::should_check_file() { + const ulonglong now = my_micro_time(); + if (now - m_last_stat_check_time < STAT_CHECK_INTERVAL_US) { + return false; + } + m_last_stat_check_time = now; + return true; +} + +/* + Since reopen() is called only when file is renamed/deleted, there is no need + to suppress logging to avoid flooding since stats itself is controlled by + should_check_file(). +*/ +enum_gcs_error Gcs_file_sink::reopen() { + if (!m_initialized) return GCS_OK; + char file_name_buffer[FN_REFLEN]; + + if (get_file_name(file_name_buffer)) { + MYSQL_GCS_LOG_ERROR("GCS debug trace reopen: error validating file name '" + << m_file_name << "'."); + return GCS_NOK; + } + + std::string new_name = file_name_buffer; + if (m_max_file_size > 0) { + if (gcs_generate_rotated_name(file_name_buffer, new_name) != 0) { + MYSQL_GCS_LOG_ERROR( + "GCS debug trace reopen: could not generate new file " + "name for '" + << file_name_buffer << "'."); + return GCS_NOK; + } + } + + File new_fd = + my_create(new_name.c_str(), 0640, O_CREAT | O_WRONLY | O_APPEND, MYF(0)); + /* + We have not closed existing file descriptors, so even if my_create fails + call to existing fd are still safe. write continues to write to detached + file descriptor without any error. + We keep the old fd active in case of error. + */ + if (new_fd < 0) { + int errno_gcs = 0; +#if defined(_WIN32) + errno_gcs = WSAGetLastError(); +#else + errno_gcs = errno; +#endif + MYSQL_GCS_LOG_ERROR("GCS debug trace reopen: error opening file '" + << new_name.c_str() << "': " << strerror(errno_gcs) + << "."); + return GCS_NOK; + } + + /* ignore any error, we have new FD */ + my_sync(m_fd, MYF(0)); + my_close(m_fd, MYF(0)); + m_fd = new_fd; + m_current_path = new_name; + m_initialized = true; + m_current_file_size = 0; + + return GCS_OK; +} + +enum_gcs_error Gcs_file_sink::rotate() { + if (!m_initialized) return GCS_OK; + char file_name_buffer[FN_REFLEN]; + + if (get_file_name(file_name_buffer)) { + if (should_log_error()) { + MYSQL_GCS_LOG_ERROR("GCS debug trace rotate: error validating file name '" + << m_file_name << "'."); + } + return GCS_NOK; + } + + std::string new_name = file_name_buffer; + if (gcs_generate_rotated_name(file_name_buffer, new_name) != 0) { + if (should_log_error()) { + MYSQL_GCS_LOG_ERROR( + "GCS debug trace rotate: could not generate new file " + "name for '" + << file_name_buffer << "'."); + } + return GCS_NOK; + } + + File new_fd = + my_create(new_name.c_str(), 0640, O_CREAT | O_WRONLY | O_APPEND, MYF(0)); + if (new_fd < 0) { + int errno_gcs = 0; +#if defined(_WIN32) + errno_gcs = WSAGetLastError(); +#else + errno_gcs = errno; +#endif + if (should_log_error()) { + MYSQL_GCS_LOG_ERROR("GCS debug trace rotate: error opening new file '" + << new_name << "': " << strerror(errno_gcs) << "."); + } + return GCS_NOK; + } + + my_sync(m_fd, MYF(0)); + my_close(m_fd, MYF(0)); + m_fd = new_fd; + m_current_path = new_name; + m_initialized = true; + m_current_file_size = 0; + + return GCS_OK; +} + void Gcs_file_sink::log_event(const std::string &message) { log_event(message.c_str(), message.length()); } @@ -446,6 +647,30 @@ void Gcs_file_sink::log_event(const char *message, size_t message_size) { written = my_write(m_fd, (const uchar *)message, message_size, MYF(0)); + if (written != MY_FILE_ERROR) { + m_current_file_size += written; + if (m_max_file_size > 0 && m_current_file_size >= m_max_file_size) { + rotate(); + } else if (should_check_file()) { + /* + Check if the current trace file was moved or deleted. + It was noted when the trace file was moved/deleted, no debug trace + was later generated in the session. + + Note: the check is done after the write (and only once per + should_check_file() interval), so if the file is removed between our + write and this check, the line just written goes to the orphaned inode + and some logs are lost until reopen() creates a new file. + */ + MY_STAT path_st, fd_st; + bool path_gone = + (my_stat(m_current_path.c_str(), &path_st, MYF(0)) == nullptr); + bool ino_changed = !path_gone && my_fstat(m_fd, &fd_st) == 0 && + path_st.st_ino != fd_st.st_ino; + if (path_gone || ino_changed) reopen(); + } + } + if (written == MY_FILE_ERROR) { int errno_gcs = 0; #if defined(_WIN32) @@ -460,13 +685,16 @@ void Gcs_file_sink::log_event(const char *message, size_t message_size) { const std::string Gcs_file_sink::get_information() const { std::string invalid("invalid"); - char file_name_buffer[FN_REFLEN]; if (!m_initialized) return invalid; - - if (get_file_name(file_name_buffer)) return invalid; - - return std::string(file_name_buffer); + /* + Although m_current_path can now be mutated from a different thread by + rotate()/reopen() during logging, it is safe to read it here without a + mutex: get_information() is only called from the initialization path (see + Gcs_xcom_interface::initialize_logging), before any concurrent logging + thread can rotate/reopen the file. + */ + return std::string(m_current_path); } void Gcs_clock_timestamp_provider::get_timestamp_as_c_string(char *buffer, diff --git a/plugin/group_replication/src/plugin.cc b/plugin/group_replication/src/plugin.cc index 0b0b48658b6a..c1673e1a7fb4 100644 --- a/plugin/group_replication/src/plugin.cc +++ b/plugin/group_replication/src/plugin.cc @@ -2840,6 +2840,17 @@ int build_gcs_parameters(Gcs_interface_parameters &gcs_module_parameters) { gcs_module_parameters.add_parameter("communication_debug_path", mysql_real_data_home); + /* + Maximum size, in bytes, of the GCS debug trace file before it is + automatically rotated. 0 disables size-based rotation. Read once here, + at Group Replication start -- like communication_debug_file/_path above. + + @note changing this sysvar takes effect at the next Group Replication start + */ + gcs_module_parameters.add_parameter( + "communication_debug_max_file_size", + std::to_string(ov.communication_debug_max_file_size_var)); + sv.deinit(); return result; } @@ -5018,6 +5029,21 @@ static MYSQL_SYSVAR_STR( "GCS_DEBUG_NONE" /* default */ ); +static MYSQL_SYSVAR_ULONG( + communication_debug_max_file_size, /* name */ + ov.communication_debug_max_file_size_var, /* var */ + PLUGIN_VAR_OPCMDARG | PLUGIN_VAR_PERSIST_AS_READ_ONLY, /* optional var */ + "Maximum size, in bytes, of the GCS debug trace file (GCS_DEBUG_TRACE) " + "before it is automatically rotated. 0 disables size-based rotation. " + "Takes effect at the next Group Replication start.", + nullptr, /* check func. */ + nullptr, /* update func. */ + DEFAULT_COMMUNICATION_DEBUG_MAX_FILE_SIZE, /* default */ + MIN_COMMUNICATION_DEBUG_MAX_FILE_SIZE, /* min */ + MAX_COMMUNICATION_DEBUG_MAX_FILE_SIZE, /* max */ + 0 /* block */ +); + static MYSQL_SYSVAR_ENUM(exit_state_action, /* name */ ov.exit_state_action_var, /* var */ PLUGIN_VAR_OPCMDARG | @@ -5516,6 +5542,7 @@ static SYS_VAR *group_replication_system_vars[] = { MYSQL_SYSVAR(flow_control_applier_threshold), MYSQL_SYSVAR(transaction_size_limit), MYSQL_SYSVAR(communication_debug_options), + MYSQL_SYSVAR(communication_debug_max_file_size), MYSQL_SYSVAR(exit_state_action), MYSQL_SYSVAR(autorejoin_tries), MYSQL_SYSVAR(unreachable_majority_timeout), From 04cb664b03ca1453682c794b70a7d62da2cd27aa Mon Sep 17 00:00:00 2001 From: Catalin Besleaga Date: Thu, 18 Jun 2026 12:53:56 +0300 Subject: [PATCH 10/32] PS-11167 [9.7]: Add DISTANCE() for VECTOR data type Implement SQL DISTANCE(vector, vector, metric) and the VECTOR_DISTANCE() synonym for vector similarity queries. Supported metrics: EUCLIDEAN (L2), EUCLIDEAN_SQUARED, MANHATTAN (L1), COSINE, and DOT. Core library (vector-common/vector_distance.*): - Runtime SIMD dispatch across Scalar, SSE4.2/NEON, AVX2, AVX-512F, and SVE2 tiers; per-kernel target attributes, no global -march=native - Dim-aware wide/narrow dispatch (dims >= 16 use widest tier; smaller vectors use 128-bit tier to avoid AVX setup overhead) - Unaligned load intrinsics throughout: VECTOR data may be misaligned; on modern CPUs unaligned and aligned loads have identical throughput when data is aligned; aligned loads would fault on misaligned inputs without a performance benefit. Loads that span a 64-byte cache-line boundary may still be slower (two line fetches); that depends on runtime address, not on loadu vs load, and is not avoided by switching to aligned intrinsics - Float32 SIMD accumulation with double-precision horizontal sum and scalar tail: preserves correctness for large dimensions and extreme values (e.g. 2e38 Euclidean distance) without sacrificing SIMD width --- mysql-test/suite/percona/include/distance.inc | 253 ++++ .../suite/percona/r/distance_cosine.result | 363 ++++++ .../suite/percona/r/distance_dot.result | 358 ++++++ .../suite/percona/r/distance_euclidean.result | 357 ++++++ .../r/distance_euclidean_squared.result | 356 ++++++ .../suite/percona/r/distance_manhattan.result | 357 ++++++ .../suite/percona/t/distance_cosine.test | 2 + mysql-test/suite/percona/t/distance_dot.test | 2 + .../suite/percona/t/distance_euclidean.test | 2 + .../percona/t/distance_euclidean_squared.test | 2 + .../suite/percona/t/distance_manhattan.test | 2 + .../suite/perfschema/r/error_log.result | 2 +- mysql-test/suite/perfschema/t/error_log.test | 3 +- share/messages_to_error_log.txt | 3 + sql/CMakeLists.txt | 1 + sql/item_create.cc | 2 + sql/item_func.h | 6 + sql/item_strfunc.cc | 146 ++- sql/item_strfunc.h | 20 + sql/mysqld.cc | 10 + unittest/gunit/CMakeLists.txt | 12 + unittest/gunit/vector_distance-t.cc | 634 ++++++++++ unittest/gunit/vector_distance_benchmark-t.cc | 306 +++++ vector-common/vector_distance.cc | 1106 +++++++++++++++++ vector-common/vector_distance.h | 115 ++ 25 files changed, 4417 insertions(+), 3 deletions(-) create mode 100644 mysql-test/suite/percona/include/distance.inc create mode 100644 mysql-test/suite/percona/r/distance_cosine.result create mode 100644 mysql-test/suite/percona/r/distance_dot.result create mode 100644 mysql-test/suite/percona/r/distance_euclidean.result create mode 100644 mysql-test/suite/percona/r/distance_euclidean_squared.result create mode 100644 mysql-test/suite/percona/r/distance_manhattan.result create mode 100644 mysql-test/suite/percona/t/distance_cosine.test create mode 100644 mysql-test/suite/percona/t/distance_dot.test create mode 100644 mysql-test/suite/percona/t/distance_euclidean.test create mode 100644 mysql-test/suite/percona/t/distance_euclidean_squared.test create mode 100644 mysql-test/suite/percona/t/distance_manhattan.test create mode 100644 unittest/gunit/vector_distance-t.cc create mode 100644 unittest/gunit/vector_distance_benchmark-t.cc create mode 100644 vector-common/vector_distance.cc create mode 100644 vector-common/vector_distance.h diff --git a/mysql-test/suite/percona/include/distance.inc b/mysql-test/suite/percona/include/distance.inc new file mode 100644 index 000000000000..b53a544e2b7f --- /dev/null +++ b/mysql-test/suite/percona/include/distance.inc @@ -0,0 +1,253 @@ +--echo # +--echo # Test coverage for vector DISTANCE() function. +--echo # + +--echo # +--echo # 0) Prepare playground. +--echo # +CREATE TABLE t1 (id INT PRIMARY KEY, v1 VECTOR(1), v2 VECTOR(2)); +INSERT INTO t1 VALUES (0, TO_VECTOR('[0]'), TO_VECTOR('[0, 0]')), + (1, TO_VECTOR('[1]'), TO_VECTOR('[1, 0]')), + (2, TO_VECTOR('[1]'), TO_VECTOR('[0, 1]')), + (3, TO_VECTOR('[2]'), TO_VECTOR('[1, 1]')), + (4, TO_VECTOR('[2]'), TO_VECTOR('[2, 0]')), + (98, TO_VECTOR('[1]'), TO_VECTOR('[2]')), + (99, NULL, NULL); +CREATE TABLE t_metric_name (id INT PRIMARY KEY, name VARCHAR(10)); +INSERT INTO t_metric_name VALUES (1, "EUCLIDEAN"), (99, NULL); + +--echo # +--echo # 1) Test how different number and types of arguments are handled. +--echo # +--echo # 1.1) Arity. +--echo # +--error ER_WRONG_PARAMCOUNT_TO_NATIVE_FCT +SELECT DISTANCE(); +--error ER_WRONG_PARAMCOUNT_TO_NATIVE_FCT +SELECT DISTANCE(TO_VECTOR("[1]")); +--error ER_WRONG_PARAMCOUNT_TO_NATIVE_FCT +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]")); +eval SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), "$metric"); +--error ER_WRONG_PARAMCOUNT_TO_NATIVE_FCT +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), TO_VECTOR("[3]"), "EUCLIDEAN"); + +--echo # +--echo # 1.2) Argument types. +--echo # +--echo # Only vectors or binary strings are allowed for first the two arguments. +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE("[1]", TO_VECTOR("[2]"), "$metric"); +eval SELECT DISTANCE(X'0000803F', TO_VECTOR("[2]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[2]"), "$metric"); +eval SELECT DISTANCE(v1, TO_VECTOR("[2]"), "$metric") FROM t1 WHERE id = 0; +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(id, TO_VECTOR("[2]"), "$metric") FROM t1 WHERE id = 0; +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(TO_VECTOR("[1]"), "[2]", "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[2]"), X'0000803F', "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[3]"), v1, "$metric") FROM t1 WHERE id = 1; +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(TO_VECTOR("[0]"), id, "$metric") FROM t1 WHERE id = 1; + +--echo # The third argument must be a string literal with value from the +--echo # fixed list of metric names. +--error ER_WRONG_ARGUMENTS +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[-1, 0]"), 1); +--error ER_WRONG_ARGUMENTS +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 0]"), CONCAT("EUCLI","DEAN")); +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2, 0]"), "euclidean"); +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[3, 0]"), "EuClIdEaN"); +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[4, 0]"), X'4555434C494445414E'); +--echo # Metric strings with embedded NUL must be rejected regardless of prefix match. +--error ER_WRONG_ARGUMENTS +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), X'4555434C494445414E00'); +--error ER_WRONG_ARGUMENTS +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), X'4555434C494445414E004A554E4B'); +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 0]"), "NOSUCHMETRIC"); +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[6, 0]"), name) FROM t_metric_name WHERE id = 1; +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[7, 0]"), NULL); + +--echo # +--echo # 1.3) NULL arguments and nullability in metadata for result. +--echo # +eval SELECT DISTANCE(NULL, TO_VECTOR("[1, 0]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[1, 1]"), NULL, "$metric"); +eval SELECT DISTANCE(v2, TO_VECTOR("[1, 0]"), "$metric") FROM t1 WHERE id = 99; +eval SELECT DISTANCE(TO_VECTOR("[1, 1]"), v2, "$metric") FROM t1 WHERE id = 99; +--echo # The third argument doesn't allow NULL values in any form. +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), NULL); +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), name) FROM t_metric_name WHERE id = 99; +--echo # The result metadata should indicate that it is nullable. +eval CREATE TABLE tt SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "$metric") AS d; +SHOW CREATE TABLE tt; +DROP TABLE tt; + +--echo # +--echo # 2) Test vector arguments length mismatch. +--echo # +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[1, 0]"), "$metric"); +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(v2, TO_VECTOR("[1]"), "$metric") FROM t1 WHERE id = 1; +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(v1, v2, "$metric") FROM t1 WHERE id = 1; +--echo # +--echo # Note that length check happens at runtime. This is well visible +--echo # when we have value stored in a vector field which is shorter than +--echo # maximum length specified at the field creation time. +eval SELECT DISTANCE(v1, v2, "$metric") FROM t1 WHERE id = 98; +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), v2, "$metric") FROM t1 WHERE id = 98; +--echo # Binary-string BLOB arguments exceeding max_dimensions (16383) are rejected. +--echo # A BLOB column is used so the argument passes the resolve-time binary-charset +--echo # type check; the max_dimensions guard fires at runtime in val_real(). +CREATE TABLE t_oversized (v MEDIUMBLOB); +INSERT INTO t_oversized VALUES (REPEAT(X'00000000', 16384)); +--error ER_WRONG_ARGUMENTS +eval SELECT DISTANCE(v, v, "$metric") FROM t_oversized; +DROP TABLE t_oversized; + +--echo # +--echo # 3) Some basic tests for different (from syntax PoV) variants of +--echo # arguments. +--echo # +eval SELECT DISTANCE(X'0000803F0000803F', X'0000000000000040', "$metric"); +eval SELECT DISTANCE(X'0000803F0000803F', TO_VECTOR("[2, 0]"), "$metric"); +eval SELECT DISTANCE(X'0000803F0000803F', v2, "$metric") FROM t1 WHERE id = 4; +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), X'000000000000803F', "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), v2, "$metric") FROM t1 WHERE id = 1; +eval SELECT DISTANCE(a.v2, b.v2, "$metric") FROM t1 AS a, t1 AS b WHERE a.id = 0 AND b.id = 4; +eval SELECT DISTANCE(v2, X'0000000000000040', "$metric") FROM t1 WHERE id = 0; +eval SELECT DISTANCE(v2, TO_VECTOR("[0, 2]"), "$metric") FROM t1 WHERE id = 0; +--echo # Non-trivial (artificial) combinations +eval SELECT DISTANCE(TO_VECTOR(CONCAT("[0", ", ", "1]")), CONCAT(X'00000000', X'00000040'), "$metric"); +--echo # The below case demonstrates that arguments to DISTANCE might not be +--echo # well-aligned in memory. +eval SELECT DISTANCE(SUBSTR(X'010000000000000040', 2), RIGHT(X'40000000000000803F', 8), "$metric"); +--echo # 9-byte blobs; SUBSTR from pos 2 → 8 bytes at offset 1 (misaligned for float). +--echo # Length must stay a multiple of 4; SUBSTR(..., 4) on 9 bytes yields 6 → ER_TO_VECTOR_CONVERSION. +eval SELECT DISTANCE(SUBSTR(X'000100000000000040', 2), SUBSTR(X'00040000000000803F', 2), "$metric"); + +--echo # +--echo # 4) Basic test for different vector values. +--echo # +--echo # Identical / collinear vectors. +eval SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 1]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[1, 0]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2.5, 2.5]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[1, 2, 3, 4, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "$metric"); +--echo # Orthogonal vectors. +eval SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[0, 1, 0]"), TO_VECTOR("[-1, 0, -1]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[1, 0, 3, 0, 5]"), TO_VECTOR("[0, 2, 0, 4, 0]"), "$metric"); +--echo # Anti-parallel vectors. +eval SELECT DISTANCE(TO_VECTOR("[-1, -1]"), TO_VECTOR("[2, 2]"), "$metric"); +if ($metric != DOT) +{ + if ($metric != MANHATTAN) + { + if ($metric != COSINE) + { + --error ER_DATA_OUT_OF_RANGE + } + } +} +eval SELECT DISTANCE(TO_VECTOR("[-2e38, 1]"), TO_VECTOR("[2e38, -1]"), "$metric"); +--echo # Distance from origin. +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[3, 4]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 12]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[0, 0, 0, 0]"), TO_VECTOR("[1, 1, 1, 1]"), "$metric"); +--echo # Mixed-sign and larger vectors. +eval SELECT DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[1, 7, 3, 16, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "$metric"); +--echo # Zero vector (behavior differs per metric). +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2, 2]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[0, 0]"), "$metric"); +eval SELECT DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[0]"), "$metric"); +--echo # Large values near float32 max. +eval SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2e38, 0]"), "$metric"); +--echo # Same value in a 16-dim vector: exercises the wide-tier SIMD overflow +--echo # fallback (dims >= 16 dispatches to the wide kernel; squaring 2e38 in +--echo # float32 overflows to +Inf, but the isfinite check falls back to scalar). +eval SELECT DISTANCE(TO_VECTOR("[2e38, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), + TO_VECTOR("[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), + "$metric"); +--echo # Symmetry: DISTANCE(a, b) = DISTANCE(b, a). +eval SELECT DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "$metric") = + DISTANCE(TO_VECTOR("[4, 5, 6]"), TO_VECTOR("[1, 2, 3]"), "$metric"); +--echo # Special IEEE 754 float32 values: NaN, +Infinity, -Infinity. +--echo # NaN/Inf input elements raise ER_DATA_OUT_OF_RANGE for all metrics (POW/EXP convention). +--error ER_DATA_OUT_OF_RANGE +eval SELECT DISTANCE(X'0000C07F', X'00000000', "$metric"); +--error ER_DATA_OUT_OF_RANGE +eval SELECT DISTANCE(X'0000807F', X'00000000', "$metric"); +--error ER_DATA_OUT_OF_RANGE +eval SELECT DISTANCE(X'000080FF', X'00000000', "$metric"); +--echo # Wide-tier SIMD path coverage (dims >= 16 dispatches to the wide kernel). +--echo # Integer-valued diffs keep float32 partial sums exact, so results are +--echo # identical across Scalar / SSE4.2 / NEON / AVX2 / AVX-512 / SVE2. +--echo # 16-dim: fills one AVX-512 register / two AVX2 / four SSE4.2 -- no scalar tail. +eval SELECT DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), + TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), + "$metric"); +--echo # 20-dim: SSE4.2 5x4 (no tail); AVX2 2x8 + 4-elem scalar tail; +--echo # AVX-512 1x16 + 4-elem scalar tail. +eval SELECT DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), + TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), + "$metric"); + +--echo # +--echo # prepared statements +--echo # +eval PREPARE s1 FROM 'SELECT DISTANCE(v1, TO_VECTOR("[2]"), "$metric") FROM t1 WHERE id = 4'; +EXECUTE s1; +EXECUTE s1; +DEALLOCATE PREPARE s1; + +--error ER_WRONG_ARGUMENTS +PREPARE stmt FROM 'SELECT DISTANCE(v1, TO_VECTOR("[2]"), "?") FROM t1 WHERE id = 4'; + +eval PREPARE stmt FROM 'SELECT DISTANCE(v1, TO_VECTOR(?), "$metric") FROM t1 WHERE id = 4'; +SET @met="[2]"; +EXECUTE stmt USING @met; +DEALLOCATE PREPARE stmt; + +--disable_warnings +--echo # +--echo # 5) Distance in query contexts. +--echo # +--echo # ORDER BY distance: nearest-neighbour pattern. +eval SELECT id FROM t1 WHERE id IN (0,1,2,3,4) + ORDER BY DISTANCE(v2, TO_VECTOR('[1, 0]'), "$metric"), id; +--echo # ORDER BY distance DESC: farthest-neighbour pattern. +eval SELECT id FROM t1 WHERE id IN (0,1,2,3,4) + ORDER BY DISTANCE(v2, TO_VECTOR('[1, 0]'), "$metric") DESC, id; +--echo # WHERE: range query filtering by distance. +eval SELECT id FROM t1 + WHERE id IN (0,1,2,3,4) AND DISTANCE(v2, TO_VECTOR('[1, 0]'), "$metric") < 1.5 + ORDER BY id; +--echo # Derived table with distance. +eval SELECT id FROM + (SELECT id, DISTANCE(v2, TO_VECTOR('[1, 0]'), "$metric") AS d + FROM t1 WHERE id IN (0,1,2,3,4)) AS sq + WHERE d IS NOT NULL ORDER BY d, id; +--enable_warnings + +if ($metric == COSINE) +{ +--echo # Zero-vector cosine in DML: must insert NULL without aborting under strict sql_mode. +CREATE TABLE tt_cosine_dml (d DOUBLE); +INSERT INTO tt_cosine_dml SELECT DISTANCE(TO_VECTOR('[0]'), TO_VECTOR('[0]'), 'COSINE'); +SELECT d FROM tt_cosine_dml; +DROP TABLE tt_cosine_dml; +} + +DROP TABLE t_metric_name; +DROP TABLE t1; diff --git a/mysql-test/suite/percona/r/distance_cosine.result b/mysql-test/suite/percona/r/distance_cosine.result new file mode 100644 index 000000000000..b4b1cb14c5a3 --- /dev/null +++ b/mysql-test/suite/percona/r/distance_cosine.result @@ -0,0 +1,363 @@ +# +# Test coverage for vector DISTANCE() function. +# +# +# 0) Prepare playground. +# +CREATE TABLE t1 (id INT PRIMARY KEY, v1 VECTOR(1), v2 VECTOR(2)); +INSERT INTO t1 VALUES (0, TO_VECTOR('[0]'), TO_VECTOR('[0, 0]')), +(1, TO_VECTOR('[1]'), TO_VECTOR('[1, 0]')), +(2, TO_VECTOR('[1]'), TO_VECTOR('[0, 1]')), +(3, TO_VECTOR('[2]'), TO_VECTOR('[1, 1]')), +(4, TO_VECTOR('[2]'), TO_VECTOR('[2, 0]')), +(98, TO_VECTOR('[1]'), TO_VECTOR('[2]')), +(99, NULL, NULL); +CREATE TABLE t_metric_name (id INT PRIMARY KEY, name VARCHAR(10)); +INSERT INTO t_metric_name VALUES (1, "EUCLIDEAN"), (99, NULL); +# +# 1) Test how different number and types of arguments are handled. +# +# 1.1) Arity. +# +SELECT DISTANCE(); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]")); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]")); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), "COSINE"); +DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), "COSINE") +0 +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), TO_VECTOR("[3]"), "EUCLIDEAN"); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +# +# 1.2) Argument types. +# +# Only vectors or binary strings are allowed for first the two arguments. +SELECT DISTANCE("[1]", TO_VECTOR("[2]"), "COSINE"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(X'0000803F', TO_VECTOR("[2]"), "COSINE"); +DISTANCE(X'0000803F', TO_VECTOR("[2]"), "COSINE") +0 +SELECT DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[2]"), "COSINE"); +DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[2]"), "COSINE") +NULL +SELECT DISTANCE(v1, TO_VECTOR("[2]"), "COSINE") FROM t1 WHERE id = 0; +DISTANCE(v1, TO_VECTOR("[2]"), "COSINE") +NULL +SELECT DISTANCE(id, TO_VECTOR("[2]"), "COSINE") FROM t1 WHERE id = 0; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1]"), "[2]", "COSINE"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[2]"), X'0000803F', "COSINE"); +DISTANCE(TO_VECTOR("[2]"), X'0000803F', "COSINE") +0 +SELECT DISTANCE(TO_VECTOR("[3]"), v1, "COSINE") FROM t1 WHERE id = 1; +DISTANCE(TO_VECTOR("[3]"), v1, "COSINE") +0 +SELECT DISTANCE(TO_VECTOR("[0]"), id, "COSINE") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +# The third argument must be a string literal with value from the +# fixed list of metric names. +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[-1, 0]"), 1); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 0]"), CONCAT("EUCLI","DEAN")); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2, 0]"), "euclidean"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2, 0]"), "euclidean") +1.4142135623730951 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[3, 0]"), "EuClIdEaN"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[3, 0]"), "EuClIdEaN") +2.23606797749979 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[4, 0]"), X'4555434C494445414E'); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[4, 0]"), X'4555434C494445414E') +3.1622776601683795 +# Metric strings with embedded NUL must be rejected regardless of prefix match. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), X'4555434C494445414E00'); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), X'4555434C494445414E004A554E4B'); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 0]"), "NOSUCHMETRIC"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[6, 0]"), name) FROM t_metric_name WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[7, 0]"), NULL); +ERROR HY000: Incorrect arguments to distance +# +# 1.3) NULL arguments and nullability in metadata for result. +# +SELECT DISTANCE(NULL, TO_VECTOR("[1, 0]"), "COSINE"); +DISTANCE(NULL, TO_VECTOR("[1, 0]"), "COSINE") +NULL +SELECT DISTANCE(TO_VECTOR("[1, 1]"), NULL, "COSINE"); +DISTANCE(TO_VECTOR("[1, 1]"), NULL, "COSINE") +NULL +SELECT DISTANCE(v2, TO_VECTOR("[1, 0]"), "COSINE") FROM t1 WHERE id = 99; +DISTANCE(v2, TO_VECTOR("[1, 0]"), "COSINE") +NULL +SELECT DISTANCE(TO_VECTOR("[1, 1]"), v2, "COSINE") FROM t1 WHERE id = 99; +DISTANCE(TO_VECTOR("[1, 1]"), v2, "COSINE") +NULL +# The third argument doesn't allow NULL values in any form. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), NULL); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), name) FROM t_metric_name WHERE id = 99; +ERROR HY000: Incorrect arguments to distance +# The result metadata should indicate that it is nullable. +CREATE TABLE tt SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "COSINE") AS d; +SHOW CREATE TABLE tt; +Table Create Table +tt CREATE TABLE `tt` ( + `d` double DEFAULT NULL +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_0900_ai_ci +DROP TABLE tt; +# +# 2) Test vector arguments length mismatch. +# +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[1, 0]"), "COSINE"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(v2, TO_VECTOR("[1]"), "COSINE") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(v1, v2, "COSINE") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +# +# Note that length check happens at runtime. This is well visible +# when we have value stored in a vector field which is shorter than +# maximum length specified at the field creation time. +SELECT DISTANCE(v1, v2, "COSINE") FROM t1 WHERE id = 98; +DISTANCE(v1, v2, "COSINE") +0 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), v2, "COSINE") FROM t1 WHERE id = 98; +ERROR HY000: Incorrect arguments to distance +# Binary-string BLOB arguments exceeding max_dimensions (16383) are rejected. +# A BLOB column is used so the argument passes the resolve-time binary-charset +# type check; the max_dimensions guard fires at runtime in val_real(). +CREATE TABLE t_oversized (v MEDIUMBLOB); +INSERT INTO t_oversized VALUES (REPEAT(X'00000000', 16384)); +SELECT DISTANCE(v, v, "COSINE") FROM t_oversized; +ERROR HY000: Incorrect arguments to distance +DROP TABLE t_oversized; +# +# 3) Some basic tests for different (from syntax PoV) variants of +# arguments. +# +SELECT DISTANCE(X'0000803F0000803F', X'0000000000000040', "COSINE"); +DISTANCE(X'0000803F0000803F', X'0000000000000040', "COSINE") +0.29289321881345254 +SELECT DISTANCE(X'0000803F0000803F', TO_VECTOR("[2, 0]"), "COSINE"); +DISTANCE(X'0000803F0000803F', TO_VECTOR("[2, 0]"), "COSINE") +0.29289321881345254 +SELECT DISTANCE(X'0000803F0000803F', v2, "COSINE") FROM t1 WHERE id = 4; +DISTANCE(X'0000803F0000803F', v2, "COSINE") +0.29289321881345254 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "COSINE"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "COSINE") +NULL +SELECT DISTANCE(TO_VECTOR("[0, 0]"), X'000000000000803F', "COSINE"); +DISTANCE(TO_VECTOR("[0, 0]"), X'000000000000803F', "COSINE") +NULL +SELECT DISTANCE(TO_VECTOR("[0, 0]"), v2, "COSINE") FROM t1 WHERE id = 1; +DISTANCE(TO_VECTOR("[0, 0]"), v2, "COSINE") +NULL +SELECT DISTANCE(a.v2, b.v2, "COSINE") FROM t1 AS a, t1 AS b WHERE a.id = 0 AND b.id = 4; +DISTANCE(a.v2, b.v2, "COSINE") +NULL +SELECT DISTANCE(v2, X'0000000000000040', "COSINE") FROM t1 WHERE id = 0; +DISTANCE(v2, X'0000000000000040', "COSINE") +NULL +SELECT DISTANCE(v2, TO_VECTOR("[0, 2]"), "COSINE") FROM t1 WHERE id = 0; +DISTANCE(v2, TO_VECTOR("[0, 2]"), "COSINE") +NULL +# Non-trivial (artificial) combinations +SELECT DISTANCE(TO_VECTOR(CONCAT("[0", ", ", "1]")), CONCAT(X'00000000', X'00000040'), "COSINE"); +DISTANCE(TO_VECTOR(CONCAT("[0", ", ", "1]")), CONCAT(X'00000000', X'00000040'), "COSINE") +0 +# The below case demonstrates that arguments to DISTANCE might not be +# well-aligned in memory. +SELECT DISTANCE(SUBSTR(X'010000000000000040', 2), RIGHT(X'40000000000000803F', 8), "COSINE"); +DISTANCE(SUBSTR(X'010000000000000040', 2), RIGHT(X'40000000000000803F', 8), "COSINE") +0 +# 9-byte blobs; SUBSTR from pos 2 → 8 bytes at offset 1 (misaligned for float). +# Length must stay a multiple of 4; SUBSTR(..., 4) on 9 bytes yields 6 → ER_TO_VECTOR_CONVERSION. +SELECT DISTANCE(SUBSTR(X'000100000000000040', 2), SUBSTR(X'00040000000000803F', 2), "COSINE"); +DISTANCE(SUBSTR(X'000100000000000040', 2), SUBSTR(X'00040000000000803F', 2), "COSINE") +0 +# +# 4) Basic test for different vector values. +# +# Identical / collinear vectors. +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 1]"), "COSINE"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 1]"), "COSINE") +0 +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[1, 0]"), "COSINE"); +DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[1, 0]"), "COSINE") +0 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2.5, 2.5]"), "COSINE"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2.5, 2.5]"), "COSINE") +0 +SELECT DISTANCE(TO_VECTOR("[1, 2, 3, 4, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "COSINE"); +DISTANCE(TO_VECTOR("[1, 2, 3, 4, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "COSINE") +0 +# Orthogonal vectors. +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "COSINE"); +DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "COSINE") +1 +SELECT DISTANCE(TO_VECTOR("[0, 1, 0]"), TO_VECTOR("[-1, 0, -1]"), "COSINE"); +DISTANCE(TO_VECTOR("[0, 1, 0]"), TO_VECTOR("[-1, 0, -1]"), "COSINE") +1 +SELECT DISTANCE(TO_VECTOR("[1, 0, 3, 0, 5]"), TO_VECTOR("[0, 2, 0, 4, 0]"), "COSINE"); +DISTANCE(TO_VECTOR("[1, 0, 3, 0, 5]"), TO_VECTOR("[0, 2, 0, 4, 0]"), "COSINE") +1 +# Anti-parallel vectors. +SELECT DISTANCE(TO_VECTOR("[-1, -1]"), TO_VECTOR("[2, 2]"), "COSINE"); +DISTANCE(TO_VECTOR("[-1, -1]"), TO_VECTOR("[2, 2]"), "COSINE") +2 +SELECT DISTANCE(TO_VECTOR("[-2e38, 1]"), TO_VECTOR("[2e38, -1]"), "COSINE"); +DISTANCE(TO_VECTOR("[-2e38, 1]"), TO_VECTOR("[2e38, -1]"), "COSINE") +2 +# Distance from origin. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "COSINE"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "COSINE") +NULL +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[3, 4]"), "COSINE"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[3, 4]"), "COSINE") +NULL +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 12]"), "COSINE"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 12]"), "COSINE") +NULL +SELECT DISTANCE(TO_VECTOR("[0, 0, 0, 0]"), TO_VECTOR("[1, 1, 1, 1]"), "COSINE"); +DISTANCE(TO_VECTOR("[0, 0, 0, 0]"), TO_VECTOR("[1, 1, 1, 1]"), "COSINE") +NULL +# Mixed-sign and larger vectors. +SELECT DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "COSINE"); +DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "COSINE") +0.025368153802923787 +SELECT DISTANCE(TO_VECTOR("[1, 7, 3, 16, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "COSINE"); +DISTANCE(TO_VECTOR("[1, 7, 3, 16, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "COSINE") +0.17366216073634244 +# Zero vector (behavior differs per metric). +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2, 2]"), "COSINE"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2, 2]"), "COSINE") +NULL +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[0, 0]"), "COSINE"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[0, 0]"), "COSINE") +NULL +SELECT DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[0]"), "COSINE"); +DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[0]"), "COSINE") +NULL +# Large values near float32 max. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2e38, 0]"), "COSINE"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2e38, 0]"), "COSINE") +NULL +# Same value in a 16-dim vector: exercises the wide-tier SIMD overflow +# fallback (dims >= 16 dispatches to the wide kernel; squaring 2e38 in +# float32 overflows to +Inf, but the isfinite check falls back to scalar). +SELECT DISTANCE(TO_VECTOR("[2e38, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +TO_VECTOR("[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +"COSINE"); +DISTANCE(TO_VECTOR("[2e38, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +TO_VECTOR("[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +"COSINE") +NULL +# Symmetry: DISTANCE(a, b) = DISTANCE(b, a). +SELECT DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "COSINE") = +DISTANCE(TO_VECTOR("[4, 5, 6]"), TO_VECTOR("[1, 2, 3]"), "COSINE"); +DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "COSINE") = +DISTANCE(TO_VECTOR("[4, 5, 6]"), TO_VECTOR("[1, 2, 3]"), "COSINE") +1 +# Special IEEE 754 float32 values: NaN, +Infinity, -Infinity. +# NaN/Inf input elements raise ER_DATA_OUT_OF_RANGE for all metrics (POW/EXP convention). +SELECT DISTANCE(X'0000C07F', X'00000000', "COSINE"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x0000c07f,0x00000000,'COSINE')' +SELECT DISTANCE(X'0000807F', X'00000000', "COSINE"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x0000807f,0x00000000,'COSINE')' +SELECT DISTANCE(X'000080FF', X'00000000', "COSINE"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x000080ff,0x00000000,'COSINE')' +# Wide-tier SIMD path coverage (dims >= 16 dispatches to the wide kernel). +# Integer-valued diffs keep float32 partial sums exact, so results are +# identical across Scalar / SSE4.2 / NEON / AVX2 / AVX-512 / SVE2. +# 16-dim: fills one AVX-512 register / two AVX2 / four SSE4.2 -- no scalar tail. +SELECT DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"COSINE"); +DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"COSINE") +1 +# 20-dim: SSE4.2 5x4 (no tail); AVX2 2x8 + 4-elem scalar tail; +# AVX-512 1x16 + 4-elem scalar tail. +SELECT DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"COSINE"); +DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"COSINE") +1 +# +# prepared statements +# +PREPARE s1 FROM 'SELECT DISTANCE(v1, TO_VECTOR("[2]"), "COSINE") FROM t1 WHERE id = 4'; +EXECUTE s1; +DISTANCE(v1, TO_VECTOR("[2]"), "COSINE") +0 +EXECUTE s1; +DISTANCE(v1, TO_VECTOR("[2]"), "COSINE") +0 +DEALLOCATE PREPARE s1; +PREPARE stmt FROM 'SELECT DISTANCE(v1, TO_VECTOR("[2]"), "?") FROM t1 WHERE id = 4'; +ERROR HY000: Incorrect arguments to distance +PREPARE stmt FROM 'SELECT DISTANCE(v1, TO_VECTOR(?), "COSINE") FROM t1 WHERE id = 4'; +SET @met="[2]"; +EXECUTE stmt USING @met; +DISTANCE(v1, TO_VECTOR(?), "COSINE") +0 +DEALLOCATE PREPARE stmt; +# +# 5) Distance in query contexts. +# +# ORDER BY distance: nearest-neighbour pattern. +SELECT id FROM t1 WHERE id IN (0,1,2,3,4) +ORDER BY DISTANCE(v2, TO_VECTOR('[1, 0]'), "COSINE"), id; +id +0 +1 +4 +3 +2 +# ORDER BY distance DESC: farthest-neighbour pattern. +SELECT id FROM t1 WHERE id IN (0,1,2,3,4) +ORDER BY DISTANCE(v2, TO_VECTOR('[1, 0]'), "COSINE") DESC, id; +id +2 +3 +1 +4 +0 +# WHERE: range query filtering by distance. +SELECT id FROM t1 +WHERE id IN (0,1,2,3,4) AND DISTANCE(v2, TO_VECTOR('[1, 0]'), "COSINE") < 1.5 +ORDER BY id; +id +1 +2 +3 +4 +# Derived table with distance. +SELECT id FROM +(SELECT id, DISTANCE(v2, TO_VECTOR('[1, 0]'), "COSINE") AS d +FROM t1 WHERE id IN (0,1,2,3,4)) AS sq +WHERE d IS NOT NULL ORDER BY d, id; +id +1 +4 +3 +2 +# Zero-vector cosine in DML: must insert NULL without aborting under strict sql_mode. +CREATE TABLE tt_cosine_dml (d DOUBLE); +INSERT INTO tt_cosine_dml SELECT DISTANCE(TO_VECTOR('[0]'), TO_VECTOR('[0]'), 'COSINE'); +SELECT d FROM tt_cosine_dml; +d +NULL +DROP TABLE tt_cosine_dml; +DROP TABLE t_metric_name; +DROP TABLE t1; diff --git a/mysql-test/suite/percona/r/distance_dot.result b/mysql-test/suite/percona/r/distance_dot.result new file mode 100644 index 000000000000..2c6f71adee23 --- /dev/null +++ b/mysql-test/suite/percona/r/distance_dot.result @@ -0,0 +1,358 @@ +# +# Test coverage for vector DISTANCE() function. +# +# +# 0) Prepare playground. +# +CREATE TABLE t1 (id INT PRIMARY KEY, v1 VECTOR(1), v2 VECTOR(2)); +INSERT INTO t1 VALUES (0, TO_VECTOR('[0]'), TO_VECTOR('[0, 0]')), +(1, TO_VECTOR('[1]'), TO_VECTOR('[1, 0]')), +(2, TO_VECTOR('[1]'), TO_VECTOR('[0, 1]')), +(3, TO_VECTOR('[2]'), TO_VECTOR('[1, 1]')), +(4, TO_VECTOR('[2]'), TO_VECTOR('[2, 0]')), +(98, TO_VECTOR('[1]'), TO_VECTOR('[2]')), +(99, NULL, NULL); +CREATE TABLE t_metric_name (id INT PRIMARY KEY, name VARCHAR(10)); +INSERT INTO t_metric_name VALUES (1, "EUCLIDEAN"), (99, NULL); +# +# 1) Test how different number and types of arguments are handled. +# +# 1.1) Arity. +# +SELECT DISTANCE(); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]")); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]")); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), "DOT"); +DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), "DOT") +-2 +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), TO_VECTOR("[3]"), "EUCLIDEAN"); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +# +# 1.2) Argument types. +# +# Only vectors or binary strings are allowed for first the two arguments. +SELECT DISTANCE("[1]", TO_VECTOR("[2]"), "DOT"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(X'0000803F', TO_VECTOR("[2]"), "DOT"); +DISTANCE(X'0000803F', TO_VECTOR("[2]"), "DOT") +-2 +SELECT DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[2]"), "DOT"); +DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[2]"), "DOT") +-0 +SELECT DISTANCE(v1, TO_VECTOR("[2]"), "DOT") FROM t1 WHERE id = 0; +DISTANCE(v1, TO_VECTOR("[2]"), "DOT") +-0 +SELECT DISTANCE(id, TO_VECTOR("[2]"), "DOT") FROM t1 WHERE id = 0; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1]"), "[2]", "DOT"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[2]"), X'0000803F', "DOT"); +DISTANCE(TO_VECTOR("[2]"), X'0000803F', "DOT") +-2 +SELECT DISTANCE(TO_VECTOR("[3]"), v1, "DOT") FROM t1 WHERE id = 1; +DISTANCE(TO_VECTOR("[3]"), v1, "DOT") +-3 +SELECT DISTANCE(TO_VECTOR("[0]"), id, "DOT") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +# The third argument must be a string literal with value from the +# fixed list of metric names. +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[-1, 0]"), 1); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 0]"), CONCAT("EUCLI","DEAN")); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2, 0]"), "euclidean"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2, 0]"), "euclidean") +1.4142135623730951 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[3, 0]"), "EuClIdEaN"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[3, 0]"), "EuClIdEaN") +2.23606797749979 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[4, 0]"), X'4555434C494445414E'); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[4, 0]"), X'4555434C494445414E') +3.1622776601683795 +# Metric strings with embedded NUL must be rejected regardless of prefix match. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), X'4555434C494445414E00'); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), X'4555434C494445414E004A554E4B'); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 0]"), "NOSUCHMETRIC"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[6, 0]"), name) FROM t_metric_name WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[7, 0]"), NULL); +ERROR HY000: Incorrect arguments to distance +# +# 1.3) NULL arguments and nullability in metadata for result. +# +SELECT DISTANCE(NULL, TO_VECTOR("[1, 0]"), "DOT"); +DISTANCE(NULL, TO_VECTOR("[1, 0]"), "DOT") +NULL +SELECT DISTANCE(TO_VECTOR("[1, 1]"), NULL, "DOT"); +DISTANCE(TO_VECTOR("[1, 1]"), NULL, "DOT") +NULL +SELECT DISTANCE(v2, TO_VECTOR("[1, 0]"), "DOT") FROM t1 WHERE id = 99; +DISTANCE(v2, TO_VECTOR("[1, 0]"), "DOT") +NULL +SELECT DISTANCE(TO_VECTOR("[1, 1]"), v2, "DOT") FROM t1 WHERE id = 99; +DISTANCE(TO_VECTOR("[1, 1]"), v2, "DOT") +NULL +# The third argument doesn't allow NULL values in any form. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), NULL); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), name) FROM t_metric_name WHERE id = 99; +ERROR HY000: Incorrect arguments to distance +# The result metadata should indicate that it is nullable. +CREATE TABLE tt SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "DOT") AS d; +SHOW CREATE TABLE tt; +Table Create Table +tt CREATE TABLE `tt` ( + `d` double DEFAULT NULL +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_0900_ai_ci +DROP TABLE tt; +# +# 2) Test vector arguments length mismatch. +# +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[1, 0]"), "DOT"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(v2, TO_VECTOR("[1]"), "DOT") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(v1, v2, "DOT") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +# +# Note that length check happens at runtime. This is well visible +# when we have value stored in a vector field which is shorter than +# maximum length specified at the field creation time. +SELECT DISTANCE(v1, v2, "DOT") FROM t1 WHERE id = 98; +DISTANCE(v1, v2, "DOT") +-2 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), v2, "DOT") FROM t1 WHERE id = 98; +ERROR HY000: Incorrect arguments to distance +# Binary-string BLOB arguments exceeding max_dimensions (16383) are rejected. +# A BLOB column is used so the argument passes the resolve-time binary-charset +# type check; the max_dimensions guard fires at runtime in val_real(). +CREATE TABLE t_oversized (v MEDIUMBLOB); +INSERT INTO t_oversized VALUES (REPEAT(X'00000000', 16384)); +SELECT DISTANCE(v, v, "DOT") FROM t_oversized; +ERROR HY000: Incorrect arguments to distance +DROP TABLE t_oversized; +# +# 3) Some basic tests for different (from syntax PoV) variants of +# arguments. +# +SELECT DISTANCE(X'0000803F0000803F', X'0000000000000040', "DOT"); +DISTANCE(X'0000803F0000803F', X'0000000000000040', "DOT") +-2 +SELECT DISTANCE(X'0000803F0000803F', TO_VECTOR("[2, 0]"), "DOT"); +DISTANCE(X'0000803F0000803F', TO_VECTOR("[2, 0]"), "DOT") +-2 +SELECT DISTANCE(X'0000803F0000803F', v2, "DOT") FROM t1 WHERE id = 4; +DISTANCE(X'0000803F0000803F', v2, "DOT") +-2 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "DOT"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "DOT") +-0 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), X'000000000000803F', "DOT"); +DISTANCE(TO_VECTOR("[0, 0]"), X'000000000000803F', "DOT") +-0 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), v2, "DOT") FROM t1 WHERE id = 1; +DISTANCE(TO_VECTOR("[0, 0]"), v2, "DOT") +-0 +SELECT DISTANCE(a.v2, b.v2, "DOT") FROM t1 AS a, t1 AS b WHERE a.id = 0 AND b.id = 4; +DISTANCE(a.v2, b.v2, "DOT") +-0 +SELECT DISTANCE(v2, X'0000000000000040', "DOT") FROM t1 WHERE id = 0; +DISTANCE(v2, X'0000000000000040', "DOT") +-0 +SELECT DISTANCE(v2, TO_VECTOR("[0, 2]"), "DOT") FROM t1 WHERE id = 0; +DISTANCE(v2, TO_VECTOR("[0, 2]"), "DOT") +-0 +# Non-trivial (artificial) combinations +SELECT DISTANCE(TO_VECTOR(CONCAT("[0", ", ", "1]")), CONCAT(X'00000000', X'00000040'), "DOT"); +DISTANCE(TO_VECTOR(CONCAT("[0", ", ", "1]")), CONCAT(X'00000000', X'00000040'), "DOT") +-2 +# The below case demonstrates that arguments to DISTANCE might not be +# well-aligned in memory. +SELECT DISTANCE(SUBSTR(X'010000000000000040', 2), RIGHT(X'40000000000000803F', 8), "DOT"); +DISTANCE(SUBSTR(X'010000000000000040', 2), RIGHT(X'40000000000000803F', 8), "DOT") +-2 +# 9-byte blobs; SUBSTR from pos 2 → 8 bytes at offset 1 (misaligned for float). +# Length must stay a multiple of 4; SUBSTR(..., 4) on 9 bytes yields 6 → ER_TO_VECTOR_CONVERSION. +SELECT DISTANCE(SUBSTR(X'000100000000000040', 2), SUBSTR(X'00040000000000803F', 2), "DOT"); +DISTANCE(SUBSTR(X'000100000000000040', 2), SUBSTR(X'00040000000000803F', 2), "DOT") +-2 +# +# 4) Basic test for different vector values. +# +# Identical / collinear vectors. +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 1]"), "DOT"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 1]"), "DOT") +-2 +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[1, 0]"), "DOT"); +DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[1, 0]"), "DOT") +-1 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2.5, 2.5]"), "DOT"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2.5, 2.5]"), "DOT") +-5 +SELECT DISTANCE(TO_VECTOR("[1, 2, 3, 4, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "DOT"); +DISTANCE(TO_VECTOR("[1, 2, 3, 4, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "DOT") +-55 +# Orthogonal vectors. +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "DOT"); +DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "DOT") +-0 +SELECT DISTANCE(TO_VECTOR("[0, 1, 0]"), TO_VECTOR("[-1, 0, -1]"), "DOT"); +DISTANCE(TO_VECTOR("[0, 1, 0]"), TO_VECTOR("[-1, 0, -1]"), "DOT") +-0 +SELECT DISTANCE(TO_VECTOR("[1, 0, 3, 0, 5]"), TO_VECTOR("[0, 2, 0, 4, 0]"), "DOT"); +DISTANCE(TO_VECTOR("[1, 0, 3, 0, 5]"), TO_VECTOR("[0, 2, 0, 4, 0]"), "DOT") +-0 +# Anti-parallel vectors. +SELECT DISTANCE(TO_VECTOR("[-1, -1]"), TO_VECTOR("[2, 2]"), "DOT"); +DISTANCE(TO_VECTOR("[-1, -1]"), TO_VECTOR("[2, 2]"), "DOT") +4 +SELECT DISTANCE(TO_VECTOR("[-2e38, 1]"), TO_VECTOR("[2e38, -1]"), "DOT"); +DISTANCE(TO_VECTOR("[-2e38, 1]"), TO_VECTOR("[2e38, -1]"), "DOT") +3.999999744228558e76 +# Distance from origin. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "DOT"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "DOT") +-0 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[3, 4]"), "DOT"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[3, 4]"), "DOT") +-0 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 12]"), "DOT"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 12]"), "DOT") +-0 +SELECT DISTANCE(TO_VECTOR("[0, 0, 0, 0]"), TO_VECTOR("[1, 1, 1, 1]"), "DOT"); +DISTANCE(TO_VECTOR("[0, 0, 0, 0]"), TO_VECTOR("[1, 1, 1, 1]"), "DOT") +-0 +# Mixed-sign and larger vectors. +SELECT DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "DOT"); +DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "DOT") +-32 +SELECT DISTANCE(TO_VECTOR("[1, 7, 3, 16, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "DOT"); +DISTANCE(TO_VECTOR("[1, 7, 3, 16, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "DOT") +-113 +# Zero vector (behavior differs per metric). +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2, 2]"), "DOT"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2, 2]"), "DOT") +-0 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[0, 0]"), "DOT"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[0, 0]"), "DOT") +-0 +SELECT DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[0]"), "DOT"); +DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[0]"), "DOT") +-0 +# Large values near float32 max. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2e38, 0]"), "DOT"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2e38, 0]"), "DOT") +-0 +# Same value in a 16-dim vector: exercises the wide-tier SIMD overflow +# fallback (dims >= 16 dispatches to the wide kernel; squaring 2e38 in +# float32 overflows to +Inf, but the isfinite check falls back to scalar). +SELECT DISTANCE(TO_VECTOR("[2e38, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +TO_VECTOR("[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +"DOT"); +DISTANCE(TO_VECTOR("[2e38, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +TO_VECTOR("[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +"DOT") +-0 +# Symmetry: DISTANCE(a, b) = DISTANCE(b, a). +SELECT DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "DOT") = +DISTANCE(TO_VECTOR("[4, 5, 6]"), TO_VECTOR("[1, 2, 3]"), "DOT"); +DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "DOT") = +DISTANCE(TO_VECTOR("[4, 5, 6]"), TO_VECTOR("[1, 2, 3]"), "DOT") +1 +# Special IEEE 754 float32 values: NaN, +Infinity, -Infinity. +# NaN/Inf input elements raise ER_DATA_OUT_OF_RANGE for all metrics (POW/EXP convention). +SELECT DISTANCE(X'0000C07F', X'00000000', "DOT"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x0000c07f,0x00000000,'DOT')' +SELECT DISTANCE(X'0000807F', X'00000000', "DOT"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x0000807f,0x00000000,'DOT')' +SELECT DISTANCE(X'000080FF', X'00000000', "DOT"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x000080ff,0x00000000,'DOT')' +# Wide-tier SIMD path coverage (dims >= 16 dispatches to the wide kernel). +# Integer-valued diffs keep float32 partial sums exact, so results are +# identical across Scalar / SSE4.2 / NEON / AVX2 / AVX-512 / SVE2. +# 16-dim: fills one AVX-512 register / two AVX2 / four SSE4.2 -- no scalar tail. +SELECT DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"DOT"); +DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"DOT") +-0 +# 20-dim: SSE4.2 5x4 (no tail); AVX2 2x8 + 4-elem scalar tail; +# AVX-512 1x16 + 4-elem scalar tail. +SELECT DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"DOT"); +DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"DOT") +-0 +# +# prepared statements +# +PREPARE s1 FROM 'SELECT DISTANCE(v1, TO_VECTOR("[2]"), "DOT") FROM t1 WHERE id = 4'; +EXECUTE s1; +DISTANCE(v1, TO_VECTOR("[2]"), "DOT") +-4 +EXECUTE s1; +DISTANCE(v1, TO_VECTOR("[2]"), "DOT") +-4 +DEALLOCATE PREPARE s1; +PREPARE stmt FROM 'SELECT DISTANCE(v1, TO_VECTOR("[2]"), "?") FROM t1 WHERE id = 4'; +ERROR HY000: Incorrect arguments to distance +PREPARE stmt FROM 'SELECT DISTANCE(v1, TO_VECTOR(?), "DOT") FROM t1 WHERE id = 4'; +SET @met="[2]"; +EXECUTE stmt USING @met; +DISTANCE(v1, TO_VECTOR(?), "DOT") +-4 +DEALLOCATE PREPARE stmt; +# +# 5) Distance in query contexts. +# +# ORDER BY distance: nearest-neighbour pattern. +SELECT id FROM t1 WHERE id IN (0,1,2,3,4) +ORDER BY DISTANCE(v2, TO_VECTOR('[1, 0]'), "DOT"), id; +id +4 +1 +3 +0 +2 +# ORDER BY distance DESC: farthest-neighbour pattern. +SELECT id FROM t1 WHERE id IN (0,1,2,3,4) +ORDER BY DISTANCE(v2, TO_VECTOR('[1, 0]'), "DOT") DESC, id; +id +0 +2 +1 +3 +4 +# WHERE: range query filtering by distance. +SELECT id FROM t1 +WHERE id IN (0,1,2,3,4) AND DISTANCE(v2, TO_VECTOR('[1, 0]'), "DOT") < 1.5 +ORDER BY id; +id +0 +1 +2 +3 +4 +# Derived table with distance. +SELECT id FROM +(SELECT id, DISTANCE(v2, TO_VECTOR('[1, 0]'), "DOT") AS d +FROM t1 WHERE id IN (0,1,2,3,4)) AS sq +WHERE d IS NOT NULL ORDER BY d, id; +id +4 +1 +3 +0 +2 +DROP TABLE t_metric_name; +DROP TABLE t1; diff --git a/mysql-test/suite/percona/r/distance_euclidean.result b/mysql-test/suite/percona/r/distance_euclidean.result new file mode 100644 index 000000000000..9a0c2ae85448 --- /dev/null +++ b/mysql-test/suite/percona/r/distance_euclidean.result @@ -0,0 +1,357 @@ +# +# Test coverage for vector DISTANCE() function. +# +# +# 0) Prepare playground. +# +CREATE TABLE t1 (id INT PRIMARY KEY, v1 VECTOR(1), v2 VECTOR(2)); +INSERT INTO t1 VALUES (0, TO_VECTOR('[0]'), TO_VECTOR('[0, 0]')), +(1, TO_VECTOR('[1]'), TO_VECTOR('[1, 0]')), +(2, TO_VECTOR('[1]'), TO_VECTOR('[0, 1]')), +(3, TO_VECTOR('[2]'), TO_VECTOR('[1, 1]')), +(4, TO_VECTOR('[2]'), TO_VECTOR('[2, 0]')), +(98, TO_VECTOR('[1]'), TO_VECTOR('[2]')), +(99, NULL, NULL); +CREATE TABLE t_metric_name (id INT PRIMARY KEY, name VARCHAR(10)); +INSERT INTO t_metric_name VALUES (1, "EUCLIDEAN"), (99, NULL); +# +# 1) Test how different number and types of arguments are handled. +# +# 1.1) Arity. +# +SELECT DISTANCE(); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]")); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]")); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), "EUCLIDEAN") +1 +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), TO_VECTOR("[3]"), "EUCLIDEAN"); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +# +# 1.2) Argument types. +# +# Only vectors or binary strings are allowed for first the two arguments. +SELECT DISTANCE("[1]", TO_VECTOR("[2]"), "EUCLIDEAN"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(X'0000803F', TO_VECTOR("[2]"), "EUCLIDEAN"); +DISTANCE(X'0000803F', TO_VECTOR("[2]"), "EUCLIDEAN") +1 +SELECT DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[2]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[2]"), "EUCLIDEAN") +2 +SELECT DISTANCE(v1, TO_VECTOR("[2]"), "EUCLIDEAN") FROM t1 WHERE id = 0; +DISTANCE(v1, TO_VECTOR("[2]"), "EUCLIDEAN") +2 +SELECT DISTANCE(id, TO_VECTOR("[2]"), "EUCLIDEAN") FROM t1 WHERE id = 0; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1]"), "[2]", "EUCLIDEAN"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[2]"), X'0000803F', "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[2]"), X'0000803F', "EUCLIDEAN") +1 +SELECT DISTANCE(TO_VECTOR("[3]"), v1, "EUCLIDEAN") FROM t1 WHERE id = 1; +DISTANCE(TO_VECTOR("[3]"), v1, "EUCLIDEAN") +2 +SELECT DISTANCE(TO_VECTOR("[0]"), id, "EUCLIDEAN") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +# The third argument must be a string literal with value from the +# fixed list of metric names. +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[-1, 0]"), 1); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 0]"), CONCAT("EUCLI","DEAN")); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2, 0]"), "euclidean"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2, 0]"), "euclidean") +1.4142135623730951 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[3, 0]"), "EuClIdEaN"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[3, 0]"), "EuClIdEaN") +2.23606797749979 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[4, 0]"), X'4555434C494445414E'); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[4, 0]"), X'4555434C494445414E') +3.1622776601683795 +# Metric strings with embedded NUL must be rejected regardless of prefix match. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), X'4555434C494445414E00'); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), X'4555434C494445414E004A554E4B'); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 0]"), "NOSUCHMETRIC"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[6, 0]"), name) FROM t_metric_name WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[7, 0]"), NULL); +ERROR HY000: Incorrect arguments to distance +# +# 1.3) NULL arguments and nullability in metadata for result. +# +SELECT DISTANCE(NULL, TO_VECTOR("[1, 0]"), "EUCLIDEAN"); +DISTANCE(NULL, TO_VECTOR("[1, 0]"), "EUCLIDEAN") +NULL +SELECT DISTANCE(TO_VECTOR("[1, 1]"), NULL, "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1, 1]"), NULL, "EUCLIDEAN") +NULL +SELECT DISTANCE(v2, TO_VECTOR("[1, 0]"), "EUCLIDEAN") FROM t1 WHERE id = 99; +DISTANCE(v2, TO_VECTOR("[1, 0]"), "EUCLIDEAN") +NULL +SELECT DISTANCE(TO_VECTOR("[1, 1]"), v2, "EUCLIDEAN") FROM t1 WHERE id = 99; +DISTANCE(TO_VECTOR("[1, 1]"), v2, "EUCLIDEAN") +NULL +# The third argument doesn't allow NULL values in any form. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), NULL); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), name) FROM t_metric_name WHERE id = 99; +ERROR HY000: Incorrect arguments to distance +# The result metadata should indicate that it is nullable. +CREATE TABLE tt SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "EUCLIDEAN") AS d; +SHOW CREATE TABLE tt; +Table Create Table +tt CREATE TABLE `tt` ( + `d` double DEFAULT NULL +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_0900_ai_ci +DROP TABLE tt; +# +# 2) Test vector arguments length mismatch. +# +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(v2, TO_VECTOR("[1]"), "EUCLIDEAN") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(v1, v2, "EUCLIDEAN") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +# +# Note that length check happens at runtime. This is well visible +# when we have value stored in a vector field which is shorter than +# maximum length specified at the field creation time. +SELECT DISTANCE(v1, v2, "EUCLIDEAN") FROM t1 WHERE id = 98; +DISTANCE(v1, v2, "EUCLIDEAN") +1 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), v2, "EUCLIDEAN") FROM t1 WHERE id = 98; +ERROR HY000: Incorrect arguments to distance +# Binary-string BLOB arguments exceeding max_dimensions (16383) are rejected. +# A BLOB column is used so the argument passes the resolve-time binary-charset +# type check; the max_dimensions guard fires at runtime in val_real(). +CREATE TABLE t_oversized (v MEDIUMBLOB); +INSERT INTO t_oversized VALUES (REPEAT(X'00000000', 16384)); +SELECT DISTANCE(v, v, "EUCLIDEAN") FROM t_oversized; +ERROR HY000: Incorrect arguments to distance +DROP TABLE t_oversized; +# +# 3) Some basic tests for different (from syntax PoV) variants of +# arguments. +# +SELECT DISTANCE(X'0000803F0000803F', X'0000000000000040', "EUCLIDEAN"); +DISTANCE(X'0000803F0000803F', X'0000000000000040', "EUCLIDEAN") +1.4142135623730951 +SELECT DISTANCE(X'0000803F0000803F', TO_VECTOR("[2, 0]"), "EUCLIDEAN"); +DISTANCE(X'0000803F0000803F', TO_VECTOR("[2, 0]"), "EUCLIDEAN") +1.4142135623730951 +SELECT DISTANCE(X'0000803F0000803F', v2, "EUCLIDEAN") FROM t1 WHERE id = 4; +DISTANCE(X'0000803F0000803F', v2, "EUCLIDEAN") +1.4142135623730951 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN") +1 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), X'000000000000803F', "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[0, 0]"), X'000000000000803F', "EUCLIDEAN") +1 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), v2, "EUCLIDEAN") FROM t1 WHERE id = 1; +DISTANCE(TO_VECTOR("[0, 0]"), v2, "EUCLIDEAN") +1 +SELECT DISTANCE(a.v2, b.v2, "EUCLIDEAN") FROM t1 AS a, t1 AS b WHERE a.id = 0 AND b.id = 4; +DISTANCE(a.v2, b.v2, "EUCLIDEAN") +2 +SELECT DISTANCE(v2, X'0000000000000040', "EUCLIDEAN") FROM t1 WHERE id = 0; +DISTANCE(v2, X'0000000000000040', "EUCLIDEAN") +2 +SELECT DISTANCE(v2, TO_VECTOR("[0, 2]"), "EUCLIDEAN") FROM t1 WHERE id = 0; +DISTANCE(v2, TO_VECTOR("[0, 2]"), "EUCLIDEAN") +2 +# Non-trivial (artificial) combinations +SELECT DISTANCE(TO_VECTOR(CONCAT("[0", ", ", "1]")), CONCAT(X'00000000', X'00000040'), "EUCLIDEAN"); +DISTANCE(TO_VECTOR(CONCAT("[0", ", ", "1]")), CONCAT(X'00000000', X'00000040'), "EUCLIDEAN") +1 +# The below case demonstrates that arguments to DISTANCE might not be +# well-aligned in memory. +SELECT DISTANCE(SUBSTR(X'010000000000000040', 2), RIGHT(X'40000000000000803F', 8), "EUCLIDEAN"); +DISTANCE(SUBSTR(X'010000000000000040', 2), RIGHT(X'40000000000000803F', 8), "EUCLIDEAN") +1 +# 9-byte blobs; SUBSTR from pos 2 → 8 bytes at offset 1 (misaligned for float). +# Length must stay a multiple of 4; SUBSTR(..., 4) on 9 bytes yields 6 → ER_TO_VECTOR_CONVERSION. +SELECT DISTANCE(SUBSTR(X'000100000000000040', 2), SUBSTR(X'00040000000000803F', 2), "EUCLIDEAN"); +DISTANCE(SUBSTR(X'000100000000000040', 2), SUBSTR(X'00040000000000803F', 2), "EUCLIDEAN") +1 +# +# 4) Basic test for different vector values. +# +# Identical / collinear vectors. +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 1]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 1]"), "EUCLIDEAN") +0 +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN") +0 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2.5, 2.5]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2.5, 2.5]"), "EUCLIDEAN") +2.1213203435596424 +SELECT DISTANCE(TO_VECTOR("[1, 2, 3, 4, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1, 2, 3, 4, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "EUCLIDEAN") +0 +# Orthogonal vectors. +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "EUCLIDEAN") +1.4142135623730951 +SELECT DISTANCE(TO_VECTOR("[0, 1, 0]"), TO_VECTOR("[-1, 0, -1]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[0, 1, 0]"), TO_VECTOR("[-1, 0, -1]"), "EUCLIDEAN") +1.7320508075688772 +SELECT DISTANCE(TO_VECTOR("[1, 0, 3, 0, 5]"), TO_VECTOR("[0, 2, 0, 4, 0]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1, 0, 3, 0, 5]"), TO_VECTOR("[0, 2, 0, 4, 0]"), "EUCLIDEAN") +7.416198487095663 +# Anti-parallel vectors. +SELECT DISTANCE(TO_VECTOR("[-1, -1]"), TO_VECTOR("[2, 2]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[-1, -1]"), TO_VECTOR("[2, 2]"), "EUCLIDEAN") +4.242640687119285 +SELECT DISTANCE(TO_VECTOR("[-2e38, 1]"), TO_VECTOR("[2e38, -1]"), "EUCLIDEAN"); +ERROR 22003: DOUBLE value is out of range in 'distance(to_vector('[-2e38, 1]'),to_vector('[2e38, -1]'),'EUCLIDEAN')' +# Distance from origin. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN") +1 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[3, 4]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[3, 4]"), "EUCLIDEAN") +5 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 12]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 12]"), "EUCLIDEAN") +13 +SELECT DISTANCE(TO_VECTOR("[0, 0, 0, 0]"), TO_VECTOR("[1, 1, 1, 1]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[0, 0, 0, 0]"), TO_VECTOR("[1, 1, 1, 1]"), "EUCLIDEAN") +2 +# Mixed-sign and larger vectors. +SELECT DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "EUCLIDEAN") +5.196152422706632 +SELECT DISTANCE(TO_VECTOR("[1, 7, 3, 16, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1, 7, 3, 16, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "EUCLIDEAN") +13 +# Zero vector (behavior differs per metric). +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2, 2]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2, 2]"), "EUCLIDEAN") +2.8284271247461903 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[0, 0]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[0, 0]"), "EUCLIDEAN") +0 +SELECT DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[0]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[0]"), "EUCLIDEAN") +0 +# Large values near float32 max. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2e38, 0]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2e38, 0]"), "EUCLIDEAN") +1.9999999360571385e38 +# Same value in a 16-dim vector: exercises the wide-tier SIMD overflow +# fallback (dims >= 16 dispatches to the wide kernel; squaring 2e38 in +# float32 overflows to +Inf, but the isfinite check falls back to scalar). +SELECT DISTANCE(TO_VECTOR("[2e38, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +TO_VECTOR("[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +"EUCLIDEAN"); +DISTANCE(TO_VECTOR("[2e38, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +TO_VECTOR("[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +"EUCLIDEAN") +1.9999999360571385e38 +# Symmetry: DISTANCE(a, b) = DISTANCE(b, a). +SELECT DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "EUCLIDEAN") = +DISTANCE(TO_VECTOR("[4, 5, 6]"), TO_VECTOR("[1, 2, 3]"), "EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "EUCLIDEAN") = +DISTANCE(TO_VECTOR("[4, 5, 6]"), TO_VECTOR("[1, 2, 3]"), "EUCLIDEAN") +1 +# Special IEEE 754 float32 values: NaN, +Infinity, -Infinity. +# NaN/Inf input elements raise ER_DATA_OUT_OF_RANGE for all metrics (POW/EXP convention). +SELECT DISTANCE(X'0000C07F', X'00000000', "EUCLIDEAN"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x0000c07f,0x00000000,'EUCLIDEAN')' +SELECT DISTANCE(X'0000807F', X'00000000', "EUCLIDEAN"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x0000807f,0x00000000,'EUCLIDEAN')' +SELECT DISTANCE(X'000080FF', X'00000000', "EUCLIDEAN"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x000080ff,0x00000000,'EUCLIDEAN')' +# Wide-tier SIMD path coverage (dims >= 16 dispatches to the wide kernel). +# Integer-valued diffs keep float32 partial sums exact, so results are +# identical across Scalar / SSE4.2 / NEON / AVX2 / AVX-512 / SVE2. +# 16-dim: fills one AVX-512 register / two AVX2 / four SSE4.2 -- no scalar tail. +SELECT DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"EUCLIDEAN") +4 +# 20-dim: SSE4.2 5x4 (no tail); AVX2 2x8 + 4-elem scalar tail; +# AVX-512 1x16 + 4-elem scalar tail. +SELECT DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"EUCLIDEAN"); +DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"EUCLIDEAN") +4.47213595499958 +# +# prepared statements +# +PREPARE s1 FROM 'SELECT DISTANCE(v1, TO_VECTOR("[2]"), "EUCLIDEAN") FROM t1 WHERE id = 4'; +EXECUTE s1; +DISTANCE(v1, TO_VECTOR("[2]"), "EUCLIDEAN") +0 +EXECUTE s1; +DISTANCE(v1, TO_VECTOR("[2]"), "EUCLIDEAN") +0 +DEALLOCATE PREPARE s1; +PREPARE stmt FROM 'SELECT DISTANCE(v1, TO_VECTOR("[2]"), "?") FROM t1 WHERE id = 4'; +ERROR HY000: Incorrect arguments to distance +PREPARE stmt FROM 'SELECT DISTANCE(v1, TO_VECTOR(?), "EUCLIDEAN") FROM t1 WHERE id = 4'; +SET @met="[2]"; +EXECUTE stmt USING @met; +DISTANCE(v1, TO_VECTOR(?), "EUCLIDEAN") +0 +DEALLOCATE PREPARE stmt; +# +# 5) Distance in query contexts. +# +# ORDER BY distance: nearest-neighbour pattern. +SELECT id FROM t1 WHERE id IN (0,1,2,3,4) +ORDER BY DISTANCE(v2, TO_VECTOR('[1, 0]'), "EUCLIDEAN"), id; +id +1 +0 +3 +4 +2 +# ORDER BY distance DESC: farthest-neighbour pattern. +SELECT id FROM t1 WHERE id IN (0,1,2,3,4) +ORDER BY DISTANCE(v2, TO_VECTOR('[1, 0]'), "EUCLIDEAN") DESC, id; +id +2 +0 +3 +4 +1 +# WHERE: range query filtering by distance. +SELECT id FROM t1 +WHERE id IN (0,1,2,3,4) AND DISTANCE(v2, TO_VECTOR('[1, 0]'), "EUCLIDEAN") < 1.5 +ORDER BY id; +id +0 +1 +2 +3 +4 +# Derived table with distance. +SELECT id FROM +(SELECT id, DISTANCE(v2, TO_VECTOR('[1, 0]'), "EUCLIDEAN") AS d +FROM t1 WHERE id IN (0,1,2,3,4)) AS sq +WHERE d IS NOT NULL ORDER BY d, id; +id +1 +0 +3 +4 +2 +DROP TABLE t_metric_name; +DROP TABLE t1; diff --git a/mysql-test/suite/percona/r/distance_euclidean_squared.result b/mysql-test/suite/percona/r/distance_euclidean_squared.result new file mode 100644 index 000000000000..c94eb1775e4f --- /dev/null +++ b/mysql-test/suite/percona/r/distance_euclidean_squared.result @@ -0,0 +1,356 @@ +# +# Test coverage for vector DISTANCE() function. +# +# +# 0) Prepare playground. +# +CREATE TABLE t1 (id INT PRIMARY KEY, v1 VECTOR(1), v2 VECTOR(2)); +INSERT INTO t1 VALUES (0, TO_VECTOR('[0]'), TO_VECTOR('[0, 0]')), +(1, TO_VECTOR('[1]'), TO_VECTOR('[1, 0]')), +(2, TO_VECTOR('[1]'), TO_VECTOR('[0, 1]')), +(3, TO_VECTOR('[2]'), TO_VECTOR('[1, 1]')), +(4, TO_VECTOR('[2]'), TO_VECTOR('[2, 0]')), +(98, TO_VECTOR('[1]'), TO_VECTOR('[2]')), +(99, NULL, NULL); +CREATE TABLE t_metric_name (id INT PRIMARY KEY, name VARCHAR(10)); +INSERT INTO t_metric_name VALUES (1, "EUCLIDEAN"), (99, NULL); +# +# 1) Test how different number and types of arguments are handled. +# +# 1.1) Arity. +# +SELECT DISTANCE(); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]")); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]")); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED") +1 +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), TO_VECTOR("[3]"), "EUCLIDEAN"); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +# +# 1.2) Argument types. +# +# Only vectors or binary strings are allowed for first the two arguments. +SELECT DISTANCE("[1]", TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(X'0000803F', TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED"); +DISTANCE(X'0000803F', TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED") +1 +SELECT DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED") +4 +SELECT DISTANCE(v1, TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 0; +DISTANCE(v1, TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED") +4 +SELECT DISTANCE(id, TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 0; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1]"), "[2]", "EUCLIDEAN_SQUARED"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[2]"), X'0000803F', "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[2]"), X'0000803F', "EUCLIDEAN_SQUARED") +1 +SELECT DISTANCE(TO_VECTOR("[3]"), v1, "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 1; +DISTANCE(TO_VECTOR("[3]"), v1, "EUCLIDEAN_SQUARED") +4 +SELECT DISTANCE(TO_VECTOR("[0]"), id, "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +# The third argument must be a string literal with value from the +# fixed list of metric names. +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[-1, 0]"), 1); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 0]"), CONCAT("EUCLI","DEAN")); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2, 0]"), "euclidean"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2, 0]"), "euclidean") +1.4142135623730951 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[3, 0]"), "EuClIdEaN"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[3, 0]"), "EuClIdEaN") +2.23606797749979 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[4, 0]"), X'4555434C494445414E'); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[4, 0]"), X'4555434C494445414E') +3.1622776601683795 +# Metric strings with embedded NUL must be rejected regardless of prefix match. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), X'4555434C494445414E00'); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), X'4555434C494445414E004A554E4B'); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 0]"), "NOSUCHMETRIC"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[6, 0]"), name) FROM t_metric_name WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[7, 0]"), NULL); +ERROR HY000: Incorrect arguments to distance +# +# 1.3) NULL arguments and nullability in metadata for result. +# +SELECT DISTANCE(NULL, TO_VECTOR("[1, 0]"), "EUCLIDEAN_SQUARED"); +DISTANCE(NULL, TO_VECTOR("[1, 0]"), "EUCLIDEAN_SQUARED") +NULL +SELECT DISTANCE(TO_VECTOR("[1, 1]"), NULL, "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1, 1]"), NULL, "EUCLIDEAN_SQUARED") +NULL +SELECT DISTANCE(v2, TO_VECTOR("[1, 0]"), "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 99; +DISTANCE(v2, TO_VECTOR("[1, 0]"), "EUCLIDEAN_SQUARED") +NULL +SELECT DISTANCE(TO_VECTOR("[1, 1]"), v2, "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 99; +DISTANCE(TO_VECTOR("[1, 1]"), v2, "EUCLIDEAN_SQUARED") +NULL +# The third argument doesn't allow NULL values in any form. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), NULL); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), name) FROM t_metric_name WHERE id = 99; +ERROR HY000: Incorrect arguments to distance +# The result metadata should indicate that it is nullable. +CREATE TABLE tt SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "EUCLIDEAN_SQUARED") AS d; +SHOW CREATE TABLE tt; +Table Create Table +tt CREATE TABLE `tt` ( + `d` double DEFAULT NULL +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_0900_ai_ci +DROP TABLE tt; +# +# 2) Test vector arguments length mismatch. +# +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN_SQUARED"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(v2, TO_VECTOR("[1]"), "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(v1, v2, "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +# +# Note that length check happens at runtime. This is well visible +# when we have value stored in a vector field which is shorter than +# maximum length specified at the field creation time. +SELECT DISTANCE(v1, v2, "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 98; +DISTANCE(v1, v2, "EUCLIDEAN_SQUARED") +1 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), v2, "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 98; +ERROR HY000: Incorrect arguments to distance +# Binary-string BLOB arguments exceeding max_dimensions (16383) are rejected. +# A BLOB column is used so the argument passes the resolve-time binary-charset +# type check; the max_dimensions guard fires at runtime in val_real(). +CREATE TABLE t_oversized (v MEDIUMBLOB); +INSERT INTO t_oversized VALUES (REPEAT(X'00000000', 16384)); +SELECT DISTANCE(v, v, "EUCLIDEAN_SQUARED") FROM t_oversized; +ERROR HY000: Incorrect arguments to distance +DROP TABLE t_oversized; +# +# 3) Some basic tests for different (from syntax PoV) variants of +# arguments. +# +SELECT DISTANCE(X'0000803F0000803F', X'0000000000000040', "EUCLIDEAN_SQUARED"); +DISTANCE(X'0000803F0000803F', X'0000000000000040', "EUCLIDEAN_SQUARED") +2 +SELECT DISTANCE(X'0000803F0000803F', TO_VECTOR("[2, 0]"), "EUCLIDEAN_SQUARED"); +DISTANCE(X'0000803F0000803F', TO_VECTOR("[2, 0]"), "EUCLIDEAN_SQUARED") +2 +SELECT DISTANCE(X'0000803F0000803F', v2, "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 4; +DISTANCE(X'0000803F0000803F', v2, "EUCLIDEAN_SQUARED") +2 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN_SQUARED") +1 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), X'000000000000803F', "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[0, 0]"), X'000000000000803F', "EUCLIDEAN_SQUARED") +1 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), v2, "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 1; +DISTANCE(TO_VECTOR("[0, 0]"), v2, "EUCLIDEAN_SQUARED") +1 +SELECT DISTANCE(a.v2, b.v2, "EUCLIDEAN_SQUARED") FROM t1 AS a, t1 AS b WHERE a.id = 0 AND b.id = 4; +DISTANCE(a.v2, b.v2, "EUCLIDEAN_SQUARED") +4 +SELECT DISTANCE(v2, X'0000000000000040', "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 0; +DISTANCE(v2, X'0000000000000040', "EUCLIDEAN_SQUARED") +4 +SELECT DISTANCE(v2, TO_VECTOR("[0, 2]"), "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 0; +DISTANCE(v2, TO_VECTOR("[0, 2]"), "EUCLIDEAN_SQUARED") +4 +# Non-trivial (artificial) combinations +SELECT DISTANCE(TO_VECTOR(CONCAT("[0", ", ", "1]")), CONCAT(X'00000000', X'00000040'), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR(CONCAT("[0", ", ", "1]")), CONCAT(X'00000000', X'00000040'), "EUCLIDEAN_SQUARED") +1 +# The below case demonstrates that arguments to DISTANCE might not be +# well-aligned in memory. +SELECT DISTANCE(SUBSTR(X'010000000000000040', 2), RIGHT(X'40000000000000803F', 8), "EUCLIDEAN_SQUARED"); +DISTANCE(SUBSTR(X'010000000000000040', 2), RIGHT(X'40000000000000803F', 8), "EUCLIDEAN_SQUARED") +1 +# 9-byte blobs; SUBSTR from pos 2 → 8 bytes at offset 1 (misaligned for float). +# Length must stay a multiple of 4; SUBSTR(..., 4) on 9 bytes yields 6 → ER_TO_VECTOR_CONVERSION. +SELECT DISTANCE(SUBSTR(X'000100000000000040', 2), SUBSTR(X'00040000000000803F', 2), "EUCLIDEAN_SQUARED"); +DISTANCE(SUBSTR(X'000100000000000040', 2), SUBSTR(X'00040000000000803F', 2), "EUCLIDEAN_SQUARED") +1 +# +# 4) Basic test for different vector values. +# +# Identical / collinear vectors. +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 1]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 1]"), "EUCLIDEAN_SQUARED") +0 +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN_SQUARED") +0 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2.5, 2.5]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2.5, 2.5]"), "EUCLIDEAN_SQUARED") +4.5 +SELECT DISTANCE(TO_VECTOR("[1, 2, 3, 4, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1, 2, 3, 4, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "EUCLIDEAN_SQUARED") +0 +# Orthogonal vectors. +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "EUCLIDEAN_SQUARED") +2 +SELECT DISTANCE(TO_VECTOR("[0, 1, 0]"), TO_VECTOR("[-1, 0, -1]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[0, 1, 0]"), TO_VECTOR("[-1, 0, -1]"), "EUCLIDEAN_SQUARED") +3 +SELECT DISTANCE(TO_VECTOR("[1, 0, 3, 0, 5]"), TO_VECTOR("[0, 2, 0, 4, 0]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1, 0, 3, 0, 5]"), TO_VECTOR("[0, 2, 0, 4, 0]"), "EUCLIDEAN_SQUARED") +55 +# Anti-parallel vectors. +SELECT DISTANCE(TO_VECTOR("[-1, -1]"), TO_VECTOR("[2, 2]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[-1, -1]"), TO_VECTOR("[2, 2]"), "EUCLIDEAN_SQUARED") +18 +SELECT DISTANCE(TO_VECTOR("[-2e38, 1]"), TO_VECTOR("[2e38, -1]"), "EUCLIDEAN_SQUARED"); +ERROR 22003: DOUBLE value is out of range in 'distance(to_vector('[-2e38, 1]'),to_vector('[2e38, -1]'),'EUCLIDEAN_SQUARED')' +# Distance from origin. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "EUCLIDEAN_SQUARED") +1 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[3, 4]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[3, 4]"), "EUCLIDEAN_SQUARED") +25 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 12]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 12]"), "EUCLIDEAN_SQUARED") +169 +SELECT DISTANCE(TO_VECTOR("[0, 0, 0, 0]"), TO_VECTOR("[1, 1, 1, 1]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[0, 0, 0, 0]"), TO_VECTOR("[1, 1, 1, 1]"), "EUCLIDEAN_SQUARED") +4 +# Mixed-sign and larger vectors. +SELECT DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "EUCLIDEAN_SQUARED") +27 +SELECT DISTANCE(TO_VECTOR("[1, 7, 3, 16, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1, 7, 3, 16, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "EUCLIDEAN_SQUARED") +169 +# Zero vector (behavior differs per metric). +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2, 2]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2, 2]"), "EUCLIDEAN_SQUARED") +8 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[0, 0]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[0, 0]"), "EUCLIDEAN_SQUARED") +0 +SELECT DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[0]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[0]"), "EUCLIDEAN_SQUARED") +0 +# Large values near float32 max. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2e38, 0]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2e38, 0]"), "EUCLIDEAN_SQUARED") +3.999999744228558e76 +# Same value in a 16-dim vector: exercises the wide-tier SIMD overflow +# fallback (dims >= 16 dispatches to the wide kernel; squaring 2e38 in +# float32 overflows to +Inf, but the isfinite check falls back to scalar). +SELECT DISTANCE(TO_VECTOR("[2e38, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +TO_VECTOR("[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +"EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[2e38, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +TO_VECTOR("[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +"EUCLIDEAN_SQUARED") +3.999999744228558e76 +# Symmetry: DISTANCE(a, b) = DISTANCE(b, a). +SELECT DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "EUCLIDEAN_SQUARED") = +DISTANCE(TO_VECTOR("[4, 5, 6]"), TO_VECTOR("[1, 2, 3]"), "EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "EUCLIDEAN_SQUARED") = +DISTANCE(TO_VECTOR("[4, 5, 6]"), TO_VECTOR("[1, 2, 3]"), "EUCLIDEAN_SQUARED") +1 +# Special IEEE 754 float32 values: NaN, +Infinity, -Infinity. +# NaN/Inf input elements raise ER_DATA_OUT_OF_RANGE for all metrics (POW/EXP convention). +SELECT DISTANCE(X'0000C07F', X'00000000', "EUCLIDEAN_SQUARED"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x0000c07f,0x00000000,'EUCLIDEAN_SQUARED')' +SELECT DISTANCE(X'0000807F', X'00000000', "EUCLIDEAN_SQUARED"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x0000807f,0x00000000,'EUCLIDEAN_SQUARED')' +SELECT DISTANCE(X'000080FF', X'00000000', "EUCLIDEAN_SQUARED"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x000080ff,0x00000000,'EUCLIDEAN_SQUARED')' +# Wide-tier SIMD path coverage (dims >= 16 dispatches to the wide kernel). +# Integer-valued diffs keep float32 partial sums exact, so results are +# identical across Scalar / SSE4.2 / NEON / AVX2 / AVX-512 / SVE2. +# 16-dim: fills one AVX-512 register / two AVX2 / four SSE4.2 -- no scalar tail. +SELECT DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"EUCLIDEAN_SQUARED") +16 +# 20-dim: SSE4.2 5x4 (no tail); AVX2 2x8 + 4-elem scalar tail; +# AVX-512 1x16 + 4-elem scalar tail. +SELECT DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"EUCLIDEAN_SQUARED"); +DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"EUCLIDEAN_SQUARED") +20 +# +# prepared statements +# +PREPARE s1 FROM 'SELECT DISTANCE(v1, TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 4'; +EXECUTE s1; +DISTANCE(v1, TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED") +0 +EXECUTE s1; +DISTANCE(v1, TO_VECTOR("[2]"), "EUCLIDEAN_SQUARED") +0 +DEALLOCATE PREPARE s1; +PREPARE stmt FROM 'SELECT DISTANCE(v1, TO_VECTOR("[2]"), "?") FROM t1 WHERE id = 4'; +ERROR HY000: Incorrect arguments to distance +PREPARE stmt FROM 'SELECT DISTANCE(v1, TO_VECTOR(?), "EUCLIDEAN_SQUARED") FROM t1 WHERE id = 4'; +SET @met="[2]"; +EXECUTE stmt USING @met; +DISTANCE(v1, TO_VECTOR(?), "EUCLIDEAN_SQUARED") +0 +DEALLOCATE PREPARE stmt; +# +# 5) Distance in query contexts. +# +# ORDER BY distance: nearest-neighbour pattern. +SELECT id FROM t1 WHERE id IN (0,1,2,3,4) +ORDER BY DISTANCE(v2, TO_VECTOR('[1, 0]'), "EUCLIDEAN_SQUARED"), id; +id +1 +0 +3 +4 +2 +# ORDER BY distance DESC: farthest-neighbour pattern. +SELECT id FROM t1 WHERE id IN (0,1,2,3,4) +ORDER BY DISTANCE(v2, TO_VECTOR('[1, 0]'), "EUCLIDEAN_SQUARED") DESC, id; +id +2 +0 +3 +4 +1 +# WHERE: range query filtering by distance. +SELECT id FROM t1 +WHERE id IN (0,1,2,3,4) AND DISTANCE(v2, TO_VECTOR('[1, 0]'), "EUCLIDEAN_SQUARED") < 1.5 +ORDER BY id; +id +0 +1 +3 +4 +# Derived table with distance. +SELECT id FROM +(SELECT id, DISTANCE(v2, TO_VECTOR('[1, 0]'), "EUCLIDEAN_SQUARED") AS d +FROM t1 WHERE id IN (0,1,2,3,4)) AS sq +WHERE d IS NOT NULL ORDER BY d, id; +id +1 +0 +3 +4 +2 +DROP TABLE t_metric_name; +DROP TABLE t1; diff --git a/mysql-test/suite/percona/r/distance_manhattan.result b/mysql-test/suite/percona/r/distance_manhattan.result new file mode 100644 index 000000000000..e2e14d1bd124 --- /dev/null +++ b/mysql-test/suite/percona/r/distance_manhattan.result @@ -0,0 +1,357 @@ +# +# Test coverage for vector DISTANCE() function. +# +# +# 0) Prepare playground. +# +CREATE TABLE t1 (id INT PRIMARY KEY, v1 VECTOR(1), v2 VECTOR(2)); +INSERT INTO t1 VALUES (0, TO_VECTOR('[0]'), TO_VECTOR('[0, 0]')), +(1, TO_VECTOR('[1]'), TO_VECTOR('[1, 0]')), +(2, TO_VECTOR('[1]'), TO_VECTOR('[0, 1]')), +(3, TO_VECTOR('[2]'), TO_VECTOR('[1, 1]')), +(4, TO_VECTOR('[2]'), TO_VECTOR('[2, 0]')), +(98, TO_VECTOR('[1]'), TO_VECTOR('[2]')), +(99, NULL, NULL); +CREATE TABLE t_metric_name (id INT PRIMARY KEY, name VARCHAR(10)); +INSERT INTO t_metric_name VALUES (1, "EUCLIDEAN"), (99, NULL); +# +# 1) Test how different number and types of arguments are handled. +# +# 1.1) Arity. +# +SELECT DISTANCE(); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]")); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]")); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), "MANHATTAN") +1 +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[2]"), TO_VECTOR("[3]"), "EUCLIDEAN"); +ERROR 42000: Incorrect parameter count in the call to native function 'DISTANCE' +# +# 1.2) Argument types. +# +# Only vectors or binary strings are allowed for first the two arguments. +SELECT DISTANCE("[1]", TO_VECTOR("[2]"), "MANHATTAN"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(X'0000803F', TO_VECTOR("[2]"), "MANHATTAN"); +DISTANCE(X'0000803F', TO_VECTOR("[2]"), "MANHATTAN") +1 +SELECT DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[2]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[2]"), "MANHATTAN") +2 +SELECT DISTANCE(v1, TO_VECTOR("[2]"), "MANHATTAN") FROM t1 WHERE id = 0; +DISTANCE(v1, TO_VECTOR("[2]"), "MANHATTAN") +2 +SELECT DISTANCE(id, TO_VECTOR("[2]"), "MANHATTAN") FROM t1 WHERE id = 0; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1]"), "[2]", "MANHATTAN"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[2]"), X'0000803F', "MANHATTAN"); +DISTANCE(TO_VECTOR("[2]"), X'0000803F', "MANHATTAN") +1 +SELECT DISTANCE(TO_VECTOR("[3]"), v1, "MANHATTAN") FROM t1 WHERE id = 1; +DISTANCE(TO_VECTOR("[3]"), v1, "MANHATTAN") +2 +SELECT DISTANCE(TO_VECTOR("[0]"), id, "MANHATTAN") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +# The third argument must be a string literal with value from the +# fixed list of metric names. +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[-1, 0]"), 1); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 0]"), CONCAT("EUCLI","DEAN")); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2, 0]"), "euclidean"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2, 0]"), "euclidean") +1.4142135623730951 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[3, 0]"), "EuClIdEaN"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[3, 0]"), "EuClIdEaN") +2.23606797749979 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[4, 0]"), X'4555434C494445414E'); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[4, 0]"), X'4555434C494445414E') +3.1622776601683795 +# Metric strings with embedded NUL must be rejected regardless of prefix match. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), X'4555434C494445414E00'); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), X'4555434C494445414E004A554E4B'); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 0]"), "NOSUCHMETRIC"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[6, 0]"), name) FROM t_metric_name WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[7, 0]"), NULL); +ERROR HY000: Incorrect arguments to distance +# +# 1.3) NULL arguments and nullability in metadata for result. +# +SELECT DISTANCE(NULL, TO_VECTOR("[1, 0]"), "MANHATTAN"); +DISTANCE(NULL, TO_VECTOR("[1, 0]"), "MANHATTAN") +NULL +SELECT DISTANCE(TO_VECTOR("[1, 1]"), NULL, "MANHATTAN"); +DISTANCE(TO_VECTOR("[1, 1]"), NULL, "MANHATTAN") +NULL +SELECT DISTANCE(v2, TO_VECTOR("[1, 0]"), "MANHATTAN") FROM t1 WHERE id = 99; +DISTANCE(v2, TO_VECTOR("[1, 0]"), "MANHATTAN") +NULL +SELECT DISTANCE(TO_VECTOR("[1, 1]"), v2, "MANHATTAN") FROM t1 WHERE id = 99; +DISTANCE(TO_VECTOR("[1, 1]"), v2, "MANHATTAN") +NULL +# The third argument doesn't allow NULL values in any form. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), NULL); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), name) FROM t_metric_name WHERE id = 99; +ERROR HY000: Incorrect arguments to distance +# The result metadata should indicate that it is nullable. +CREATE TABLE tt SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "MANHATTAN") AS d; +SHOW CREATE TABLE tt; +Table Create Table +tt CREATE TABLE `tt` ( + `d` double DEFAULT NULL +) ENGINE=InnoDB DEFAULT CHARSET=utf8mb4 COLLATE=utf8mb4_0900_ai_ci +DROP TABLE tt; +# +# 2) Test vector arguments length mismatch. +# +SELECT DISTANCE(TO_VECTOR("[1]"), TO_VECTOR("[1, 0]"), "MANHATTAN"); +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(v2, TO_VECTOR("[1]"), "MANHATTAN") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +SELECT DISTANCE(v1, v2, "MANHATTAN") FROM t1 WHERE id = 1; +ERROR HY000: Incorrect arguments to distance +# +# Note that length check happens at runtime. This is well visible +# when we have value stored in a vector field which is shorter than +# maximum length specified at the field creation time. +SELECT DISTANCE(v1, v2, "MANHATTAN") FROM t1 WHERE id = 98; +DISTANCE(v1, v2, "MANHATTAN") +1 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), v2, "MANHATTAN") FROM t1 WHERE id = 98; +ERROR HY000: Incorrect arguments to distance +# Binary-string BLOB arguments exceeding max_dimensions (16383) are rejected. +# A BLOB column is used so the argument passes the resolve-time binary-charset +# type check; the max_dimensions guard fires at runtime in val_real(). +CREATE TABLE t_oversized (v MEDIUMBLOB); +INSERT INTO t_oversized VALUES (REPEAT(X'00000000', 16384)); +SELECT DISTANCE(v, v, "MANHATTAN") FROM t_oversized; +ERROR HY000: Incorrect arguments to distance +DROP TABLE t_oversized; +# +# 3) Some basic tests for different (from syntax PoV) variants of +# arguments. +# +SELECT DISTANCE(X'0000803F0000803F', X'0000000000000040', "MANHATTAN"); +DISTANCE(X'0000803F0000803F', X'0000000000000040', "MANHATTAN") +2 +SELECT DISTANCE(X'0000803F0000803F', TO_VECTOR("[2, 0]"), "MANHATTAN"); +DISTANCE(X'0000803F0000803F', TO_VECTOR("[2, 0]"), "MANHATTAN") +2 +SELECT DISTANCE(X'0000803F0000803F', v2, "MANHATTAN") FROM t1 WHERE id = 4; +DISTANCE(X'0000803F0000803F', v2, "MANHATTAN") +2 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "MANHATTAN") +1 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), X'000000000000803F', "MANHATTAN"); +DISTANCE(TO_VECTOR("[0, 0]"), X'000000000000803F', "MANHATTAN") +1 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), v2, "MANHATTAN") FROM t1 WHERE id = 1; +DISTANCE(TO_VECTOR("[0, 0]"), v2, "MANHATTAN") +1 +SELECT DISTANCE(a.v2, b.v2, "MANHATTAN") FROM t1 AS a, t1 AS b WHERE a.id = 0 AND b.id = 4; +DISTANCE(a.v2, b.v2, "MANHATTAN") +2 +SELECT DISTANCE(v2, X'0000000000000040', "MANHATTAN") FROM t1 WHERE id = 0; +DISTANCE(v2, X'0000000000000040', "MANHATTAN") +2 +SELECT DISTANCE(v2, TO_VECTOR("[0, 2]"), "MANHATTAN") FROM t1 WHERE id = 0; +DISTANCE(v2, TO_VECTOR("[0, 2]"), "MANHATTAN") +2 +# Non-trivial (artificial) combinations +SELECT DISTANCE(TO_VECTOR(CONCAT("[0", ", ", "1]")), CONCAT(X'00000000', X'00000040'), "MANHATTAN"); +DISTANCE(TO_VECTOR(CONCAT("[0", ", ", "1]")), CONCAT(X'00000000', X'00000040'), "MANHATTAN") +1 +# The below case demonstrates that arguments to DISTANCE might not be +# well-aligned in memory. +SELECT DISTANCE(SUBSTR(X'010000000000000040', 2), RIGHT(X'40000000000000803F', 8), "MANHATTAN"); +DISTANCE(SUBSTR(X'010000000000000040', 2), RIGHT(X'40000000000000803F', 8), "MANHATTAN") +1 +# 9-byte blobs; SUBSTR from pos 2 → 8 bytes at offset 1 (misaligned for float). +# Length must stay a multiple of 4; SUBSTR(..., 4) on 9 bytes yields 6 → ER_TO_VECTOR_CONVERSION. +SELECT DISTANCE(SUBSTR(X'000100000000000040', 2), SUBSTR(X'00040000000000803F', 2), "MANHATTAN"); +DISTANCE(SUBSTR(X'000100000000000040', 2), SUBSTR(X'00040000000000803F', 2), "MANHATTAN") +1 +# +# 4) Basic test for different vector values. +# +# Identical / collinear vectors. +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 1]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[1, 1]"), "MANHATTAN") +0 +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[1, 0]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[1, 0]"), "MANHATTAN") +0 +SELECT DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2.5, 2.5]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[1, 1]"), TO_VECTOR("[2.5, 2.5]"), "MANHATTAN") +3 +SELECT DISTANCE(TO_VECTOR("[1, 2, 3, 4, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[1, 2, 3, 4, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "MANHATTAN") +0 +# Orthogonal vectors. +SELECT DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[1, 0]"), TO_VECTOR("[0, 1]"), "MANHATTAN") +2 +SELECT DISTANCE(TO_VECTOR("[0, 1, 0]"), TO_VECTOR("[-1, 0, -1]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[0, 1, 0]"), TO_VECTOR("[-1, 0, -1]"), "MANHATTAN") +3 +SELECT DISTANCE(TO_VECTOR("[1, 0, 3, 0, 5]"), TO_VECTOR("[0, 2, 0, 4, 0]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[1, 0, 3, 0, 5]"), TO_VECTOR("[0, 2, 0, 4, 0]"), "MANHATTAN") +15 +# Anti-parallel vectors. +SELECT DISTANCE(TO_VECTOR("[-1, -1]"), TO_VECTOR("[2, 2]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[-1, -1]"), TO_VECTOR("[2, 2]"), "MANHATTAN") +6 +SELECT DISTANCE(TO_VECTOR("[-2e38, 1]"), TO_VECTOR("[2e38, -1]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[-2e38, 1]"), TO_VECTOR("[2e38, -1]"), "MANHATTAN") +3.999999872114277e38 +# Distance from origin. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[1, 0]"), "MANHATTAN") +1 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[3, 4]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[3, 4]"), "MANHATTAN") +7 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 12]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[5, 12]"), "MANHATTAN") +17 +SELECT DISTANCE(TO_VECTOR("[0, 0, 0, 0]"), TO_VECTOR("[1, 1, 1, 1]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[0, 0, 0, 0]"), TO_VECTOR("[1, 1, 1, 1]"), "MANHATTAN") +4 +# Mixed-sign and larger vectors. +SELECT DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "MANHATTAN") +9 +SELECT DISTANCE(TO_VECTOR("[1, 7, 3, 16, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[1, 7, 3, 16, 5]"), TO_VECTOR("[1, 2, 3, 4, 5]"), "MANHATTAN") +17 +# Zero vector (behavior differs per metric). +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2, 2]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2, 2]"), "MANHATTAN") +4 +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[0, 0]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[0, 0]"), "MANHATTAN") +0 +SELECT DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[0]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[0]"), TO_VECTOR("[0]"), "MANHATTAN") +0 +# Large values near float32 max. +SELECT DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2e38, 0]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[0, 0]"), TO_VECTOR("[2e38, 0]"), "MANHATTAN") +1.9999999360571385e38 +# Same value in a 16-dim vector: exercises the wide-tier SIMD overflow +# fallback (dims >= 16 dispatches to the wide kernel; squaring 2e38 in +# float32 overflows to +Inf, but the isfinite check falls back to scalar). +SELECT DISTANCE(TO_VECTOR("[2e38, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +TO_VECTOR("[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +"MANHATTAN"); +DISTANCE(TO_VECTOR("[2e38, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +TO_VECTOR("[0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]"), +"MANHATTAN") +1.9999999360571385e38 +# Symmetry: DISTANCE(a, b) = DISTANCE(b, a). +SELECT DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "MANHATTAN") = +DISTANCE(TO_VECTOR("[4, 5, 6]"), TO_VECTOR("[1, 2, 3]"), "MANHATTAN"); +DISTANCE(TO_VECTOR("[1, 2, 3]"), TO_VECTOR("[4, 5, 6]"), "MANHATTAN") = +DISTANCE(TO_VECTOR("[4, 5, 6]"), TO_VECTOR("[1, 2, 3]"), "MANHATTAN") +1 +# Special IEEE 754 float32 values: NaN, +Infinity, -Infinity. +# NaN/Inf input elements raise ER_DATA_OUT_OF_RANGE for all metrics (POW/EXP convention). +SELECT DISTANCE(X'0000C07F', X'00000000', "MANHATTAN"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x0000c07f,0x00000000,'MANHATTAN')' +SELECT DISTANCE(X'0000807F', X'00000000', "MANHATTAN"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x0000807f,0x00000000,'MANHATTAN')' +SELECT DISTANCE(X'000080FF', X'00000000', "MANHATTAN"); +ERROR 22003: DOUBLE value is out of range in 'distance(0x000080ff,0x00000000,'MANHATTAN')' +# Wide-tier SIMD path coverage (dims >= 16 dispatches to the wide kernel). +# Integer-valued diffs keep float32 partial sums exact, so results are +# identical across Scalar / SSE4.2 / NEON / AVX2 / AVX-512 / SVE2. +# 16-dim: fills one AVX-512 register / two AVX2 / four SSE4.2 -- no scalar tail. +SELECT DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"MANHATTAN"); +DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"MANHATTAN") +16 +# 20-dim: SSE4.2 5x4 (no tail); AVX2 2x8 + 4-elem scalar tail; +# AVX-512 1x16 + 4-elem scalar tail. +SELECT DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"MANHATTAN"); +DISTANCE(TO_VECTOR("[1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0]"), +TO_VECTOR("[0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1,0,1]"), +"MANHATTAN") +20 +# +# prepared statements +# +PREPARE s1 FROM 'SELECT DISTANCE(v1, TO_VECTOR("[2]"), "MANHATTAN") FROM t1 WHERE id = 4'; +EXECUTE s1; +DISTANCE(v1, TO_VECTOR("[2]"), "MANHATTAN") +0 +EXECUTE s1; +DISTANCE(v1, TO_VECTOR("[2]"), "MANHATTAN") +0 +DEALLOCATE PREPARE s1; +PREPARE stmt FROM 'SELECT DISTANCE(v1, TO_VECTOR("[2]"), "?") FROM t1 WHERE id = 4'; +ERROR HY000: Incorrect arguments to distance +PREPARE stmt FROM 'SELECT DISTANCE(v1, TO_VECTOR(?), "MANHATTAN") FROM t1 WHERE id = 4'; +SET @met="[2]"; +EXECUTE stmt USING @met; +DISTANCE(v1, TO_VECTOR(?), "MANHATTAN") +0 +DEALLOCATE PREPARE stmt; +# +# 5) Distance in query contexts. +# +# ORDER BY distance: nearest-neighbour pattern. +SELECT id FROM t1 WHERE id IN (0,1,2,3,4) +ORDER BY DISTANCE(v2, TO_VECTOR('[1, 0]'), "MANHATTAN"), id; +id +1 +0 +3 +4 +2 +# ORDER BY distance DESC: farthest-neighbour pattern. +SELECT id FROM t1 WHERE id IN (0,1,2,3,4) +ORDER BY DISTANCE(v2, TO_VECTOR('[1, 0]'), "MANHATTAN") DESC, id; +id +2 +0 +3 +4 +1 +# WHERE: range query filtering by distance. +SELECT id FROM t1 +WHERE id IN (0,1,2,3,4) AND DISTANCE(v2, TO_VECTOR('[1, 0]'), "MANHATTAN") < 1.5 +ORDER BY id; +id +0 +1 +3 +4 +# Derived table with distance. +SELECT id FROM +(SELECT id, DISTANCE(v2, TO_VECTOR('[1, 0]'), "MANHATTAN") AS d +FROM t1 WHERE id IN (0,1,2,3,4)) AS sq +WHERE d IS NOT NULL ORDER BY d, id; +id +1 +0 +3 +4 +2 +DROP TABLE t_metric_name; +DROP TABLE t1; diff --git a/mysql-test/suite/percona/t/distance_cosine.test b/mysql-test/suite/percona/t/distance_cosine.test new file mode 100644 index 000000000000..3875a94aacfe --- /dev/null +++ b/mysql-test/suite/percona/t/distance_cosine.test @@ -0,0 +1,2 @@ +let $metric = COSINE; +--source ../include/distance.inc diff --git a/mysql-test/suite/percona/t/distance_dot.test b/mysql-test/suite/percona/t/distance_dot.test new file mode 100644 index 000000000000..bf77176908f5 --- /dev/null +++ b/mysql-test/suite/percona/t/distance_dot.test @@ -0,0 +1,2 @@ +let $metric = DOT; +--source ../include/distance.inc diff --git a/mysql-test/suite/percona/t/distance_euclidean.test b/mysql-test/suite/percona/t/distance_euclidean.test new file mode 100644 index 000000000000..ecb97385dcb8 --- /dev/null +++ b/mysql-test/suite/percona/t/distance_euclidean.test @@ -0,0 +1,2 @@ +let $metric = EUCLIDEAN; +--source ../include/distance.inc diff --git a/mysql-test/suite/percona/t/distance_euclidean_squared.test b/mysql-test/suite/percona/t/distance_euclidean_squared.test new file mode 100644 index 000000000000..a5b0f6552c1e --- /dev/null +++ b/mysql-test/suite/percona/t/distance_euclidean_squared.test @@ -0,0 +1,2 @@ +let $metric = EUCLIDEAN_SQUARED; +--source ../include/distance.inc diff --git a/mysql-test/suite/percona/t/distance_manhattan.test b/mysql-test/suite/percona/t/distance_manhattan.test new file mode 100644 index 000000000000..76edbf2a6b43 --- /dev/null +++ b/mysql-test/suite/percona/t/distance_manhattan.test @@ -0,0 +1,2 @@ +let $metric = MANHATTAN; +--source ../include/distance.inc diff --git a/mysql-test/suite/perfschema/r/error_log.result b/mysql-test/suite/perfschema/r/error_log.result index 5e511bf7baed..8084f8b39e92 100644 --- a/mysql-test/suite/perfschema/r/error_log.result +++ b/mysql-test/suite/perfschema/r/error_log.result @@ -94,7 +94,7 @@ AND error_code NOT IN("MY-010097", "MY-015641", "MY-013884", "MY-013908", "MY-013930", "MY-014013", "MY-000068", "MY-014066", "MY-014067", "MY-014068", "MY-013932", "MY-010101", "MY-010099", "MY-011070", -"MY-011872", "MY-011874", "MY-015140") +"MY-011872", "MY-011874", "MY-015140", "MY-048400") ORDER BY logged, error_code; prio error_code subsystem System MY-015015 Server diff --git a/mysql-test/suite/perfschema/t/error_log.test b/mysql-test/suite/perfschema/t/error_log.test index 2335be74c21d..43a76ecdfdfb 100644 --- a/mysql-test/suite/perfschema/t/error_log.test +++ b/mysql-test/suite/perfschema/t/error_log.test @@ -144,6 +144,7 @@ SELECT prio,error_code,subsystem,data # Percona-added: # MY-010097: ER_SEC_FILE_PRIV_EMPTY, emitted for --secure-log-path # MY-015641: ER_SERVER_WARN_CONTAINER_IGNORED, depends on whether server runs in a container +# MY-048400: ER_VECTOR_DISTANCE_SIMD_DISPATCH, emits to inform about the SIMD/scalar used to calculate vector_distance SELECT prio,error_code,subsystem FROM performance_schema.error_log WHERE logged>=@startup @@ -159,7 +160,7 @@ SELECT prio,error_code,subsystem "MY-013884", "MY-013908", "MY-013930", "MY-014013", "MY-000068", "MY-014066", "MY-014067", "MY-014068", "MY-013932", "MY-010101", "MY-010099", "MY-011070", - "MY-011872", "MY-011874", "MY-015140") + "MY-011872", "MY-011874", "MY-015140", "MY-048400") ORDER BY logged, error_code; --echo diff --git a/share/messages_to_error_log.txt b/share/messages_to_error_log.txt index 931b2d731ae3..ed3102c3a750 100644 --- a/share/messages_to_error_log.txt +++ b/share/messages_to_error_log.txt @@ -14049,6 +14049,9 @@ start-error-number 48300 # start-error-number 48400 +ER_VECTOR_DISTANCE_SIMD_DISPATCH + eng "For VECTOR_DISTANCE using %s" + # # End of Percona Server 9.7 error log messages # diff --git a/sql/CMakeLists.txt b/sql/CMakeLists.txt index b1fe25b17847..5d741cda1d5b 100644 --- a/sql/CMakeLists.txt +++ b/sql/CMakeLists.txt @@ -368,6 +368,7 @@ SET(SQL_SHARED_SOURCES auth/sha2_password_common.cc auth/sha2_password.cc ../vector-common/vector_conversion.cc + ../vector-common/vector_distance.cc ssl_wrapper_service.cc bootstrap.cc check_stack.cc diff --git a/sql/item_create.cc b/sql/item_create.cc index 02a5cd6f6cb9..2c04b5ef2584 100644 --- a/sql/item_create.cc +++ b/sql/item_create.cc @@ -1400,6 +1400,7 @@ static const std::pair func_array[] = { {"DAYOFWEEK", SQL_FACTORY(Dayofweek_instantiator)}, {"DAYOFYEAR", SQL_FN(Item_func_dayofyear, 1)}, {"DEGREES", SQL_FN(Item_func_degrees, 1)}, + {"DISTANCE", SQL_FN(Item_func_vector_distance, 3)}, {"ELT", SQL_FN_V(Item_func_elt, 2, MAX_ARGLIST_SIZE)}, {"ETAG", SQL_FN_V(Item_func_etag, 1, MAX_ARGLIST_SIZE)}, {"EXP", SQL_FN(Item_func_exp, 1)}, @@ -1657,6 +1658,7 @@ static const std::pair func_array[] = { {"FROM_VECTOR", SQL_FN(Item_func_from_vector, 1)}, {"VECTOR_TO_STRING", SQL_FN(Item_func_from_vector, 1)}, {"VECTOR_DIM", SQL_FN(Item_func_vector_dim, 1)}, + {"VECTOR_DISTANCE", SQL_FN(Item_func_vector_distance, 3)}, {"UCASE", SQL_FN(Item_func_upper, 1)}, {"UNCOMPRESS", SQL_FN(Item_func_uncompress, 1)}, {"UNCOMPRESSED_LENGTH", SQL_FN(Item_func_uncompressed_length, 1)}, diff --git a/sql/item_func.h b/sql/item_func.h index e36f51d8ba8d..d0ef3ed1b547 100644 --- a/sql/item_func.h +++ b/sql/item_func.h @@ -364,6 +364,7 @@ class Item_func : public Item_result_field { ETAG_FUNC, CURRENT_USER_IN_FUNC, CURRENT_ROLE_IN_FUNC, + VECTOR_DISTANCE_FUNC }; enum optimize_type { OPTIMIZE_NONE, @@ -858,6 +859,11 @@ class Item_real_func : public Item_func { set_data_type_double(); } + Item_real_func(const POS &pos, Item *a, Item *b, Item *c) + : Item_func(pos, a, b, c) { + set_data_type_double(); + } + explicit Item_real_func(mem_root_deque *list) : Item_func(list) { set_data_type_double(); } diff --git a/sql/item_strfunc.cc b/sql/item_strfunc.cc index f819473493bf..1cb6dd03c6f1 100644 --- a/sql/item_strfunc.cc +++ b/sql/item_strfunc.cc @@ -37,7 +37,7 @@ #include #include #include -#include // std::isfinite +#include // std::isfinite, std::isnan #include // size_t #include #include @@ -135,6 +135,7 @@ #include "typelib.h" #include "unhex.h" #include "vector-common/vector_conversion.h" // from_string_to_vector, from_vector_to_string +#include "vector-common/vector_distance.h" // vector_distance_euclidean_squared, vector_distance_cosine, vector_distance_dot extern uint *my_aes_opmode_key_sizes; @@ -4270,6 +4271,149 @@ String *Item_func_from_vector::val_str_ascii(String *str) { return &buffer; } +bool Item_func_vector_distance::do_itemize(Parse_context *pc, Item **res) { + if (skip_itemize(res)) return false; + if (Item_real_func::do_itemize(pc, res)) return true; + // Unsafe for statement-based replication: results depend on the SIMD tier + // dispatched at runtime (AVX-512/AVX2/SSE4.2/NEON/scalar), which can differ + // between source and replica hardware and yield slightly different + // floating-point results. + pc->thd->lex->set_stmt_unsafe(LEX::BINLOG_STMT_UNSAFE_SYSTEM_FUNCTION); + return false; +} + +bool Item_func_vector_distance::resolve_type(THD *thd) { + if (param_type_is_default(thd, 0, 2, MYSQL_TYPE_VECTOR)) { + return true; + } + + for (uint i = 0; i < 2; ++i) { + if (!(args[i]->data_type() == MYSQL_TYPE_VECTOR || + (args[i]->result_type() == STRING_RESULT && + args[i]->collation.collation == &my_charset_bin))) { + my_error(ER_WRONG_ARGUMENTS, MYF(0), func_name()); + return true; + } + } + + // Let us prohibit non-literal metric names right away, to make + // optimizer life easier. This is not something going to happen + // in practice anyway. + if (!args[2]->basic_const_item()) { + my_error(ER_WRONG_ARGUMENTS, MYF(0), func_name()); + return true; + } + + String tmp, *metric_n = args[2]->val_str_ascii(&tmp); + + if (metric_n == nullptr) { + my_error(ER_WRONG_ARGUMENTS, MYF(0), func_name()); + return true; + } + + static constexpr struct { + std::string_view name; + metric_type metric; + } kMetrics[] = { + {"euclidean", EUCLIDEAN}, {"euclidean_squared", EUCLIDEAN_SQUARED}, + {"cosine", COSINE}, {"dot", DOT_PRODUCT}, + {"manhattan", MANHATTAN}, + }; + + // my_strnncoll is length-aware: embedded NUL + trailing bytes cannot match. + const auto *it = std::find_if( + std::begin(kMetrics), std::end(kMetrics), [&](const auto &candidate) { + return !my_strnncoll(&my_charset_latin1, + pointer_cast(metric_n->ptr()), + metric_n->length(), + pointer_cast(candidate.name.data()), + candidate.name.size()); + }); + if (it == std::end(kMetrics)) { + my_error(ER_WRONG_ARGUMENTS, MYF(0), func_name()); + return true; + } + m_metric = it->metric; + + // Cosine can return NULL for zero-length vectors at runtime. Mark nullable + // unconditionally so that derived columns and metadata (SHOW CREATE TABLE) + // reflect the true nullability of the function regardless of whether the + // input arguments are themselves nullable. + set_nullable(true); + + return false; +} + +double Item_func_vector_distance::val_real() { + assert(fixed); + null_value = false; + + String buff_a, buff_b; + String *a = args[0]->val_str(&buff_a); + if (a == nullptr || a->ptr() == nullptr) { + return error_real(); + } + + uint32 a_dims = get_dimensions(a->length(), Field_vector::precision); + if (a_dims == UINT32_MAX) { + my_error(ER_TO_VECTOR_CONVERSION, MYF(0), a->length(), a->ptr()); + return error_real(); + } + if (a_dims > Field_vector::max_dimensions) { + my_error(ER_WRONG_ARGUMENTS, MYF(0), func_name()); + return error_real(); + } + + String *b = args[1]->val_str(&buff_b); + if (b == nullptr || b->ptr() == nullptr) { + return error_real(); + } + + uint32 b_dims = get_dimensions(b->length(), Field_vector::precision); + if (b_dims == UINT32_MAX) { + my_error(ER_TO_VECTOR_CONVERSION, MYF(0), b->length(), b->ptr()); + return error_real(); + } + if (b_dims > Field_vector::max_dimensions) { + my_error(ER_WRONG_ARGUMENTS, MYF(0), func_name()); + return error_real(); + } + + if (a_dims != b_dims) { + my_error(ER_WRONG_ARGUMENTS, MYF(0), func_name()); + return error_real(); + } + + switch (m_metric) { + case EUCLIDEAN: + return check_float_overflow(std::sqrt( + vector_distance_euclidean_squared(a->ptr(), b->ptr(), a_dims))); + case EUCLIDEAN_SQUARED: + return check_float_overflow( + vector_distance_euclidean_squared(a->ptr(), b->ptr(), a_dims)); + case COSINE: { + const double dist = vector_distance_cosine(a->ptr(), b->ptr(), a_dims); + if (std::isinf(dist)) { + // +Inf sentinel from vector_distance_cosine: zero-vector(s) → undefined + // cosine → NULL + null_value = true; + return 0.0; + } + // NaN/Inf input elements propagated through → ER_DATA_OUT_OF_RANGE + return check_float_overflow(dist); + } + case DOT_PRODUCT: + return check_float_overflow( + -vector_distance_dot(a->ptr(), b->ptr(), a_dims)); + case MANHATTAN: + return check_float_overflow( + vector_distance_manhattan(a->ptr(), b->ptr(), a_dims)); + default: + assert(false); + return 0.0; + } +} + String *Item_func_uncompress::val_str(String *str) { assert(fixed); String *res = args[0]->val_str(str); diff --git a/sql/item_strfunc.h b/sql/item_strfunc.h index 14e6334d4554..51babbfc9e6c 100644 --- a/sql/item_strfunc.h +++ b/sql/item_strfunc.h @@ -1310,6 +1310,26 @@ class Item_func_from_vector final : public Item_str_ascii_func { String *val_str_ascii(String *str) override; }; +class Item_func_vector_distance final : public Item_real_func { + enum metric_type { + EUCLIDEAN, + EUCLIDEAN_SQUARED, + COSINE, + DOT_PRODUCT, + MANHATTAN + }; + metric_type m_metric{EUCLIDEAN}; + + public: + Item_func_vector_distance(const POS &pos, Item *a, Item *b, Item *c) + : Item_real_func(pos, a, b, c) {} + bool do_itemize(Parse_context *pc, Item **res) override; + bool resolve_type(THD *thd) override; + const char *func_name() const override { return "distance"; } + enum Functype functype() const override { return VECTOR_DISTANCE_FUNC; } + double val_real() override; +}; + class Item_func_uncompress final : public Item_str_func { String buffer; diff --git a/sql/mysqld.cc b/sql/mysqld.cc index 5ca85725e42d..2fdc058d846b 100644 --- a/sql/mysqld.cc +++ b/sql/mysqld.cc @@ -927,6 +927,7 @@ MySQL clients support the protocol: #include "thr_lock.h" #include "thr_mutex.h" #include "typelib.h" +#include "vector-common/vector_distance.h" // init_vector_distance_functions #include "violite.h" #ifdef WITH_PERFSCHEMA_STORAGE_ENGINE @@ -8427,6 +8428,7 @@ static int init_server_components() { We need to call each of these following functions to ensure that all things are initialized so that unireg_abort() doesn't fail */ + init_vector_distance_functions(); mdl_init(); partitioning_init(); if (table_def_init() || hostname_cache_init(host_cache_size)) @@ -8505,6 +8507,14 @@ static int init_server_components() { */ if (setup_error_log_components()) unireg_abort(MYSQLD_ABORT_EXIT); + if (!is_help_or_validate_option()) { + char vector_distance_msg[256]; + vector_distance_dispatch_description(vector_distance_msg, + sizeof(vector_distance_msg)); + LogErr(INFORMATION_LEVEL, ER_VECTOR_DISTANCE_SIMD_DISPATCH, + vector_distance_msg); + } + if (MDL_context_backup_manager::init()) { LogErr(ERROR_LEVEL, ER_OOM); unireg_abort(MYSQLD_ABORT_EXIT); diff --git a/unittest/gunit/CMakeLists.txt b/unittest/gunit/CMakeLists.txt index d86102bc19d4..426fbbbbbd3d 100644 --- a/unittest/gunit/CMakeLists.txt +++ b/unittest/gunit/CMakeLists.txt @@ -182,6 +182,7 @@ SET(TESTS unhex utf8alias val_int_compare + vector_distance ) LIST(TRANSFORM TESTS APPEND "-t.cc" OUTPUT_VARIABLE ALL_SMALL_TESTS) @@ -365,6 +366,7 @@ DISABLE_MISSING_PROFILE_WARNING() LIST(TRANSFORM SERVER_TESTS APPEND "-t.cc" OUTPUT_VARIABLE ALL_LARGE_TESTS) SET(SQL_GUNIT_LIB_SOURCE + ${CMAKE_SOURCE_DIR}/vector-common/vector_distance.cc ${CMAKE_SOURCE_DIR}/sql/filesort_utils.cc ${CMAKE_SOURCE_DIR}/sql/mdl.cc ${CMAKE_SOURCE_DIR}/sql/sql_list.cc @@ -422,6 +424,16 @@ FOREACH(test ${TESTS}) ) ENDFOREACH() +# vector_distance per-tier benchmark — all SIMD tiers × 6 sizes × euclidean/cosine/dot_product. +# Unavailable tiers (wrong ISA or CPU) are skipped at runtime via GTEST_SKIP(). +MYSQL_ADD_EXECUTABLE(vector_distance_benchmark-t vector_distance_benchmark-t.cc + COMPILE_DEFINITIONS ${DISABLE_PSI_DEFINITIONS} + ENABLE_EXPORTS + EXCLUDE_FROM_ALL + LINK_LIBRARIES sqlgunitlib gunit_small extra::boost + SKIP_INSTALL +) + # Disable by default, since it dumps a stack trace. # We don't want ppl to think there was a segfault or something. # See also the mtr test main.print_stacktrace diff --git a/unittest/gunit/vector_distance-t.cc b/unittest/gunit/vector_distance-t.cc new file mode 100644 index 000000000000..da558d55c006 --- /dev/null +++ b/unittest/gunit/vector_distance-t.cc @@ -0,0 +1,634 @@ +/* Copyright (c) 2026, Percona and/or its affiliates. + + This program is free software; you can redistribute it and/or modify + it under the terms of the GNU General Public License, version 2.0, + as published by the Free Software Foundation. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License, version 2.0, for more details. + + You should have received a copy of the GNU General Public License + along with this program; if not, write to the Free Software + Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA */ + +#include + +#include +#include +#include +#include +#include +#include +#include +#include + +#include "my_pointer_arithmetic.h" +#include "vector-common/vector_distance.h" + +namespace vector_distance_unittest { + +// --------------------------------------------------------------------------- +// Reference implementations — always scalar, double precision. +// Used to verify SIMD paths; independent of vector_distance.cc internals. +// --------------------------------------------------------------------------- + +static double ref_euclidean_squared(const float *a, const float *b, + uint32_t n) { + double sum = 0.0; + for (uint32_t i = 0; i < n; i++) { + const double d = a[i] - b[i]; + sum += d * d; + } + return sum; +} + +static double ref_euclidean(const float *a, const float *b, uint32_t n) { + return std::sqrt(ref_euclidean_squared(a, b, n)); +} + +// Smallest positive offset from base that is not alignof(float)-aligned. +// A fixed +1 byte offset is not reliable: when the buffer sits at address +// % alignof(float) == alignof(float) - 1 (seen on Apple Silicon stacks), +// base+1 is still float-aligned. +static size_t misaligned_float_offset(uintptr_t base) { + for (size_t offset = 1; offset < alignof(float); ++offset) { + if ((base + offset) % alignof(float) != 0) return offset; + } + assert(false); + return 1; +} + +// Mirrors Item_func_vector_distance EUCLIDEAN branch. +static double euclidean_l2(const char *a, const char *b, uint32_t dims) { + return std::sqrt(vector_distance_euclidean_squared(a, b, dims)); +} + +static double ref_dot_product(const float *a, const float *b, uint32_t n) { + double ab = 0.0; + for (uint32_t i = 0; i < n; i++) ab += (double)a[i] * b[i]; + return ab; +} + +static double ref_cosine(const float *a, const float *b, uint32_t n) { + double ab = 0.0, na = 0.0, nb = 0.0; + for (uint32_t i = 0; i < n; i++) { + ab += a[i] * b[i]; + na += a[i] * a[i]; + nb += b[i] * b[i]; + } + const double denom = std::sqrt(na * nb); + if (denom == 0.0) return std::numeric_limits::infinity(); + return 1.0 - ab / denom; +} + +static double ref_manhattan(const float *a, const float *b, uint32_t n) { + double result = 0.0; + for (uint32_t i = 0; i < n; i++) result += std::fabs((double)a[i] - b[i]); + return result; +} + +// --------------------------------------------------------------------------- +// Fixture — initialises dispatch pointers once per test suite +// --------------------------------------------------------------------------- + +class VectorDistanceTest : public ::testing::Test { + protected: + static void SetUpTestSuite() { init_vector_distance_functions(); } +}; + +// --------------------------------------------------------------------------- +// Known-value correctness +// --------------------------------------------------------------------------- + +TEST_F(VectorDistanceTest, EuclideanSquaredKnownValues) { + // [0,0] → [3,4] = 25.0 (squared 3-4-5 right triangle) + alignas(32) float a[] = {0.0f, 0.0f}; + alignas(32) float b[] = {3.0f, 4.0f}; + EXPECT_NEAR( + vector_distance_euclidean_squared((const char *)a, (const char *)b, 2), + 25.0, 1e-6); + + // Identical vectors → distance 0 + alignas(32) float c[] = {1.0f, 2.0f, 3.0f}; + EXPECT_NEAR( + vector_distance_euclidean_squared((const char *)c, (const char *)c, 3), + 0.0, 1e-9); +} + +TEST_F(VectorDistanceTest, EuclideanKnownValues) { + // [0,0] → [3,4] = 5.0 (3-4-5 right triangle) + alignas(32) float a[] = {0.0f, 0.0f}; + alignas(32) float b[] = {3.0f, 4.0f}; + EXPECT_NEAR(euclidean_l2((const char *)a, (const char *)b, 2), 5.0, 1e-6); + + // Identical vectors → distance 0 + alignas(32) float c[] = {1.0f, 2.0f, 3.0f}; + EXPECT_NEAR(euclidean_l2((const char *)c, (const char *)c, 3), 0.0, 1e-9); +} + +TEST_F(VectorDistanceTest, CosineKnownValues) { + // Identical unit vectors → distance 0 + alignas(32) float same[] = {1.0f, 0.0f, 0.0f}; + EXPECT_NEAR(vector_distance_cosine((const char *)same, (const char *)same, 3), + 0.0, 1e-6); + + // Orthogonal vectors → distance 1 + alignas(32) float x[] = {1.0f, 0.0f}; + alignas(32) float y[] = {0.0f, 1.0f}; + EXPECT_NEAR(vector_distance_cosine((const char *)x, (const char *)y, 2), 1.0, + 1e-6); + + // Anti-parallel → distance 2 + alignas(32) float pos[] = {1.0f, 1.0f}; + alignas(32) float neg[] = {-1.0f, -1.0f}; + EXPECT_NEAR(vector_distance_cosine((const char *)pos, (const char *)neg, 2), + 2.0, 1e-6); +} + +TEST_F(VectorDistanceTest, DotProductKnownValues) { + // Orthogonal vectors → dot product 0 + alignas(32) float x[] = {1.0f, 0.0f}; + alignas(32) float y[] = {0.0f, 1.0f}; + EXPECT_NEAR(vector_distance_dot((const char *)x, (const char *)y, 2), 0.0, + 1e-9); + + // Identical unit vector → dot product 1 + alignas(32) float u[] = {1.0f, 0.0f, 0.0f}; + EXPECT_NEAR(vector_distance_dot((const char *)u, (const char *)u, 3), 1.0, + 1e-9); + + // Known values: [1,2,3]·[4,5,6] = 4+10+18 = 32 + alignas(32) float a[] = {1.0f, 2.0f, 3.0f}; + alignas(32) float b[] = {4.0f, 5.0f, 6.0f}; + EXPECT_NEAR(vector_distance_dot((const char *)a, (const char *)b, 3), 32.0, + 1e-6); +} + +TEST_F(VectorDistanceTest, ManhattanKnownValues) { + // Identical vectors → distance 0 + alignas(32) float same[] = {1.0f, 2.0f, 3.0f}; + EXPECT_NEAR( + vector_distance_manhattan((const char *)same, (const char *)same, 3), 0.0, + 1e-9); + + // [0,0] → [3,4] = |3| + |4| = 7 (compare: Euclidean gives 5) + alignas(32) float a[] = {0.0f, 0.0f}; + alignas(32) float b[] = {3.0f, 4.0f}; + EXPECT_NEAR(vector_distance_manhattan((const char *)a, (const char *)b, 2), + 7.0, 1e-6); + + // [1,7,3,16,5] → [1,2,3,4,5] = 0+5+0+12+0 = 17 + alignas(32) float c[] = {1.0f, 7.0f, 3.0f, 16.0f, 5.0f}; + alignas(32) float d[] = {1.0f, 2.0f, 3.0f, 4.0f, 5.0f}; + EXPECT_NEAR(vector_distance_manhattan((const char *)c, (const char *)d, 5), + 17.0, 1e-6); +} + +// --------------------------------------------------------------------------- +// Zero-vector guard: cosine must return +Inf sentinel (val_real maps to NULL) +// --------------------------------------------------------------------------- + +TEST_F(VectorDistanceTest, CosineZeroVectorReturnsInf) { + alignas(32) float z[] = {0.0f, 0.0f}; + alignas(32) float a[] = {1.0f, 2.0f}; + EXPECT_TRUE( + std::isinf(vector_distance_cosine((const char *)z, (const char *)a, 2))); + EXPECT_TRUE( + std::isinf(vector_distance_cosine((const char *)a, (const char *)z, 2))); + EXPECT_TRUE( + std::isinf(vector_distance_cosine((const char *)z, (const char *)z, 2))); +} + +// --------------------------------------------------------------------------- +// Unaligned path: misaligned buffer must give the same result as aligned +// --------------------------------------------------------------------------- + +static void CheckUnalignedMatchesAligned(uint32_t dims) { + SCOPED_TRACE(::testing::Message() << "dims=" << dims); + + std::mt19937 rng(dims); + std::uniform_real_distribution dist(-10.0f, 10.0f); + std::vector fa(dims), fb(dims); + for (auto &x : fa) x = dist(rng); + for (auto &x : fb) x = dist(rng); + + std::vector buf_a(dims * sizeof(float) + alignof(float)); + std::vector buf_b(dims * sizeof(float) + alignof(float)); + const size_t off_a = + misaligned_float_offset(reinterpret_cast(buf_a.data())); + const size_t off_b = + misaligned_float_offset(reinterpret_cast(buf_b.data())); + std::memcpy(buf_a.data() + off_a, fa.data(), dims * sizeof(float)); + std::memcpy(buf_b.data() + off_b, fb.data(), dims * sizeof(float)); + + const char *ma = buf_a.data() + off_a; + const char *mb = buf_b.data() + off_b; + ASSERT_FALSE(is_aligned_to(ma, alignof(float))); + ASSERT_FALSE(is_aligned_to(mb, alignof(float))); + + const char *aa = (const char *)fa.data(); + const char *ab = (const char *)fb.data(); + + EXPECT_DOUBLE_EQ(vector_distance_euclidean_squared(aa, ab, dims), + vector_distance_euclidean_squared(ma, mb, dims)); + EXPECT_DOUBLE_EQ(euclidean_l2(aa, ab, dims), euclidean_l2(ma, mb, dims)); + EXPECT_DOUBLE_EQ(vector_distance_cosine(aa, ab, dims), + vector_distance_cosine(ma, mb, dims)); + EXPECT_DOUBLE_EQ(vector_distance_dot(aa, ab, dims), + vector_distance_dot(ma, mb, dims)); + EXPECT_DOUBLE_EQ(vector_distance_manhattan(aa, ab, dims), + vector_distance_manhattan(ma, mb, dims)); +} + +TEST_F(VectorDistanceTest, UnalignedMatchesAligned) { + // dims=8 is below VECTOR_DISTANCE_WIDE_MIN_DIMS (narrow tier: SSE4.2/NEON/ + // scalar). dims=35 is above it and odd, so it exercises the wide tier's + // (AVX2/AVX-512/SVE2) loadu path together with its tail handling. + CheckUnalignedMatchesAligned(8); + CheckUnalignedMatchesAligned(35); +} + +// --------------------------------------------------------------------------- +// SIMD parity: aligned result matches double-precision scalar reference. +// On CPUs without SIMD both sides run the same scalar code, so the test +// degenerates into an identity check — still a useful correctness signal. +// --------------------------------------------------------------------------- + +TEST_F(VectorDistanceTest, EuclideanSquaredParityWithReference) { + std::mt19937 rng(42); + std::uniform_real_distribution dist(-10.0f, 10.0f); + + for (uint32_t dims : {4u, 8u, 16u, 32u, 128u, 512u}) { + // Use heap vectors; malloc guarantees at least 16-byte alignment, + // which satisfies our alignof(float)=4 dispatch gate. + std::vector a(dims), b(dims); + for (auto &x : a) x = dist(rng); + for (auto &x : b) x = dist(rng); + + const double got = vector_distance_euclidean_squared( + (const char *)a.data(), (const char *)b.data(), dims); + const double ref = ref_euclidean_squared(a.data(), b.data(), dims); + // Allow 0.01% relative tolerance for float-precision SIMD accumulation. + EXPECT_NEAR(got, ref, ref * 1e-4 + 1e-9) << "dims=" << dims; + } +} + +TEST_F(VectorDistanceTest, EuclideanParityWithReference) { + std::mt19937 rng(42); + std::uniform_real_distribution dist(-10.0f, 10.0f); + + for (uint32_t dims : {4u, 8u, 16u, 32u, 128u, 512u}) { + std::vector a(dims), b(dims); + for (auto &x : a) x = dist(rng); + for (auto &x : b) x = dist(rng); + + const double got = + euclidean_l2((const char *)a.data(), (const char *)b.data(), dims); + const double ref = ref_euclidean(a.data(), b.data(), dims); + EXPECT_NEAR(got, ref, ref * 1e-4 + 1e-9) << "dims=" << dims; + } +} + +TEST_F(VectorDistanceTest, CosineParityWithReference) { + std::mt19937 rng(123); + std::uniform_real_distribution dist(-5.0f, 5.0f); + + for (uint32_t dims : {4u, 8u, 16u, 32u, 128u, 512u}) { + std::vector a(dims), b(dims); + for (auto &x : a) x = dist(rng); + for (auto &x : b) x = dist(rng); + + const double got = vector_distance_cosine((const char *)a.data(), + (const char *)b.data(), dims); + const double ref = ref_cosine(a.data(), b.data(), dims); + EXPECT_NEAR(got, ref, 1e-4) << "dims=" << dims; + } +} + +TEST_F(VectorDistanceTest, DotProductParityWithReference) { + std::mt19937 rng(77); + std::uniform_real_distribution dist(-10.0f, 10.0f); + + for (uint32_t dims : {4u, 8u, 16u, 32u, 128u, 512u}) { + std::vector a(dims), b(dims); + for (auto &x : a) x = dist(rng); + for (auto &x : b) x = dist(rng); + + const double got = vector_distance_dot((const char *)a.data(), + (const char *)b.data(), dims); + const double ref = ref_dot_product(a.data(), b.data(), dims); + EXPECT_NEAR(got, ref, std::abs(ref) * 1e-4 + 1e-9) << "dims=" << dims; + } +} + +TEST_F(VectorDistanceTest, ManhattanParityWithReference) { + std::mt19937 rng(55); + std::uniform_real_distribution dist(-10.0f, 10.0f); + + for (uint32_t dims : {4u, 8u, 16u, 32u, 128u, 512u}) { + std::vector a(dims), b(dims); + for (auto &x : a) x = dist(rng); + for (auto &x : b) x = dist(rng); + + const double got = vector_distance_manhattan((const char *)a.data(), + (const char *)b.data(), dims); + const double ref = ref_manhattan(a.data(), b.data(), dims); + EXPECT_NEAR(got, ref, ref * 1e-4 + 1e-9) << "dims=" << dims; + } +} + +// --------------------------------------------------------------------------- +// Per-tier parity tests +// +// A separate parameterized fixture calls init_vector_distance_functions_tier() +// for each registered tier, skipping tiers that are unavailable on this CPU or +// build. This ensures every SIMD kernel is tested for correctness independently +// — including inferior tiers on CPUs that support a higher one. +// +// The existing VectorDistanceTest suite is untouched; it still exercises the +// production path via init_vector_distance_functions() (highest tier on this +// CPU). +// --------------------------------------------------------------------------- + +static const char *tier_name(VectorDistanceTier tier) { + switch (tier) { + case VectorDistanceTier::Scalar: + return "Scalar"; + case VectorDistanceTier::Sse42: + return "Sse42"; + case VectorDistanceTier::Avx2: + return "Avx2"; + case VectorDistanceTier::Avx512f: + return "Avx512f"; + case VectorDistanceTier::Neon: + return "Neon"; + case VectorDistanceTier::Sve2: + return "Sve2"; + } + return "Unknown"; +} + +class VectorDistanceTierParityTest + : public ::testing::TestWithParam { + protected: + void SetUp() override { + const VectorDistanceTier t = GetParam(); + if (!vector_distance_tier_available(t)) + GTEST_SKIP() << tier_name(t) << " not available on this CPU/build"; + init_vector_distance_functions_tier(t); +#if defined(__x86_64__) || defined(_M_X64) + if (t == VectorDistanceTier::Avx2 || t == VectorDistanceTier::Avx512f) { + EXPECT_EQ(vector_distance_wide_tier(), t); + EXPECT_EQ(vector_distance_narrow_tier(), + vector_distance_tier_available(VectorDistanceTier::Sse42) + ? VectorDistanceTier::Sse42 + : VectorDistanceTier::Scalar); + } +#endif + } + void TearDown() override { init_vector_distance_functions(); } +}; + +TEST_P(VectorDistanceTierParityTest, EuclideanSquaredParityPerTier) { + std::mt19937 rng(42); + std::uniform_real_distribution dist(-10.0f, 10.0f); + + // 16383 is + // 16·1023+15 for AVX-512 + // 8·2047+7 for AVX2 + // 4·4095+3 for SSE/NEON + // so every tail loop is tested. + for (uint32_t dims : {4u, 8u, 32u, 128u, 1024u, 16383u}) { + std::vector a(dims), b(dims); + for (auto &x : a) x = dist(rng); + for (auto &x : b) x = dist(rng); + + const double got = vector_distance_euclidean_squared( + (const char *)a.data(), (const char *)b.data(), dims); + const double ref = ref_euclidean_squared(a.data(), b.data(), dims); + EXPECT_NEAR(got, ref, ref * 1e-4 + 1e-9) + << "tier=" << tier_name(GetParam()) << " dims=" << dims; + } +} + +TEST_P(VectorDistanceTierParityTest, EuclideanParityPerTier) { + std::mt19937 rng(42); + std::uniform_real_distribution dist(-10.0f, 10.0f); + + for (uint32_t dims : {4u, 8u, 32u, 128u, 1024u, 16383u}) { + std::vector a(dims), b(dims); + for (auto &x : a) x = dist(rng); + for (auto &x : b) x = dist(rng); + + const double got = + euclidean_l2((const char *)a.data(), (const char *)b.data(), dims); + const double ref = ref_euclidean(a.data(), b.data(), dims); + EXPECT_NEAR(got, ref, ref * 1e-4 + 1e-9) + << "tier=" << tier_name(GetParam()) << " dims=" << dims; + } +} + +TEST_P(VectorDistanceTierParityTest, CosineParityPerTier) { + std::mt19937 rng(123); + std::uniform_real_distribution dist(-5.0f, 5.0f); + + for (uint32_t dims : {4u, 8u, 32u, 128u, 1024u, 16383u}) { + std::vector a(dims), b(dims); + for (auto &x : a) x = dist(rng); + for (auto &x : b) x = dist(rng); + + const double got = vector_distance_cosine((const char *)a.data(), + (const char *)b.data(), dims); + const double ref = ref_cosine(a.data(), b.data(), dims); + EXPECT_NEAR(got, ref, 1e-4) + << "tier=" << tier_name(GetParam()) << " dims=" << dims; + } +} + +TEST_P(VectorDistanceTierParityTest, DotProductParityPerTier) { + std::mt19937 rng(77); + std::uniform_real_distribution dist(-10.0f, 10.0f); + + for (uint32_t dims : {4u, 8u, 32u, 128u, 1024u, 16383u}) { + std::vector a(dims), b(dims); + for (auto &x : a) x = dist(rng); + for (auto &x : b) x = dist(rng); + + const double got = vector_distance_dot((const char *)a.data(), + (const char *)b.data(), dims); + const double ref = ref_dot_product(a.data(), b.data(), dims); + EXPECT_NEAR(got, ref, std::abs(ref) * 1e-4 + 1e-9) + << "tier=" << tier_name(GetParam()) << " dims=" << dims; + } +} + +TEST_P(VectorDistanceTierParityTest, ManhattanParityPerTier) { + std::mt19937 rng(55); + std::uniform_real_distribution dist(-10.0f, 10.0f); + + for (uint32_t dims : {4u, 8u, 32u, 128u, 1024u, 16383u}) { + std::vector a(dims), b(dims); + for (auto &x : a) x = dist(rng); + for (auto &x : b) x = dist(rng); + + const double got = vector_distance_manhattan((const char *)a.data(), + (const char *)b.data(), dims); + const double ref = ref_manhattan(a.data(), b.data(), dims); + EXPECT_NEAR(got, ref, ref * 1e-4 + 1e-9) + << "tier=" << tier_name(GetParam()) << " dims=" << dims; + } +} + +TEST_P(VectorDistanceTierParityTest, OverflowFallbackPerTier) { + // A 16-dim vector (>= 16 triggers the wide kernel) with one element = 2e38. + // (2e38)^2 ~ 4e76 overflows float32 (FLT_MAX ~ 3.4e38); the SIMD accumulator + // becomes +Inf without the fallback. The scalar path uses double throughout + // and returns a finite result. Verify the fix: result must be finite and + // equal to the scalar reference. + constexpr uint32_t dims = 16; + std::vector a(dims, 0.0f), b(dims, 0.0f); + a[0] = 2e38f; + + // Euclidean squared: scalar = (2e38)^2; broken SIMD would give +Inf (-> SQL + // NULL). + const double got_e = vector_distance_euclidean_squared( + (const char *)a.data(), (const char *)b.data(), dims); + EXPECT_TRUE(std::isfinite(got_e)) << "tier=" << tier_name(GetParam()); + EXPECT_EQ(got_e, ref_euclidean_squared(a.data(), b.data(), dims)) + << "tier=" << tier_name(GetParam()); + + // Euclidean L2: scalar = 2e38; mirrors SQL EUCLIDEAN (sqrt of squared). + const double got_l2 = + euclidean_l2((const char *)a.data(), (const char *)b.data(), dims); + EXPECT_TRUE(std::isfinite(got_l2)) << "tier=" << tier_name(GetParam()); + EXPECT_EQ(got_l2, ref_euclidean(a.data(), b.data(), dims)) + << "tier=" << tier_name(GetParam()); + + // Manhattan: scalar = 2e38. + const double got_m = vector_distance_manhattan((const char *)a.data(), + (const char *)b.data(), dims); + EXPECT_TRUE(std::isfinite(got_m)) << "tier=" << tier_name(GetParam()); + EXPECT_EQ(got_m, ref_manhattan(a.data(), b.data(), dims)) + << "tier=" << tier_name(GetParam()); + + // Dot: a[0]*b2[0] = 2e38*2e38 overflows float32; scalar = 4e76 (finite in + // double). + std::vector b2(dims, 0.0f); + b2[0] = 2e38f; + const double got_d = vector_distance_dot((const char *)a.data(), + (const char *)b2.data(), dims); + EXPECT_TRUE(std::isfinite(got_d)) << "tier=" << tier_name(GetParam()); + EXPECT_EQ(got_d, ref_dot_product(a.data(), b2.data(), dims)) + << "tier=" << tier_name(GetParam()); + + // Cosine: a == a (same pointer) => cosine distance = 0. + // Float32 norm overflow to Inf causes 1 - Inf/Inf = NaN without the fix. + const double got_c = vector_distance_cosine((const char *)a.data(), + (const char *)a.data(), dims); + EXPECT_NEAR(got_c, 0.0, 1e-9) << "tier=" << tier_name(GetParam()); +} + +static std::string tier_param_name( + const ::testing::TestParamInfo &info) { + return tier_name(info.param); +} + +#if defined(__x86_64__) || defined(_M_X64) +INSTANTIATE_TEST_SUITE_P(AllTiers, VectorDistanceTierParityTest, + ::testing::Values(VectorDistanceTier::Scalar, + VectorDistanceTier::Sse42, + VectorDistanceTier::Avx2, + VectorDistanceTier::Avx512f), + tier_param_name); +#elif defined(__aarch64__) || defined(_M_ARM64) +INSTANTIATE_TEST_SUITE_P(AllTiers, VectorDistanceTierParityTest, + ::testing::Values(VectorDistanceTier::Scalar, + VectorDistanceTier::Neon, + VectorDistanceTier::Sve2), + tier_param_name); +#else +INSTANTIATE_TEST_SUITE_P(AllTiers, VectorDistanceTierParityTest, + ::testing::Values(VectorDistanceTier::Scalar), + tier_param_name); +#endif + +// --------------------------------------------------------------------------- +// Dispatch description / tier reporting +// --------------------------------------------------------------------------- + +class VectorDistanceDispatchTest : public ::testing::Test { + protected: + void TearDown() override { init_vector_distance_functions(); } +}; + +TEST_F(VectorDistanceDispatchTest, ProductionInitIsIdempotent) { + init_vector_distance_functions(); + const auto wide = vector_distance_wide_tier(); + const auto narrow = vector_distance_narrow_tier(); + init_vector_distance_functions(); + EXPECT_EQ(vector_distance_wide_tier(), wide); + EXPECT_EQ(vector_distance_narrow_tier(), narrow); +} + +TEST_F(VectorDistanceDispatchTest, ProductionInitRestoredAfterScalarOverride) { + init_vector_distance_functions(); + const VectorDistanceTier prod_wide = vector_distance_wide_tier(); + const VectorDistanceTier prod_narrow = vector_distance_narrow_tier(); + + init_vector_distance_functions_tier(VectorDistanceTier::Scalar); + EXPECT_EQ(vector_distance_wide_tier(), VectorDistanceTier::Scalar); + EXPECT_EQ(vector_distance_narrow_tier(), VectorDistanceTier::Scalar); + + init_vector_distance_functions(); + EXPECT_EQ(vector_distance_wide_tier(), prod_wide); + EXPECT_EQ(vector_distance_narrow_tier(), prod_narrow); +} + +TEST_F(VectorDistanceDispatchTest, DescriptionIsFragmentForLogMessage) { + init_vector_distance_functions(); + char msg[256]; + const size_t len = vector_distance_dispatch_description(msg, sizeof(msg)); + EXPECT_GT(len, 0u); + // The description is spliced into ER_VECTOR_DISTANCE_SIMD_DISPATCH's + // "For VECTOR_DISTANCE using %s", so it is a sentence fragment: a leading + // space, no repeated "DISTANCE()"/"VECTOR_DISTANCE" wording of its own. + EXPECT_EQ(msg[0], ' '); + EXPECT_EQ(std::string(msg).find("DISTANCE"), std::string::npos); +} + +TEST_F(VectorDistanceDispatchTest, ForcedScalarTierUpdatesDescription) { + init_vector_distance_functions_tier(VectorDistanceTier::Scalar); + EXPECT_EQ(vector_distance_wide_tier(), VectorDistanceTier::Scalar); + EXPECT_EQ(vector_distance_narrow_tier(), VectorDistanceTier::Scalar); + + char msg[256]; + vector_distance_dispatch_description(msg, sizeof(msg)); + EXPECT_NE(std::string(msg).find("software scalar"), std::string::npos); +} + +#if defined(__x86_64__) || defined(_M_X64) +TEST_F(VectorDistanceDispatchTest, + SplitDispatchMentionsDimensionsWhenWideDiffers) { + if (!vector_distance_tier_available(VectorDistanceTier::Avx2) || + !vector_distance_tier_available(VectorDistanceTier::Sse42)) { + GTEST_SKIP() << "requires AVX2 wide path and SSE4.2 narrow path"; + } + + init_vector_distance_functions(); + ASSERT_NE(vector_distance_wide_tier(), vector_distance_narrow_tier()); + + char msg[256]; + vector_distance_dispatch_description(msg, sizeof(msg)); + const std::string ge = + "dimensions >= " + std::to_string(VECTOR_DISTANCE_WIDE_MIN_DIMS); + const std::string lt = + "dimensions < " + std::to_string(VECTOR_DISTANCE_WIDE_MIN_DIMS); + EXPECT_NE(std::string(msg).find(ge), std::string::npos); + EXPECT_NE(std::string(msg).find(lt), std::string::npos); + EXPECT_NE(std::string(msg).find("SSE4.2"), std::string::npos); +} +#endif + +} // namespace vector_distance_unittest diff --git a/unittest/gunit/vector_distance_benchmark-t.cc b/unittest/gunit/vector_distance_benchmark-t.cc new file mode 100644 index 000000000000..1e966d02665f --- /dev/null +++ b/unittest/gunit/vector_distance_benchmark-t.cc @@ -0,0 +1,306 @@ +/* Copyright (c) 2026, Percona and/or its affiliates. + + This program is free software; you can redistribute it and/or modify + it under the terms of the GNU General Public License, version 2.0, + as published by the Free Software Foundation. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License, version 2.0, for more details. + + You should have received a copy of the GNU General Public License + along with this program; if not, write to the Free Software + Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA */ + +/** + @file vector_distance_benchmark-t.cc + + Per-tier microbenchmarks for vector_distance_euclidean_squared(), + vector_distance_cosine(), vector_distance_dot(), and + vector_distance_manhattan(). + + Euclidean (L2) benchmarks apply std::sqrt() on top of the squared kernel, + mirroring the SQL EUCLIDEAN path in Item_func_vector_distance::val_real. + EuclideanSquared benchmarks measure the kernel alone (SQL EUCLIDEAN_SQUARED). + + Every combination of (metric × tier × size) is registered as a separate + Google Test case via the BENCHMARK() macro. At the start of each case, + vector_distance_tier_available() is checked and GTEST_SKIP() is called when + the tier is not supported by the current CPU or build. This means: + + - On a Scalar-only x86 host: only Scalar tests execute; Sse42/Avx2/Avx512f + are skipped. + - On an AVX-512 capable host: all four x86 tiers run, so inferior-tier + throughput is also measured. + - On aarch64 without a SVE2 build: Neon runs, Sve2 is skipped. + + Sizes benchmarked: 4, 8, 32, 128, 1024, 16383 (float32 elements). + Metrics: Euclidean, EuclideanSquared, Cosine, DotProduct, Manhattan. + + Tiers registered per platform: + x86_64 — Scalar, Sse42, Avx2, Avx512f + aarch64 — Scalar, Neon, Sve2 + other — Scalar only +*/ + +#include + +#include +#include +#include +#include + +#include "unittest/gunit/benchmark.h" +#include "vector-common/vector_distance.h" + +namespace vector_distance_tier_bench { + +// Volatile sink prevents the optimizer from discarding computed distances. +static volatile double bench_sink; + +enum class Metric { + Euclidean, + EuclideanSquared, + Cosine, + DotProduct, + Manhattan +}; + +static void fill_random(float *data, uint32_t n, uint32_t seed) { + std::mt19937 rng(seed); + std::uniform_real_distribution dist(-1.0f, 1.0f); + for (uint32_t i = 0; i < n; i++) data[i] = dist(rng); +} + +// --------------------------------------------------------------------------- +// Generic benchmark body +// --------------------------------------------------------------------------- + +template +static void bench_impl(size_t num_iterations) { +#ifndef NDEBUG + // benchmark.cc calls StartBenchmarkTiming() before invoking func(), which + // conflicts with our inner StartBenchmarkTiming() and triggers the + // assert(!timer_running) in debug builds. Timings are meaningless in debug + // mode regardless; skip cleanly instead. + GTEST_SKIP() << "Benchmarks skipped in debug builds " + "(build with -DWITH_DEBUG=OFF for meaningful results)"; +#endif + if (!vector_distance_tier_available(kTier)) { + const char *names[] = {"Scalar", "Sse42", "Avx2", + "Avx512f", "Neon", "Sve2"}; + const int idx = static_cast(kTier); + GTEST_SKIP() << names[idx] << " not available on this CPU/build"; + } + + init_vector_distance_functions_tier(kTier); + + std::vector a(kDims), b(kDims); + fill_random(a.data(), kDims, 1); + fill_random(b.data(), kDims, 2); + + StartBenchmarkTiming(); + for (size_t i = 0; i < num_iterations; i++) { + if constexpr (kMetric == Metric::Cosine) + bench_sink = vector_distance_cosine((const char *)a.data(), + (const char *)b.data(), kDims); + else if constexpr (kMetric == Metric::DotProduct) + bench_sink = vector_distance_dot((const char *)a.data(), + (const char *)b.data(), kDims); + else if constexpr (kMetric == Metric::Manhattan) + bench_sink = vector_distance_manhattan((const char *)a.data(), + (const char *)b.data(), kDims); + else if constexpr (kMetric == Metric::Euclidean) + bench_sink = std::sqrt(vector_distance_euclidean_squared( + (const char *)a.data(), (const char *)b.data(), kDims)); + else + bench_sink = vector_distance_euclidean_squared( + (const char *)a.data(), (const char *)b.data(), kDims); + } + StopBenchmarkTiming(); + SetBytesProcessed(num_iterations * kDims * sizeof(float) * 2); +} + +// --------------------------------------------------------------------------- +// Macro machinery +// --------------------------------------------------------------------------- + +// Expands M(size) for each of the six benchmark sizes. +#define FOR_EACH_SIZE(M) M(4) M(8) M(32) M(128) M(1024) M(16383) + +// Registers one Euclidean (L2 with sqrt) benchmark for a given tier + size. +#define BENCH_EUCLIDEAN_ONE(tier_enum, tier_label, dims) \ + static void BenchEuclidean_##tier_label##_##dims(size_t n) { \ + bench_impl(n); \ + } \ + BENCHMARK(BenchEuclidean_##tier_label##_##dims) + +// Registers one EuclideanSquared benchmark for a given tier + size. +#define BENCH_EUCLIDEAN_SQUARED_ONE(tier_enum, tier_label, dims) \ + static void BenchEuclideanSquared_##tier_label##_##dims(size_t n) { \ + bench_impl( \ + n); \ + } \ + BENCHMARK(BenchEuclideanSquared_##tier_label##_##dims) + +// Registers one Cosine benchmark function for a given tier + size. +#define BENCH_COSINE_ONE(tier_enum, tier_label, dims) \ + static void BenchCosine_##tier_label##_##dims(size_t n) { \ + bench_impl(n); \ + } \ + BENCHMARK(BenchCosine_##tier_label##_##dims) + +// Per-size expanders (one per size, named so the tier macro can paste them). +#define BENCH_EUCLIDEAN_Scalar(dims) BENCH_EUCLIDEAN_ONE(Scalar, Scalar, dims) +#define BENCH_EUCLIDEAN_Sse42(dims) BENCH_EUCLIDEAN_ONE(Sse42, Sse42, dims) +#define BENCH_EUCLIDEAN_Avx2(dims) BENCH_EUCLIDEAN_ONE(Avx2, Avx2, dims) +#define BENCH_EUCLIDEAN_Avx512f(dims) \ + BENCH_EUCLIDEAN_ONE(Avx512f, Avx512f, dims) +#define BENCH_EUCLIDEAN_Neon(dims) BENCH_EUCLIDEAN_ONE(Neon, Neon, dims) +#define BENCH_EUCLIDEAN_Sve2(dims) BENCH_EUCLIDEAN_ONE(Sve2, Sve2, dims) + +#define BENCH_EUCLIDEAN_SQUARED_Scalar(dims) \ + BENCH_EUCLIDEAN_SQUARED_ONE(Scalar, Scalar, dims) +#define BENCH_EUCLIDEAN_SQUARED_Sse42(dims) \ + BENCH_EUCLIDEAN_SQUARED_ONE(Sse42, Sse42, dims) +#define BENCH_EUCLIDEAN_SQUARED_Avx2(dims) \ + BENCH_EUCLIDEAN_SQUARED_ONE(Avx2, Avx2, dims) +#define BENCH_EUCLIDEAN_SQUARED_Avx512f(dims) \ + BENCH_EUCLIDEAN_SQUARED_ONE(Avx512f, Avx512f, dims) +#define BENCH_EUCLIDEAN_SQUARED_Neon(dims) \ + BENCH_EUCLIDEAN_SQUARED_ONE(Neon, Neon, dims) +#define BENCH_EUCLIDEAN_SQUARED_Sve2(dims) \ + BENCH_EUCLIDEAN_SQUARED_ONE(Sve2, Sve2, dims) + +#define BENCH_COSINE_Scalar(dims) BENCH_COSINE_ONE(Scalar, Scalar, dims) +#define BENCH_COSINE_Sse42(dims) BENCH_COSINE_ONE(Sse42, Sse42, dims) +#define BENCH_COSINE_Avx2(dims) BENCH_COSINE_ONE(Avx2, Avx2, dims) +#define BENCH_COSINE_Avx512f(dims) BENCH_COSINE_ONE(Avx512f, Avx512f, dims) +#define BENCH_COSINE_Neon(dims) BENCH_COSINE_ONE(Neon, Neon, dims) +#define BENCH_COSINE_Sve2(dims) BENCH_COSINE_ONE(Sve2, Sve2, dims) + +// Registers one DotProduct benchmark function for a given tier + size. +#define BENCH_DOT_PRODUCT_ONE(tier_enum, tier_label, dims) \ + static void BenchDotProduct_##tier_label##_##dims(size_t n) { \ + bench_impl(n); \ + } \ + BENCHMARK(BenchDotProduct_##tier_label##_##dims) + +#define BENCH_DOT_PRODUCT_Scalar(dims) \ + BENCH_DOT_PRODUCT_ONE(Scalar, Scalar, dims) +#define BENCH_DOT_PRODUCT_Sse42(dims) BENCH_DOT_PRODUCT_ONE(Sse42, Sse42, dims) +#define BENCH_DOT_PRODUCT_Avx2(dims) BENCH_DOT_PRODUCT_ONE(Avx2, Avx2, dims) +#define BENCH_DOT_PRODUCT_Avx512f(dims) \ + BENCH_DOT_PRODUCT_ONE(Avx512f, Avx512f, dims) +#define BENCH_DOT_PRODUCT_Neon(dims) BENCH_DOT_PRODUCT_ONE(Neon, Neon, dims) +#define BENCH_DOT_PRODUCT_Sve2(dims) BENCH_DOT_PRODUCT_ONE(Sve2, Sve2, dims) + +// Registers one Manhattan benchmark function for a given tier + size. +#define BENCH_MANHATTAN_ONE(tier_enum, tier_label, dims) \ + static void BenchManhattan_##tier_label##_##dims(size_t n) { \ + bench_impl(n); \ + } \ + BENCHMARK(BenchManhattan_##tier_label##_##dims) + +#define BENCH_MANHATTAN_Scalar(dims) BENCH_MANHATTAN_ONE(Scalar, Scalar, dims) +#define BENCH_MANHATTAN_Sse42(dims) BENCH_MANHATTAN_ONE(Sse42, Sse42, dims) +#define BENCH_MANHATTAN_Avx2(dims) BENCH_MANHATTAN_ONE(Avx2, Avx2, dims) +#define BENCH_MANHATTAN_Avx512f(dims) \ + BENCH_MANHATTAN_ONE(Avx512f, Avx512f, dims) +#define BENCH_MANHATTAN_Neon(dims) BENCH_MANHATTAN_ONE(Neon, Neon, dims) +#define BENCH_MANHATTAN_Sve2(dims) BENCH_MANHATTAN_ONE(Sve2, Sve2, dims) + +// --------------------------------------------------------------------------- +// Tier registrations — all tiers are always declared; unavailable ones skip. +// --------------------------------------------------------------------------- + +// Tier 0 — Scalar (all platforms) +FOR_EACH_SIZE(BENCH_EUCLIDEAN_Scalar) +FOR_EACH_SIZE(BENCH_EUCLIDEAN_SQUARED_Scalar) +FOR_EACH_SIZE(BENCH_COSINE_Scalar) +FOR_EACH_SIZE(BENCH_DOT_PRODUCT_Scalar) +FOR_EACH_SIZE(BENCH_MANHATTAN_Scalar) + +// x86_64 tiers +#if defined(__x86_64__) || defined(_M_X64) + +FOR_EACH_SIZE(BENCH_EUCLIDEAN_Sse42) +FOR_EACH_SIZE(BENCH_EUCLIDEAN_SQUARED_Sse42) +FOR_EACH_SIZE(BENCH_COSINE_Sse42) +FOR_EACH_SIZE(BENCH_DOT_PRODUCT_Sse42) +FOR_EACH_SIZE(BENCH_MANHATTAN_Sse42) + +FOR_EACH_SIZE(BENCH_EUCLIDEAN_Avx2) +FOR_EACH_SIZE(BENCH_EUCLIDEAN_SQUARED_Avx2) +FOR_EACH_SIZE(BENCH_COSINE_Avx2) +FOR_EACH_SIZE(BENCH_DOT_PRODUCT_Avx2) +FOR_EACH_SIZE(BENCH_MANHATTAN_Avx2) + +FOR_EACH_SIZE(BENCH_EUCLIDEAN_Avx512f) +FOR_EACH_SIZE(BENCH_EUCLIDEAN_SQUARED_Avx512f) +FOR_EACH_SIZE(BENCH_COSINE_Avx512f) +FOR_EACH_SIZE(BENCH_DOT_PRODUCT_Avx512f) +FOR_EACH_SIZE(BENCH_MANHATTAN_Avx512f) + +#endif // x86_64 + +// aarch64 tiers +#if defined(__aarch64__) || defined(_M_ARM64) + +FOR_EACH_SIZE(BENCH_EUCLIDEAN_Neon) +FOR_EACH_SIZE(BENCH_EUCLIDEAN_SQUARED_Neon) +FOR_EACH_SIZE(BENCH_COSINE_Neon) +FOR_EACH_SIZE(BENCH_DOT_PRODUCT_Neon) +FOR_EACH_SIZE(BENCH_MANHATTAN_Neon) + +FOR_EACH_SIZE(BENCH_EUCLIDEAN_Sve2) +FOR_EACH_SIZE(BENCH_EUCLIDEAN_SQUARED_Sve2) +FOR_EACH_SIZE(BENCH_COSINE_Sve2) +FOR_EACH_SIZE(BENCH_DOT_PRODUCT_Sve2) +FOR_EACH_SIZE(BENCH_MANHATTAN_Sve2) + +#endif // aarch64 + +// --------------------------------------------------------------------------- +// Cleanup macros +// --------------------------------------------------------------------------- + +#undef BENCH_MANHATTAN_Sve2 +#undef BENCH_MANHATTAN_Neon +#undef BENCH_MANHATTAN_Avx512f +#undef BENCH_MANHATTAN_Avx2 +#undef BENCH_MANHATTAN_Sse42 +#undef BENCH_MANHATTAN_Scalar +#undef BENCH_MANHATTAN_ONE +#undef BENCH_DOT_PRODUCT_Sve2 +#undef BENCH_DOT_PRODUCT_Neon +#undef BENCH_DOT_PRODUCT_Avx512f +#undef BENCH_DOT_PRODUCT_Avx2 +#undef BENCH_DOT_PRODUCT_Sse42 +#undef BENCH_DOT_PRODUCT_Scalar +#undef BENCH_DOT_PRODUCT_ONE +#undef BENCH_COSINE_Sve2 +#undef BENCH_COSINE_Neon +#undef BENCH_COSINE_Avx512f +#undef BENCH_COSINE_Avx2 +#undef BENCH_COSINE_Sse42 +#undef BENCH_COSINE_Scalar +#undef BENCH_EUCLIDEAN_SQUARED_Sve2 +#undef BENCH_EUCLIDEAN_SQUARED_Neon +#undef BENCH_EUCLIDEAN_SQUARED_Avx512f +#undef BENCH_EUCLIDEAN_SQUARED_Avx2 +#undef BENCH_EUCLIDEAN_SQUARED_Sse42 +#undef BENCH_EUCLIDEAN_SQUARED_Scalar +#undef BENCH_EUCLIDEAN_SQUARED_ONE +#undef BENCH_EUCLIDEAN_Sve2 +#undef BENCH_EUCLIDEAN_Neon +#undef BENCH_EUCLIDEAN_Avx512f +#undef BENCH_EUCLIDEAN_Avx2 +#undef BENCH_EUCLIDEAN_Sse42 +#undef BENCH_EUCLIDEAN_Scalar +#undef BENCH_COSINE_ONE +#undef BENCH_EUCLIDEAN_ONE +#undef FOR_EACH_SIZE + +} // namespace vector_distance_tier_bench diff --git a/vector-common/vector_distance.cc b/vector-common/vector_distance.cc new file mode 100644 index 000000000000..fb6ae83a6351 --- /dev/null +++ b/vector-common/vector_distance.cc @@ -0,0 +1,1106 @@ +/* Copyright (c) 2026, Percona and/or its affiliates. + + This program is free software; you can redistribute it and/or modify + it under the terms of the GNU General Public License, version 2.0, + as published by the Free Software Foundation. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License, version 2.0, for more details. + + You should have received a copy of the GNU General Public License + along with this program; if not, write to the Free Software + Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA */ + +/** + @file vector-common/vector_distance.cc + + Euclidean, cosine and dot-product distance for VECTOR columns (float32). + + Public entry points — vector_distance_euclidean_squared(), + vector_distance_cosine(), vector_distance_dot(), vector_distance_manhattan() + — are declared in + vector-common/vector_distance.h. Call init_vector_distance_functions() once + before first use so the dispatch pointers are set to the best kernel for the + host CPU. + + All kernels accept const char * and any byte alignment. SIMD paths use + unaligned load intrinsics (_mm_loadu_ps, _mm256_loadu_ps, _mm512_loadu_ps, + vld1q_f32, svld1_f32) rather than aligned variants: VECTOR payloads may + come from columns or misaligned SUBSTR blobs, and on modern x86/ARM CPUs + unaligned loads have the same throughput as aligned loads when the data + happens to be aligned. Aligned intrinsics would fault on misaligned + addresses without improving performance. The one remaining slowdown is + cache-line crossing: if a load spans a 64-byte cache-line boundary the + CPU must fetch from two lines, which costs extra regardless of whether + the instruction is MOVUPS or MOVAPS — that penalty depends on runtime + address, not on choosing loadu vs load. Scalar tails use memcpy to + avoid UB on misaligned float *. + + Dim-aware dispatch (Opt 2) + -------------------------- + Wide-tier kernels (AVX2, AVX-512, SVE2) have a minimum useful dimension: the + SIMD body only fires when dims ≥ register_width (8 for AVX2, 16 for AVX-512). + Below that threshold, calling a wide-tier kernel pays register-init overhead + with no SIMD benefit. + + To avoid this regression, the public wrappers use two function pointer sets: + g_* — the widest available tier; used for dims ≥ + VECTOR_DISTANCE_WIDE_MIN_DIMS g_*_narrow — SSE4.2 / NEON (fills at dim ≥ 4); + used for dims < VECTOR_DISTANCE_WIDE_MIN_DIMS + + SIMD tier model + --------------- + Kernels are grouped into four tiers. init_vector_distance_functions() selects + the highest tier the CPU and OS support at runtime. Each SIMD function is + compiled with its own GCC/Clang target attribute so this translation unit + stays at the baseline ISA; no global -mavx2 / -march=native is required. + + +------+----------+--------------------------------+-------------------+ + | Tier | Name | Optimization target | Register width | + +------+----------+--------------------------------+-------------------+ + | 0 | Scalar | Any x86_64 / ARM64 (fallback) | 32/64-bit | + | 1 | Legacy | SSE4.2 (Intel/AMD) / NEON (ARM)| 128-bit | + | 2 | Standard | AVX2 + FMA (Intel/AMD) | 256-bit | + | 3 | Ultra | AVX-512 (Intel/AMD) / SVE2(ARM)| 512-bit+ (VLA) | + +------+----------+--------------------------------+-------------------+ + + Tier 0 — Scalar (all architectures) + euclidean_scalar, cosine_scalar, dot_product_scalar, manhattan_scalar + Default function pointers; also forced by + init_vector_distance_functions_tier(VectorDistanceTier::Scalar). + + Tier 1 — Legacy + x86_64: euclidean_sse, cosine_sse, dot_product_sse, manhattan_sse + (target("sse4.2"), 4 floats/iter) cpu_has_sse42() + aarch64: euclidean_neon, cosine_neon, dot_product_neon, manhattan_neon + (target("+simd"), 4 floats/iter) Always enabled on ARMv8. + + Tier 2 — Standard (x86_64 only) + euclidean_avx2, cosine_avx2, dot_product_avx2, manhattan_avx2 + (target("avx2"), 8 floats/iter) cpu_has_avx2_fma() + + Tier 3 — Ultra + x86_64: euclidean_avx512, cosine_avx512, dot_product_avx512, + manhattan_avx512 (target("avx512f"), 16 floats/iter) cpu_has_avx512f() + aarch64: euclidean_sve2, cosine_sve2, dot_product_sve2, manhattan_sve2 + (target("+sve2"), scalable VLA) + + Runtime dispatch (init_vector_distance_functions) + ------------------------------------ + x86_64: AVX-512 -> AVX2+FMA -> SSE4.2 -> scalar (wide) + SSE4.2 -> scalar (narrow, dims < VECTOR_DISTANCE_WIDE_MIN_DIMS) + aarch64: NEON or SVE2 (wide); NEON (narrow) + other: scalar only (no-op init) + _WIN32 (x64 and ARM64): scalar only — SIMD tiers compiled out at build time. + + Euclidean kernels (euclidean_*) return sum((a[i]-b[i])²) without sqrt. + SQL applies std::sqrt for the EUCLIDEAN metric; EUCLIDEAN_SQUARED uses the + kernel result directly. + + Cosine distance returns +Inf as a sentinel when either vector has zero norm + (undefined cosine); the SQL layer (Item_func_vector_distance::val_real) + detects it via std::isinf and maps it to NULL. +*/ + +#include "vector-common/vector_distance.h" + +#include +#include +#include +#include +#include +#include + +#include "mysql/attribute.h" // MY_ATTRIBUTE + +// Platform guards — mirror ut0crc32.h:53-69 +// +// On _WIN32 (x64 and ARM64) SIMD tiers are not wired up; scalar-only, +// same pragmatic approach as CRC32_DEFAULT in ut0crc32.h. +#if !defined(_WIN32) +#if defined(__x86_64__) || defined(_M_X64) +#define VECTOR_DISTANCE_x86_64 +#elif defined(__aarch64__) || defined(_M_ARM64) +#define VECTOR_DISTANCE_AARCH64 +#endif +#endif + +#if !defined(VECTOR_DISTANCE_x86_64) && !defined(VECTOR_DISTANCE_AARCH64) +#define VECTOR_DISTANCE_DEFAULT +#endif + +// SVE2 is opt-in: it requires the toolchain to compile the unit with +// __ARM_FEATURE_SVE2 (e.g. -march=armv8-a+sve2 or armv9-a). Without that we +// keep NEON-only behaviour and avoid pulling in . +#if defined(VECTOR_DISTANCE_AARCH64) && defined(__ARM_FEATURE_SVE2) +#define VECTOR_DISTANCE_HAS_SVE2 +#endif + +// --------------------------------------------------------------------------- +// Scalar kernels — always compiled, safe on every architecture. +// Take const char * so they can handle any byte alignment: dereferencing a +// float * that is not suitably aligned is UB, so each element is fetched +// with memcpy into a local float instead. Compilers see the fixed 4-byte +// size and emit a single load instruction — no function call, no overhead. +// Scalar path accumulates in double. Inputs are float32, but double +// avoids overflow/precision loss (e.g. (2e38)² is Inf in float, finite +// in double) and matches the SIMD reduction path below. +// --------------------------------------------------------------------------- + +static double euclidean_scalar(const char *a_raw, const char *b_raw, + uint32_t dims) { + double result = 0.0; + for (uint32_t i = 0; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + const double d = av - bv; + result += d * d; + } + return result; +} + +static double cosine_scalar(const char *a_raw, const char *b_raw, + uint32_t dims) { + double ab = 0.0, norm_a = 0.0, norm_b = 0.0; + for (uint32_t i = 0; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + ab += (double)av * bv; + norm_a += (double)av * av; + norm_b += (double)bv * bv; + } + const double denom = sqrt(norm_a * norm_b); + // +Inf sentinel: zero-denom means undefined cosine (zero-vector input). + if (denom == 0.0) return std::numeric_limits::infinity(); + return 1.0 - ab / denom; +} + +static double dot_product_scalar(const char *a_raw, const char *b_raw, + uint32_t dims) { + double ab = 0.0; + for (uint32_t i = 0; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + ab += (double)av * bv; + } + return ab; +} + +static double manhattan_scalar(const char *a_raw, const char *b_raw, + uint32_t dims) { + double result = 0.0; + for (uint32_t i = 0; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + const double d = (double)av - bv; + result += std::fabs(d); + } + return result; +} + +// --------------------------------------------------------------------------- +// Function pointers — initialized to scalar; init_vector_distance_functions() +// may promote. Wide pointers (g_*) are used for dims ≥ +// VECTOR_DISTANCE_WIDE_MIN_DIMS. Narrow pointers (g_*_narrow) are used for dims +// < VECTOR_DISTANCE_WIDE_MIN_DIMS to avoid dispatching to AVX-512/AVX2 when the +// SIMD loop cannot fire (needs ≥ 16/8 elements). +// --------------------------------------------------------------------------- + +using vector_distance_fn_t = double (*)(const char *, const char *, uint32_t); + +static vector_distance_fn_t g_euclidean = euclidean_scalar; +static vector_distance_fn_t g_cosine = cosine_scalar; +static vector_distance_fn_t g_dot_product = dot_product_scalar; +static vector_distance_fn_t g_manhattan = manhattan_scalar; + +static vector_distance_fn_t g_euclidean_narrow = euclidean_scalar; +static vector_distance_fn_t g_cosine_narrow = cosine_scalar; +static vector_distance_fn_t g_dot_product_narrow = dot_product_scalar; +static vector_distance_fn_t g_manhattan_narrow = manhattan_scalar; + +static VectorDistanceTier g_wide_tier = VectorDistanceTier::Scalar; +static VectorDistanceTier g_narrow_tier = VectorDistanceTier::Scalar; + +// --------------------------------------------------------------------------- +// SIMD kernels — see file header for the full tier table and dispatch order. +// --------------------------------------------------------------------------- + +#ifdef VECTOR_DISTANCE_x86_64 + +#include + +// CPU feature detection ----------------------------------------------------- + +static bool cpu_has_sse42() { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_cpu_supports("sse4.2"); +#else + return false; +#endif +} + +static bool cpu_has_avx2_fma() { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_cpu_supports("avx2") && __builtin_cpu_supports("fma"); +#else + return false; +#endif +} + +// __builtin_cpu_supports already gates AVX-512 on OS support: libgcc and +// compiler-rt only report the feature when OSXSAVE is set and XCR0 enables +// the full ZMM/opmask state, so no manual XGETBV check is needed (hnswlib +// does it by hand only because it reads raw CPUID bits). +static bool cpu_has_avx512f() { +#if defined(__GNUC__) || defined(__clang__) + return __builtin_cpu_supports("avx512f"); +#else + return false; +#endif +} + +// SIMD loops accumulate in float32 registers for full vector width. +// Each kernel promotes to double once (horizontal sum + scalar tail); +// float32-only reduction would not be faster and loses precision. +// However, individual float32 products (d*d for Euclidean, a[i]*b[i] for +// cosine/dot) can overflow to +Inf for extreme float32 inputs (e.g. element +// difference near FLT_MAX). Each kernel checks for a non-finite horizontal +// sum and falls back to the scalar path, which uses double throughout. + +// Tier 1 — SSE4.2, 4 floats per iteration ----------------------------------- + +// Horizontal sum: reduce the 4 float lanes of an SSE register to one float +// (SSE counterpart of the AVX-512 _mm512_reduce_add_ps intrinsic). +MY_ATTRIBUTE((target("sse4.2"))) +static float hsum128_sse(__m128 v) { + __m128 s = _mm_hadd_ps(v, v); + s = _mm_hadd_ps(s, s); + return _mm_cvtss_f32(s); +} + +MY_ATTRIBUTE((target("sse4.2"))) +static double euclidean_sse(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + __m128 sum = _mm_setzero_ps(); + uint32_t i = 0; + for (; i + 4 <= dims; i += 4) { + __m128 d = _mm_sub_ps(_mm_loadu_ps(a + i), _mm_loadu_ps(b + i)); + sum = _mm_add_ps(sum, _mm_mul_ps(d, d)); + } + double result = hsum128_sse(sum); + if (!std::isfinite(result)) return euclidean_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + const double d = av - bv; + result += d * d; + } + return result; +} + +MY_ATTRIBUTE((target("sse4.2"))) +static double cosine_sse(const char *a_raw, const char *b_raw, uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + __m128 vec_ab = _mm_setzero_ps(); + __m128 vec_norm_a = _mm_setzero_ps(); + __m128 vec_norm_b = _mm_setzero_ps(); + uint32_t i = 0; + for (; i + 4 <= dims; i += 4) { + __m128 vec_a = _mm_loadu_ps(a + i); + __m128 vec_b = _mm_loadu_ps(b + i); + vec_ab = _mm_add_ps(vec_ab, _mm_mul_ps(vec_a, vec_b)); + vec_norm_a = _mm_add_ps(vec_norm_a, _mm_mul_ps(vec_a, vec_a)); + vec_norm_b = _mm_add_ps(vec_norm_b, _mm_mul_ps(vec_b, vec_b)); + } + double ab = hsum128_sse(vec_ab); + double norm_a = hsum128_sse(vec_norm_a); + double norm_b = hsum128_sse(vec_norm_b); + if (!std::isfinite(ab) || !std::isfinite(norm_a) || !std::isfinite(norm_b)) + return cosine_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + ab += (double)av * bv; + norm_a += (double)av * av; + norm_b += (double)bv * bv; + } + const double denom = sqrt(norm_a * norm_b); + if (denom == 0.0) return std::numeric_limits::infinity(); + return 1.0 - ab / denom; +} + +MY_ATTRIBUTE((target("sse4.2"))) +static double dot_product_sse(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + __m128 vec_ab = _mm_setzero_ps(); + uint32_t i = 0; + for (; i + 4 <= dims; i += 4) + vec_ab = _mm_add_ps(vec_ab, + _mm_mul_ps(_mm_loadu_ps(a + i), _mm_loadu_ps(b + i))); + double ab = hsum128_sse(vec_ab); + if (!std::isfinite(ab)) return dot_product_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + ab += (double)av * bv; + } + return ab; +} + +MY_ATTRIBUTE((target("sse4.2"))) +static double manhattan_sse(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + // SSE has no float abs intrinsic. -0.0f is a float with only the sign bit + // set, so andnot(sign_mask, d) clears each lane's sign bit: fabs(d) per lane. + const __m128 sign_mask = _mm_set1_ps(-0.0f); + __m128 sum = _mm_setzero_ps(); + uint32_t i = 0; + for (; i + 4 <= dims; i += 4) { + __m128 d = _mm_sub_ps(_mm_loadu_ps(a + i), _mm_loadu_ps(b + i)); + sum = _mm_add_ps(sum, _mm_andnot_ps(sign_mask, d)); + } + double result = hsum128_sse(sum); + if (!std::isfinite(result)) return manhattan_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + result += std::fabs((double)av - bv); + } + return result; +} + +// Tier 2 — AVX2 + FMA, 8 floats per iteration ------------------------------- + +// Horizontal sum: reduce the 8 float lanes of an AVX2 register to one float +// (AVX2 counterpart of the AVX-512 _mm512_reduce_add_ps intrinsic). +// Lambdas don't inherit a function's target attribute in GCC; use a static +// helper so _mm_hadd_ps and friends are compiled with the AVX2 ISA. +MY_ATTRIBUTE((target("avx2"))) +static float hsum256(__m256 v) { + __m128 lo = _mm256_castps256_ps128(v); + __m128 hi = _mm256_extractf128_ps(v, 1); + __m128 s = _mm_add_ps(lo, hi); + s = _mm_hadd_ps(s, s); + s = _mm_hadd_ps(s, s); + return _mm_cvtss_f32(s); +} + +MY_ATTRIBUTE((target("avx2,fma"))) +static double euclidean_avx2(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + __m256 sum = _mm256_setzero_ps(); + uint32_t i = 0; + for (; i + 8 <= dims; i += 8) { + __m256 d = _mm256_sub_ps(_mm256_loadu_ps(a + i), _mm256_loadu_ps(b + i)); + sum = _mm256_fmadd_ps(d, d, sum); + } + double result = hsum256(sum); + if (!std::isfinite(result)) return euclidean_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + const double d = av - bv; + result += d * d; + } + return result; +} + +MY_ATTRIBUTE((target("avx2,fma"))) +static double cosine_avx2(const char *a_raw, const char *b_raw, uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + __m256 vec_ab = _mm256_setzero_ps(); + __m256 vec_norm_a = _mm256_setzero_ps(); + __m256 vec_norm_b = _mm256_setzero_ps(); + uint32_t i = 0; + for (; i + 8 <= dims; i += 8) { + __m256 vec_a = _mm256_loadu_ps(a + i); + __m256 vec_b = _mm256_loadu_ps(b + i); + vec_ab = _mm256_fmadd_ps(vec_a, vec_b, vec_ab); + vec_norm_a = _mm256_fmadd_ps(vec_a, vec_a, vec_norm_a); + vec_norm_b = _mm256_fmadd_ps(vec_b, vec_b, vec_norm_b); + } + double ab = hsum256(vec_ab); + double norm_a = hsum256(vec_norm_a); + double norm_b = hsum256(vec_norm_b); + if (!std::isfinite(ab) || !std::isfinite(norm_a) || !std::isfinite(norm_b)) + return cosine_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + ab += (double)av * bv; + norm_a += (double)av * av; + norm_b += (double)bv * bv; + } + const double denom = sqrt(norm_a * norm_b); + if (denom == 0.0) return std::numeric_limits::infinity(); + return 1.0 - ab / denom; +} + +MY_ATTRIBUTE((target("avx2,fma"))) +static double dot_product_avx2(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + __m256 vec_ab = _mm256_setzero_ps(); + uint32_t i = 0; + for (; i + 8 <= dims; i += 8) + vec_ab = + _mm256_fmadd_ps(_mm256_loadu_ps(a + i), _mm256_loadu_ps(b + i), vec_ab); + double ab = hsum256(vec_ab); + if (!std::isfinite(ab)) return dot_product_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + ab += (double)av * bv; + } + return ab; +} + +MY_ATTRIBUTE((target("avx2"))) +static double manhattan_avx2(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + // AVX2 has no float abs intrinsic. -0.0f is a float with only the sign bit + // set, so andnot(sign_mask, d) clears each lane's sign bit: fabs(d) per lane. + const __m256 sign_mask = _mm256_set1_ps(-0.0f); + __m256 sum = _mm256_setzero_ps(); + uint32_t i = 0; + for (; i + 8 <= dims; i += 8) { + __m256 d = _mm256_sub_ps(_mm256_loadu_ps(a + i), _mm256_loadu_ps(b + i)); + sum = _mm256_add_ps(sum, _mm256_andnot_ps(sign_mask, d)); + } + double result = hsum256(sum); + if (!std::isfinite(result)) return manhattan_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + result += std::fabs((double)av - bv); + } + return result; +} + +// Tier 3 — AVX-512F, 16 floats per iteration -------------------------------- + +MY_ATTRIBUTE((target("avx512f"))) +static double euclidean_avx512(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + __m512 sum = _mm512_setzero_ps(); + uint32_t i = 0; + for (; i + 16 <= dims; i += 16) { + __m512 d = _mm512_sub_ps(_mm512_loadu_ps(a + i), _mm512_loadu_ps(b + i)); + sum = _mm512_fmadd_ps(d, d, sum); + } + double result = _mm512_reduce_add_ps(sum); + if (!std::isfinite(result)) return euclidean_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + const double d = av - bv; + result += d * d; + } + return result; +} + +MY_ATTRIBUTE((target("avx512f"))) +static double cosine_avx512(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + __m512 vec_ab = _mm512_setzero_ps(); + __m512 vec_norm_a = _mm512_setzero_ps(); + __m512 vec_norm_b = _mm512_setzero_ps(); + uint32_t i = 0; + for (; i + 16 <= dims; i += 16) { + __m512 vec_a = _mm512_loadu_ps(a + i); + __m512 vec_b = _mm512_loadu_ps(b + i); + vec_ab = _mm512_fmadd_ps(vec_a, vec_b, vec_ab); + vec_norm_a = _mm512_fmadd_ps(vec_a, vec_a, vec_norm_a); + vec_norm_b = _mm512_fmadd_ps(vec_b, vec_b, vec_norm_b); + } + double ab = _mm512_reduce_add_ps(vec_ab); + double norm_a = _mm512_reduce_add_ps(vec_norm_a); + double norm_b = _mm512_reduce_add_ps(vec_norm_b); + if (!std::isfinite(ab) || !std::isfinite(norm_a) || !std::isfinite(norm_b)) + return cosine_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + ab += (double)av * bv; + norm_a += (double)av * av; + norm_b += (double)bv * bv; + } + const double denom = sqrt(norm_a * norm_b); + if (denom == 0.0) return std::numeric_limits::infinity(); + return 1.0 - ab / denom; +} + +MY_ATTRIBUTE((target("avx512f"))) +static double dot_product_avx512(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + __m512 vec_ab = _mm512_setzero_ps(); + uint32_t i = 0; + for (; i + 16 <= dims; i += 16) + vec_ab = + _mm512_fmadd_ps(_mm512_loadu_ps(a + i), _mm512_loadu_ps(b + i), vec_ab); + double ab = _mm512_reduce_add_ps(vec_ab); + if (!std::isfinite(ab)) return dot_product_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + ab += (double)av * bv; + } + return ab; +} + +MY_ATTRIBUTE((target("avx512f"))) +static double manhattan_avx512(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + __m512 sum = _mm512_setzero_ps(); + uint32_t i = 0; + for (; i + 16 <= dims; i += 16) { + __m512 d = _mm512_sub_ps(_mm512_loadu_ps(a + i), _mm512_loadu_ps(b + i)); + sum = _mm512_add_ps(sum, _mm512_abs_ps(d)); + } + double result = _mm512_reduce_add_ps(sum); + if (!std::isfinite(result)) return manhattan_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + result += std::fabs((double)av - bv); + } + return result; +} + +#endif // VECTOR_DISTANCE_x86_64 + +#ifdef VECTOR_DISTANCE_AARCH64 + +#include + +// Tier 1 — NEON (Advanced SIMD), 4 floats per iteration --------------------- +// Fused multiply-add (vfmaq_f32 -> FMLA) is baseline ARMv8-A, so unlike the +// x86 SSE4.2 tier no separate feature gate is needed; single rounding matches +// the AVX2/AVX-512/SVE2 tiers. + +// Horizontal sum: reduce the 4 float lanes of a NEON register to one float +// (NEON counterpart of the AVX-512 _mm512_reduce_add_ps intrinsic). +// Same lambda issue applies on NEON; extract as a static attributed helper. +MY_ATTRIBUTE((target("+simd"))) +static float hsum4(float32x4_t v) { + float32x2_t s = vadd_f32(vget_low_f32(v), vget_high_f32(v)); + s = vpadd_f32(s, s); + return vget_lane_f32(s, 0); +} + +MY_ATTRIBUTE((target("+simd"))) +static double euclidean_neon(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + float32x4_t sum = vdupq_n_f32(0.0f); + uint32_t i = 0; + for (; i + 4 <= dims; i += 4) { + float32x4_t d = vsubq_f32(vld1q_f32(a + i), vld1q_f32(b + i)); + sum = vfmaq_f32(sum, d, d); + } + double result = hsum4(sum); + if (!std::isfinite(result)) return euclidean_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + const double d = av - bv; + result += d * d; + } + return result; +} + +MY_ATTRIBUTE((target("+simd"))) +static double cosine_neon(const char *a_raw, const char *b_raw, uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + float32x4_t vec_ab = vdupq_n_f32(0.0f); + float32x4_t vec_norm_a = vdupq_n_f32(0.0f); + float32x4_t vec_norm_b = vdupq_n_f32(0.0f); + uint32_t i = 0; + for (; i + 4 <= dims; i += 4) { + float32x4_t vec_a = vld1q_f32(a + i); + float32x4_t vec_b = vld1q_f32(b + i); + vec_ab = vfmaq_f32(vec_ab, vec_a, vec_b); + vec_norm_a = vfmaq_f32(vec_norm_a, vec_a, vec_a); + vec_norm_b = vfmaq_f32(vec_norm_b, vec_b, vec_b); + } + double ab = hsum4(vec_ab); + double norm_a = hsum4(vec_norm_a); + double norm_b = hsum4(vec_norm_b); + if (!std::isfinite(ab) || !std::isfinite(norm_a) || !std::isfinite(norm_b)) + return cosine_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + ab += (double)av * bv; + norm_a += (double)av * av; + norm_b += (double)bv * bv; + } + const double denom = sqrt(norm_a * norm_b); + if (denom == 0.0) return std::numeric_limits::infinity(); + return 1.0 - ab / denom; +} + +MY_ATTRIBUTE((target("+simd"))) +static double dot_product_neon(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + float32x4_t vec_ab = vdupq_n_f32(0.0f); + uint32_t i = 0; + for (; i + 4 <= dims; i += 4) + vec_ab = vfmaq_f32(vec_ab, vld1q_f32(a + i), vld1q_f32(b + i)); + double ab = hsum4(vec_ab); + if (!std::isfinite(ab)) return dot_product_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + ab += (double)av * bv; + } + return ab; +} + +MY_ATTRIBUTE((target("+simd"))) +static double manhattan_neon(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + float32x4_t sum = vdupq_n_f32(0.0f); + uint32_t i = 0; + for (; i + 4 <= dims; i += 4) { + float32x4_t d = vsubq_f32(vld1q_f32(a + i), vld1q_f32(b + i)); + sum = vaddq_f32(sum, vabsq_f32(d)); + } + double result = hsum4(sum); + if (!std::isfinite(result)) return manhattan_scalar(a_raw, b_raw, dims); + for (; i < dims; i++) { + float av, bv; + memcpy(&av, a_raw + i * sizeof(float), sizeof(float)); + memcpy(&bv, b_raw + i * sizeof(float), sizeof(float)); + result += std::fabs((double)av - bv); + } + return result; +} + +#ifdef VECTOR_DISTANCE_HAS_SVE2 + +#include +#include + +// HWCAP2_SVE2 may not be exposed by older libc headers; the bit is stable +// in the Linux kernel UAPI (linux/include/uapi/asm-generic/hwcap.h). +#ifndef HWCAP2_SVE2 +#define HWCAP2_SVE2 (1UL << 1) +#endif + +// Tier 3 — SVE2, scalable (VLA) — predicated loads handle any alignment ----- + +static bool cpu_has_sve2() { +#if defined(__linux__) + return (getauxval(AT_HWCAP2) & HWCAP2_SVE2) != 0; +#else + return false; +#endif +} + +MY_ATTRIBUTE((target("+sve2"))) +static double euclidean_sve2(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + svfloat32_t sum = svdup_n_f32(0.0f); + uint32_t i = 0; + svbool_t pg = svwhilelt_b32_u32(i, dims); + while (svptest_first(svptrue_b32(), pg)) { + svfloat32_t vec_a = svld1_f32(pg, a + i); + svfloat32_t vec_b = svld1_f32(pg, b + i); + svfloat32_t d = svsub_f32_x(pg, vec_a, vec_b); + // Merging form keeps inactive lanes of sum unchanged on the tail. + sum = svmla_f32_m(pg, sum, d, d); + i += svcntw(); + pg = svwhilelt_b32_u32(i, dims); + } + const double result = svaddv_f32(svptrue_b32(), sum); + if (!std::isfinite(result)) return euclidean_scalar(a_raw, b_raw, dims); + return result; +} + +MY_ATTRIBUTE((target("+sve2"))) +static double cosine_sve2(const char *a_raw, const char *b_raw, uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + svfloat32_t vec_ab = svdup_n_f32(0.0f); + svfloat32_t vec_norm_a = svdup_n_f32(0.0f); + svfloat32_t vec_norm_b = svdup_n_f32(0.0f); + uint32_t i = 0; + svbool_t pg = svwhilelt_b32_u32(i, dims); + while (svptest_first(svptrue_b32(), pg)) { + svfloat32_t vec_a = svld1_f32(pg, a + i); + svfloat32_t vec_b = svld1_f32(pg, b + i); + vec_ab = svmla_f32_m(pg, vec_ab, vec_a, vec_b); + vec_norm_a = svmla_f32_m(pg, vec_norm_a, vec_a, vec_a); + vec_norm_b = svmla_f32_m(pg, vec_norm_b, vec_b, vec_b); + i += svcntw(); + pg = svwhilelt_b32_u32(i, dims); + } + const double ab = svaddv_f32(svptrue_b32(), vec_ab); + const double norm_a = svaddv_f32(svptrue_b32(), vec_norm_a); + const double norm_b = svaddv_f32(svptrue_b32(), vec_norm_b); + if (!std::isfinite(ab) || !std::isfinite(norm_a) || !std::isfinite(norm_b)) + return cosine_scalar(a_raw, b_raw, dims); + const double denom = sqrt(norm_a * norm_b); + if (denom == 0.0) return std::numeric_limits::infinity(); + return 1.0 - ab / denom; +} + +MY_ATTRIBUTE((target("+sve2"))) +static double dot_product_sve2(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + svfloat32_t vec_ab = svdup_n_f32(0.0f); + uint32_t i = 0; + svbool_t pg = svwhilelt_b32_u32(i, dims); + while (svptest_first(svptrue_b32(), pg)) { + svfloat32_t vec_a = svld1_f32(pg, a + i); + svfloat32_t vec_b = svld1_f32(pg, b + i); + vec_ab = svmla_f32_m(pg, vec_ab, vec_a, vec_b); + i += svcntw(); + pg = svwhilelt_b32_u32(i, dims); + } + const double ab = svaddv_f32(svptrue_b32(), vec_ab); + if (!std::isfinite(ab)) return dot_product_scalar(a_raw, b_raw, dims); + return ab; +} + +MY_ATTRIBUTE((target("+sve2"))) +static double manhattan_sve2(const char *a_raw, const char *b_raw, + uint32_t dims) { + const float *a = reinterpret_cast(a_raw); + const float *b = reinterpret_cast(b_raw); + svfloat32_t sum = svdup_n_f32(0.0f); + uint32_t i = 0; + svbool_t pg = svwhilelt_b32_u32(i, dims); + while (svptest_first(svptrue_b32(), pg)) { + svfloat32_t vec_a = svld1_f32(pg, a + i); + svfloat32_t vec_b = svld1_f32(pg, b + i); + svfloat32_t d = svsub_f32_x(pg, vec_a, vec_b); + sum = svadd_f32_m(pg, sum, svabs_f32_x(pg, d)); + i += svcntw(); + pg = svwhilelt_b32_u32(i, dims); + } + const double result = svaddv_f32(svptrue_b32(), sum); + if (!std::isfinite(result)) return manhattan_scalar(a_raw, b_raw, dims); + return result; +} + +#endif // VECTOR_DISTANCE_HAS_SVE2 + +#endif // VECTOR_DISTANCE_AARCH64 + +// Tier dispatch helpers — must follow all kernel definitions above. + +static void apply_wide_tier(VectorDistanceTier tier) { + g_wide_tier = tier; + switch (tier) { + case VectorDistanceTier::Scalar: + g_euclidean = euclidean_scalar; + g_cosine = cosine_scalar; + g_dot_product = dot_product_scalar; + g_manhattan = manhattan_scalar; + break; +#ifdef VECTOR_DISTANCE_x86_64 + case VectorDistanceTier::Sse42: + g_euclidean = euclidean_sse; + g_cosine = cosine_sse; + g_dot_product = dot_product_sse; + g_manhattan = manhattan_sse; + break; + case VectorDistanceTier::Avx2: + g_euclidean = euclidean_avx2; + g_cosine = cosine_avx2; + g_dot_product = dot_product_avx2; + g_manhattan = manhattan_avx2; + break; + case VectorDistanceTier::Avx512f: + g_euclidean = euclidean_avx512; + g_cosine = cosine_avx512; + g_dot_product = dot_product_avx512; + g_manhattan = manhattan_avx512; + break; +#endif +#ifdef VECTOR_DISTANCE_AARCH64 + case VectorDistanceTier::Neon: + g_euclidean = euclidean_neon; + g_cosine = cosine_neon; + g_dot_product = dot_product_neon; + g_manhattan = manhattan_neon; + break; +#ifdef VECTOR_DISTANCE_HAS_SVE2 + case VectorDistanceTier::Sve2: + g_euclidean = euclidean_sve2; + g_cosine = cosine_sve2; + g_dot_product = dot_product_sve2; + g_manhattan = manhattan_sve2; + break; +#endif +#endif + default: + break; + } +} + +static void apply_narrow_tier(VectorDistanceTier tier) { + g_narrow_tier = tier; + switch (tier) { + case VectorDistanceTier::Scalar: + g_euclidean_narrow = euclidean_scalar; + g_cosine_narrow = cosine_scalar; + g_dot_product_narrow = dot_product_scalar; + g_manhattan_narrow = manhattan_scalar; + break; +#ifdef VECTOR_DISTANCE_x86_64 + case VectorDistanceTier::Sse42: + g_euclidean_narrow = euclidean_sse; + g_cosine_narrow = cosine_sse; + g_dot_product_narrow = dot_product_sse; + g_manhattan_narrow = manhattan_sse; + break; +#endif +#ifdef VECTOR_DISTANCE_AARCH64 + case VectorDistanceTier::Neon: + g_euclidean_narrow = euclidean_neon; + g_cosine_narrow = cosine_neon; + g_dot_product_narrow = dot_product_neon; + g_manhattan_narrow = manhattan_neon; + break; +#ifdef VECTOR_DISTANCE_HAS_SVE2 + case VectorDistanceTier::Sve2: + g_euclidean_narrow = euclidean_sve2; + g_cosine_narrow = cosine_sve2; + g_dot_product_narrow = dot_product_sve2; + g_manhattan_narrow = manhattan_sve2; + break; +#endif +#endif + default: + break; + } +} + +// init_vector_distance_functions — promote g_* and g_*_narrow (see file header) + +void init_vector_distance_functions() { + apply_wide_tier(VectorDistanceTier::Scalar); + apply_narrow_tier(VectorDistanceTier::Scalar); +#ifdef VECTOR_DISTANCE_x86_64 + // x86_64: Tier 3 -> Tier 2 -> Tier 1 -> Tier 0 (wide) + if (cpu_has_avx512f()) { + apply_wide_tier(VectorDistanceTier::Avx512f); + } else if (cpu_has_avx2_fma()) { + apply_wide_tier(VectorDistanceTier::Avx2); + } else if (cpu_has_sse42()) { + apply_wide_tier(VectorDistanceTier::Sse42); + } + // Narrow path: SSE4.2 fills at dim ≥ 4; use it when available. + if (cpu_has_sse42()) { + apply_narrow_tier(VectorDistanceTier::Sse42); + } +#endif +#ifdef VECTOR_DISTANCE_AARCH64 + // Tier 1 (NEON), optionally Tier 3 (SVE2) + apply_wide_tier(VectorDistanceTier::Neon); + apply_narrow_tier(VectorDistanceTier::Neon); +#ifdef VECTOR_DISTANCE_HAS_SVE2 + if (cpu_has_sve2()) { // Tier 3 over Tier 1 (wide only) + apply_wide_tier(VectorDistanceTier::Sve2); + } +#endif +#endif +} + +bool vector_distance_tier_available(VectorDistanceTier tier) { + switch (tier) { + case VectorDistanceTier::Scalar: + return true; +#ifdef VECTOR_DISTANCE_x86_64 + case VectorDistanceTier::Sse42: + return cpu_has_sse42(); + case VectorDistanceTier::Avx2: + return cpu_has_avx2_fma(); + case VectorDistanceTier::Avx512f: + return cpu_has_avx512f(); +#endif +#ifdef VECTOR_DISTANCE_AARCH64 + case VectorDistanceTier::Neon: + return true; // mandatory on ARMv8 + case VectorDistanceTier::Sve2: +#ifdef VECTOR_DISTANCE_HAS_SVE2 + return cpu_has_sve2(); +#else + return false; +#endif +#endif + default: + return false; + } +} + +void init_vector_distance_functions_tier(VectorDistanceTier tier) { + switch (tier) { + case VectorDistanceTier::Scalar: + apply_wide_tier(VectorDistanceTier::Scalar); + apply_narrow_tier(VectorDistanceTier::Scalar); + break; +#ifdef VECTOR_DISTANCE_x86_64 + case VectorDistanceTier::Sse42: + apply_wide_tier(VectorDistanceTier::Sse42); + apply_narrow_tier(VectorDistanceTier::Sse42); + break; + case VectorDistanceTier::Avx2: + apply_wide_tier(VectorDistanceTier::Avx2); + if (cpu_has_sse42()) apply_narrow_tier(VectorDistanceTier::Sse42); + break; + case VectorDistanceTier::Avx512f: + apply_wide_tier(VectorDistanceTier::Avx512f); + if (cpu_has_sse42()) apply_narrow_tier(VectorDistanceTier::Sse42); + break; +#endif +#ifdef VECTOR_DISTANCE_AARCH64 + case VectorDistanceTier::Neon: + apply_wide_tier(VectorDistanceTier::Neon); + apply_narrow_tier(VectorDistanceTier::Neon); + break; +#ifdef VECTOR_DISTANCE_HAS_SVE2 + case VectorDistanceTier::Sve2: + apply_wide_tier(VectorDistanceTier::Sve2); + apply_narrow_tier(VectorDistanceTier::Neon); + break; +#endif +#endif + default: + // Unsupported tier on this build; callers must check + // vector_distance_tier_available() first. + assert(false); + break; + } +} + +VectorDistanceTier vector_distance_wide_tier() { return g_wide_tier; } + +VectorDistanceTier vector_distance_narrow_tier() { return g_narrow_tier; } + +const char *vector_distance_tier_label(VectorDistanceTier tier) { + switch (tier) { + case VectorDistanceTier::Scalar: + return "software scalar"; + case VectorDistanceTier::Sse42: + return "SSE4.2"; + case VectorDistanceTier::Avx2: + return "AVX2 and FMA"; + case VectorDistanceTier::Avx512f: + return "AVX-512F"; + case VectorDistanceTier::Neon: + return "NEON"; + case VectorDistanceTier::Sve2: + return "SVE2"; + } + return "unknown"; +} + +size_t vector_distance_dispatch_description(char *buf, size_t buf_len) { + if (buf == nullptr || buf_len == 0) return 0; + + const VectorDistanceTier wide = g_wide_tier; + const VectorDistanceTier narrow = g_narrow_tier; + int n; + if (wide == narrow) { + if (wide == VectorDistanceTier::Scalar) { + n = snprintf(buf, buf_len, " software scalar."); + } else { + n = snprintf(buf, buf_len, " hardware accelerated %s.", + vector_distance_tier_label(wide)); + } + } else { + n = snprintf( + buf, buf_len, + " hardware accelerated %s " + "(dimensions >= %u) and %s (dimensions < %u).", + vector_distance_tier_label(wide), VECTOR_DISTANCE_WIDE_MIN_DIMS, + vector_distance_tier_label(narrow), VECTOR_DISTANCE_WIDE_MIN_DIMS); + } + if (n < 0) { + buf[0] = '\0'; + return 0; + } + if (static_cast(n) >= buf_len) return buf_len - 1; + return static_cast(n); +} + +// Public wrappers — dim-aware dispatch: narrow path for dims < +// VECTOR_DISTANCE_WIDE_MIN_DIMS avoids sending small inputs to AVX-512/AVX2 +// where the SIMD loop cannot fire. + +double vector_distance_euclidean_squared(const char *a, const char *b, + uint32_t dims) { + return (dims < VECTOR_DISTANCE_WIDE_MIN_DIMS ? g_euclidean_narrow + : g_euclidean)(a, b, dims); +} + +double vector_distance_cosine(const char *a, const char *b, uint32_t dims) { + return (dims < VECTOR_DISTANCE_WIDE_MIN_DIMS ? g_cosine_narrow : g_cosine)( + a, b, dims); +} + +double vector_distance_dot(const char *a, const char *b, uint32_t dims) { + return (dims < VECTOR_DISTANCE_WIDE_MIN_DIMS ? g_dot_product_narrow + : g_dot_product)(a, b, dims); +} + +double vector_distance_manhattan(const char *a, const char *b, uint32_t dims) { + return (dims < VECTOR_DISTANCE_WIDE_MIN_DIMS ? g_manhattan_narrow + : g_manhattan)(a, b, dims); +} diff --git a/vector-common/vector_distance.h b/vector-common/vector_distance.h new file mode 100644 index 000000000000..9e46571f12b4 --- /dev/null +++ b/vector-common/vector_distance.h @@ -0,0 +1,115 @@ +/* Copyright (c) 2026, Percona and/or its affiliates. + + This program is free software; you can redistribute it and/or modify + it under the terms of the GNU General Public License, version 2.0, + as published by the Free Software Foundation. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License, version 2.0, for more details. + + You should have received a copy of the GNU General Public License + along with this program; if not, write to the Free Software + Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA */ + +#pragma once + +#include +#include + +/** Set SIMD function pointers before first use; safe to call repeatedly. */ +void init_vector_distance_functions(); + +/** + SIMD tier identifiers. Values match the tier table in vector_distance.cc. + Tiers that do not apply to the host platform (e.g. SSE42 on aarch64) are + always reported as unavailable by vector_distance_tier_available(). +*/ +enum class VectorDistanceTier { + Scalar = 0, ///< Tier 0 — plain C++ scalar + Sse42, ///< Tier 1 x86_64 — SSE4.2, 128-bit + Avx2, ///< Tier 2 x86_64 — AVX2 + FMA, 256-bit + Avx512f, ///< Tier 3 x86_64 — AVX-512F, 512-bit + Neon, ///< Tier 1 aarch64 — NEON, 128-bit + Sve2, ///< Tier 3 aarch64 — SVE2, scalable (VLA) +}; + +/** + Return true when @p tier can be used on the current CPU and build. + Scalar always returns true. Architecture-specific tiers return false on the + wrong platform. SVE2 additionally requires the binary to have been compiled + with __ARM_FEATURE_SVE2 and the OS kernel to advertise HWCAP2_SVE2. +*/ +bool vector_distance_tier_available(VectorDistanceTier tier); + +/** + Set all dispatch pointers (wide and narrow) directly to @p tier without + call_once protection. Intended for benchmarks that want to force a specific + tier on a host that may support a higher one. Use + VectorDistanceTier::Scalar to force scalar kernels. Callers must verify + vector_distance_tier_available() first. +*/ +void init_vector_distance_functions_tier(VectorDistanceTier tier); + +/** Minimum dims for the wide SIMD kernel; below this, g_*_narrow is used. */ +static constexpr uint32_t VECTOR_DISTANCE_WIDE_MIN_DIMS = 16; + +/** Wide-path tier selected by init_vector_distance_functions() (dims >= + * VECTOR_DISTANCE_WIDE_MIN_DIMS). */ +VectorDistanceTier vector_distance_wide_tier(); + +/** Narrow-path tier selected by init_vector_distance_functions() (dims < + * VECTOR_DISTANCE_WIDE_MIN_DIMS). */ +VectorDistanceTier vector_distance_narrow_tier(); + +/** + Stable English label for logs/tests, e.g. "AVX-512F", "AVX2 and FMA", + "software scalar". +*/ +const char *vector_distance_tier_label(VectorDistanceTier tier); + +/** + Write a human-readable dispatch summary into @p buf (NUL-terminated). + Returns the number of bytes written, excluding the terminating NUL. +*/ +size_t vector_distance_dispatch_description(char *buf, size_t buf_len); + +/** + Compute squared Euclidean distance: sum((a[i]-b[i])²). + No sqrt — intended for ranking and as the shared kernel for SQL EUCLIDEAN + (where sqrt is applied in Item_func_vector_distance::val_real). + Accepts any byte alignment; SIMD kernels use unaligned loads internally. + Returns double; SIMD accumulates in float32, reduction uses double for + precision on large dims and extreme float32 values. +*/ +double vector_distance_euclidean_squared(const char *a, const char *b, + uint32_t dims); + +/** + Compute cosine distance between two float vectors encoded as raw bytes. + Returns +Inf when either vector is all-zeros (undefined cosine — true + zero-denominator); returns NaN when input elements are NaN/Inf (bad-data + propagation). The caller must distinguish these two cases. + Returns double; SIMD accumulates in float32, reduction uses double for + precision on large dims and extreme float32 values. +*/ +double vector_distance_cosine(const char *a, const char *b, uint32_t dims); + +/** + Compute dot product (inner product) between two float vectors encoded as raw + bytes. Returns sum(a[i]*b[i]). Higher values indicate greater similarity. + Always finite for finite inputs — no NaN edge cases. + Returns double; SIMD accumulates in float32, reduction uses double for + precision on large dims and extreme float32 values. +*/ +double vector_distance_dot(const char *a, const char *b, uint32_t dims); + +/** + Compute Manhattan (L1) distance between two float vectors encoded as raw + bytes. Returns sum(|a[i] - b[i]|). Always >= 0 for finite inputs. No NaN + edge cases. + Returns double; SIMD accumulates in float32, reduction uses double for + precision on large dims and extreme float32 values. +*/ +double vector_distance_manhattan(const char *a, const char *b, uint32_t dims); From 47f605c826f3546b880ed2d661e66fda5a7ba3d0 Mon Sep 17 00:00:00 2001 From: Oleksiy Lukin Date: Tue, 7 Jul 2026 12:42:10 +0300 Subject: [PATCH 11/32] PS-10102 [8.4] fix of crash when KMIP server is unreachable https://perconadev.atlassian.net/browse/PS-10102 When component_keyring_kmip is loaded but keyring initialization fails (e.g. KMIP server not running or misconfigured), g_keyring_operations remains a null unique_ptr. Any subsequent call to a keyring service method (generate, store, remove, read, encrypt, decrypt, iterator) dereferences this null pointer, triggering an assertion in unique_ptr::operator*() and aborting the server. Add a null guard at the start of every service method that dereferences g_keyring_operations, returning true (error) immediately when the keyring has not been successfully initialized. --- .../keyring_encryption_service_definition.cc | 2 ++ .../keyring_generator_service_definition.cc | 1 + .../keyring_keys_metadata_iterator_service_definition.cc | 6 ++++++ .../keyring_reader_service_definition.cc | 4 ++++ .../keyring_writer_service_definition.cc | 2 ++ 5 files changed, 15 insertions(+) diff --git a/components/keyrings/keyring_kmip/service_implementation/keyring_encryption_service_definition.cc b/components/keyrings/keyring_kmip/service_implementation/keyring_encryption_service_definition.cc index 089d1974f91f..33947b90bf6c 100644 --- a/components/keyrings/keyring_kmip/service_implementation/keyring_encryption_service_definition.cc +++ b/components/keyrings/keyring_kmip/service_implementation/keyring_encryption_service_definition.cc @@ -49,6 +49,7 @@ DEFINE_BOOL_METHOD(Keyring_aes_service_impl::encrypt, const unsigned char *data_buffer, size_t data_buffer_length, unsigned char *out_buffer, size_t out_buffer_length, size_t *out_length)) { + if (!g_keyring_operations) return true; return aes_encrypt_template< Keyring_kmip_backend, keyring_common::data::Data_extension>( @@ -63,6 +64,7 @@ DEFINE_BOOL_METHOD(Keyring_aes_service_impl::decrypt, const unsigned char *data_buffer, size_t data_buffer_length, unsigned char *out_buffer, size_t out_buffer_length, size_t *out_length)) { + if (!g_keyring_operations) return true; return aes_decrypt_template< Keyring_kmip_backend, keyring_common::data::Data_extension>( diff --git a/components/keyrings/keyring_kmip/service_implementation/keyring_generator_service_definition.cc b/components/keyrings/keyring_kmip/service_implementation/keyring_generator_service_definition.cc index 5a07ad18f482..7420fc3e5887 100644 --- a/components/keyrings/keyring_kmip/service_implementation/keyring_generator_service_definition.cc +++ b/components/keyrings/keyring_kmip/service_implementation/keyring_generator_service_definition.cc @@ -38,6 +38,7 @@ namespace service_definition { DEFINE_BOOL_METHOD(Keyring_generator_service_impl::generate, (const char *data_id, const char *auth_id, const char *data_type, size_t data_size)) { + if (!g_keyring_operations) return true; return generate_template( data_id, auth_id, data_type, data_size, *g_keyring_operations, *g_component_callbacks); diff --git a/components/keyrings/keyring_kmip/service_implementation/keyring_keys_metadata_iterator_service_definition.cc b/components/keyrings/keyring_kmip/service_implementation/keyring_keys_metadata_iterator_service_definition.cc index ebc19834f633..9e407310d22b 100644 --- a/components/keyrings/keyring_kmip/service_implementation/keyring_keys_metadata_iterator_service_definition.cc +++ b/components/keyrings/keyring_kmip/service_implementation/keyring_keys_metadata_iterator_service_definition.cc @@ -46,6 +46,7 @@ using keyring_kmip::IdExt; namespace service_definition { DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::init, (my_h_keyring_keys_metadata_iterator * forward_iterator)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; bool retval = init_keys_metadata_iterator_template( it, *g_keyring_operations, *g_component_callbacks); @@ -57,6 +58,7 @@ DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::init, DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::deinit, (my_h_keyring_keys_metadata_iterator forward_iterator)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset( reinterpret_cast> *>(forward_iterator)); @@ -66,6 +68,7 @@ DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::deinit, DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::is_valid, (my_h_keyring_keys_metadata_iterator forward_iterator)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset( reinterpret_cast> *>(forward_iterator)); @@ -78,6 +81,7 @@ DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::is_valid, DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::next, (my_h_keyring_keys_metadata_iterator forward_iterator)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset( reinterpret_cast> *>(forward_iterator)); @@ -91,6 +95,7 @@ DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::next, DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::get_length, (my_h_keyring_keys_metadata_iterator forward_iterator, size_t *data_id_length, size_t *auth_id_length)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset( reinterpret_cast> *>(forward_iterator)); @@ -106,6 +111,7 @@ DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::get, (my_h_keyring_keys_metadata_iterator forward_iterator, char *data_id, size_t data_id_length, char *auth_id, size_t auth_id_length)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset( reinterpret_cast> *>(forward_iterator)); diff --git a/components/keyrings/keyring_kmip/service_implementation/keyring_reader_service_definition.cc b/components/keyrings/keyring_kmip/service_implementation/keyring_reader_service_definition.cc index e74c22f35dd4..ebdd168f1341 100644 --- a/components/keyrings/keyring_kmip/service_implementation/keyring_reader_service_definition.cc +++ b/components/keyrings/keyring_kmip/service_implementation/keyring_reader_service_definition.cc @@ -44,6 +44,7 @@ namespace service_definition { DEFINE_BOOL_METHOD(Keyring_reader_service_impl::init, (const char *data_id, const char *auth_id, my_h_keyring_reader_object *reader_object)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; int retval = init_reader_template( data_id, auth_id, it, *g_keyring_operations, *g_component_callbacks); @@ -55,6 +56,7 @@ DEFINE_BOOL_METHOD(Keyring_reader_service_impl::init, DEFINE_BOOL_METHOD(Keyring_reader_service_impl::deinit, (my_h_keyring_reader_object reader_object)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset(reinterpret_cast> *>(reader_object)); return deinit_reader_template(it, *g_keyring_operations, @@ -64,6 +66,7 @@ DEFINE_BOOL_METHOD(Keyring_reader_service_impl::deinit, DEFINE_BOOL_METHOD(Keyring_reader_service_impl::fetch_length, (my_h_keyring_reader_object reader_object, size_t *data_size, size_t *data_type_size)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset(reinterpret_cast> *>(reader_object)); bool retval = fetch_length_template( @@ -79,6 +82,7 @@ DEFINE_BOOL_METHOD(Keyring_reader_service_impl::fetch, unsigned char *data_buffer, size_t data_buffer_length, size_t *data_size, char *data_type_buffer, size_t data_type_buffer_length, size_t *data_type_size)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset(reinterpret_cast> *>(reader_object)); bool retval = fetch_template( diff --git a/components/keyrings/keyring_kmip/service_implementation/keyring_writer_service_definition.cc b/components/keyrings/keyring_kmip/service_implementation/keyring_writer_service_definition.cc index 56bd409ed07b..0e2b73c985e1 100644 --- a/components/keyrings/keyring_kmip/service_implementation/keyring_writer_service_definition.cc +++ b/components/keyrings/keyring_kmip/service_implementation/keyring_writer_service_definition.cc @@ -39,6 +39,7 @@ DEFINE_BOOL_METHOD(Keyring_writer_service_impl::store, (const char *data_id, const char *auth_id, const unsigned char *data, size_t data_size, const char *data_type)) { + if (!g_keyring_operations) return true; return store_template(data_id, auth_id, data, data_size, data_type, *g_keyring_operations, *g_component_callbacks); @@ -46,6 +47,7 @@ DEFINE_BOOL_METHOD(Keyring_writer_service_impl::store, DEFINE_BOOL_METHOD(Keyring_writer_service_impl::remove, (const char *data_id, const char *auth_id)) { + if (!g_keyring_operations) return true; return remove_template( data_id, auth_id, *g_keyring_operations, *g_component_callbacks); } From e8ccf541c2e14c6395826053be341f9edaae26c7 Mon Sep 17 00:00:00 2001 From: Oleksiy Lukin Date: Thu, 9 Jul 2026 14:33:18 +0300 Subject: [PATCH 12/32] PS-10102 [8.4] fix of component_keyring_kmip MTR tests https://perconadev.atlassian.net/browse/PS-10102 Various fixes of component_keyring_kmip MTR tests to run successfully on PyKMIP and CosmianKMS servers. Some tests are excluded from parallel run. The CosmianKMS KMIP server now is main for testing, but PyKMIP server is still supported --- .../keyring_tests/innodb/table_encrypt_3.inc | 2 +- .../inc/keyring_udf_test_kmip.inc | 17 ++++-- .../inc/setup_component.inc | 35 ++++++++++-- .../inc/setup_component_customized.inc | 22 ++++++-- .../inc/teardown_component.inc | 14 ++++- .../inc/teardown_component_customized.inc | 16 ++++++ .../t/clone_remote_encrypt.test | 1 + .../t/encrypt_explicit.result | 4 +- .../t/encrypt_explicit.test | 1 - .../t/keyring_component_status.result | 4 ++ .../t/keyring_component_status.test | 31 ++++++++++- .../t/log_encrypt_1.test | 3 +- .../t/log_encrypt_5.result | 6 +-- .../t/log_encrypt_6.result | 9 ++-- .../t/log_encrypt_kill.result | 12 ++--- .../t/mysql_ts_alter_encrypt_1.test | 1 + .../t/mysql_ts_alter_encrypt_2.result | 4 -- .../t/mysql_ts_alter_encrypt_2.test | 1 + .../t/table_encrypt_2.test | 1 + .../t/table_encrypt_3.result | 2 +- .../t/table_encrypt_3.test | 1 + .../t/table_encrypt_4.test | 1 + .../t/tablespace_encrypt_10.result | 54 ++++++++++++------- .../t/tablespace_encrypt_10.test | 1 + .../t/tablespace_encrypt_11.test | 1 + .../t/tablespace_encrypt_3.test | 1 + .../t/tablespace_encrypt_7.test | 1 + 27 files changed, 191 insertions(+), 55 deletions(-) diff --git a/mysql-test/include/keyring_tests/innodb/table_encrypt_3.inc b/mysql-test/include/keyring_tests/innodb/table_encrypt_3.inc index cd2c0bc84119..cf268ad4d7d1 100644 --- a/mysql-test/include/keyring_tests/innodb/table_encrypt_3.inc +++ b/mysql-test/include/keyring_tests/innodb/table_encrypt_3.inc @@ -388,7 +388,7 @@ CREATE PROCEDURE tde_db.rotate_master_key() begin declare i int default 1; declare has_error int default 0; - while (i <= 500) DO + while (i <= 100) DO ALTER INSTANCE ROTATE INNODB MASTER KEY; set i = i + 1; end while; diff --git a/mysql-test/suite/component_keyring_kmip/inc/keyring_udf_test_kmip.inc b/mysql-test/suite/component_keyring_kmip/inc/keyring_udf_test_kmip.inc index 608fe298074b..d78ed61cc6c4 100644 --- a/mysql-test/suite/component_keyring_kmip/inc/keyring_udf_test_kmip.inc +++ b/mysql-test/suite/component_keyring_kmip/inc/keyring_udf_test_kmip.inc @@ -20,6 +20,20 @@ --echo # ---------------------------------------------------------------------- +--disable_result_log +--disable_query_log +--disable_abort_on_error +# Best-effort cleanup for strict KMIP servers that reject duplicate key names. +SELECT keyring_key_remove('AES_g1'); +SELECT keyring_key_remove('AES_g2'); +SELECT keyring_key_remove('AES_g3'); +SELECT keyring_key_remove('AES_s1'); +SELECT keyring_key_remove('AES_s2'); +SELECT keyring_key_remove('AES_s3'); +--enable_abort_on_error +--enable_query_log +--enable_result_log + --echo # Tests for AES key type --echo # keyring_key_generate tests SELECT keyring_key_generate('AES_g1', 'AES', 16); @@ -115,6 +129,3 @@ DROP FUNCTION keyring_key_type_fetch; DROP FUNCTION keyring_key_length_fetch; UNINSTALL PLUGIN keyring_udf; --echo # ---------------------------------------------------------------------- - - - diff --git a/mysql-test/suite/component_keyring_kmip/inc/setup_component.inc b/mysql-test/suite/component_keyring_kmip/inc/setup_component.inc index 5da074b5d224..56f4ec2b5a8d 100644 --- a/mysql-test/suite/component_keyring_kmip/inc/setup_component.inc +++ b/mysql-test/suite/component_keyring_kmip/inc/setup_component.inc @@ -10,6 +10,21 @@ --echo # ---------------------------------------------------------------------- --echo # Setup +# Ensure the standard binary manifest (read_local_manifest: true) is in place. +# A previous customized-setup test may have replaced or removed it; restoring +# it here guarantees that the instance manifest is always read at server start. +--perl + use strict; + use File::Basename; + my $content = '{ "read_local_manifest": true }'; + my $manifest_file_ext = ".my"; + my ($exename, $path, $suffix) = fileparse($ENV{'MYSQLD'}, qr/\.[^.]*/); + my $manifest_file_path = $path.$exename.$manifest_file_ext; + open(my $mh, "> $manifest_file_path") or die "Cannot write $manifest_file_path: $!"; + print $mh $content or die; + close($mh) or die; +EOF + --let PLUGIN_DIR_OPT = $KEYRING_KMIP_COMPONENT_OPT # Data directory location @@ -22,8 +37,23 @@ # Create local keyring config --let KEYRING_KMIP_PATH = `SELECT CONCAT( '$MYSQLTEST_VARDIR', '/keyring_kmip')` ---let KEYRING_CONFIG_CONTENT = `SELECT CONCAT('{ \"path\": \"', '$KEYRING_KMIP_PATH', '\", \"server_addr\": \"', '$KMIP_ADDR', '\", \"server_port\": \"', '$KMIP_PORT', '\", \"client_ca\": \"', '$KMIP_CLIENT_CA', '\", \"client_key\": \"', '$KMIP_CLIENT_KEY', '\", \"server_ca\": \"', '$KMIP_SERVER_CA', '\", \"object_group\": \"', 'test-object-group', '\" }')` ---source include/keyring_tests/helper/local_keyring_create_config.inc +--echo # Creating local configuration file for keyring component: $COMPONENT_NAME +--perl + use strict; + my $vardir = $ENV{'MYSQLTEST_VARDIR'}; + my $config_file = $ENV{'CURRENT_DATADIR'} . "/" . $ENV{'COMPONENT_NAME'} . ".cnf"; + open my $uuid_fh, '<', '/proc/sys/kernel/random/uuid' or die "Cannot open /proc/sys/kernel/random/uuid: $!"; + my $uuid = <$uuid_fh>; + close $uuid_fh or die "Cannot close /proc/sys/kernel/random/uuid: $!"; + chomp $uuid; + $uuid =~ s/-//g; + $uuid = substr($uuid, 0, 24); + my $keyring_path = $vardir . '/keyring_kmip_' . $uuid; + my $config_content = '{ "path": "' . $keyring_path . '", "server_addr": "127.0.0.1", "server_port": "5696", "client_ca": "/tmp/certs/mysql-client-cert.pem", "client_key": "/tmp/certs/mysql-client-key.pem", "server_ca": "/tmp/certs/vault-kmip-ca.pem", "object_group": "test-object-group-' . $uuid . '" }'; + open my $cfh, '>', $config_file or die "Cannot write $config_file: $!"; + print $cfh $config_content or die "Cannot write $config_file: $!"; + close $cfh or die "Cannot close $config_file: $!"; +EOF # Create local manifest file for current server instance --let LOCAL_MANIFEST_CONTENT = `SELECT CONCAT('{ \"components\": \"file://', '$COMPONENT_NAME', '\" }')` @@ -32,4 +62,3 @@ # Restart server with manifest file --source include/keyring_tests/helper/start_server_with_manifest.inc --echo # ---------------------------------------------------------------------- - diff --git a/mysql-test/suite/component_keyring_kmip/inc/setup_component_customized.inc b/mysql-test/suite/component_keyring_kmip/inc/setup_component_customized.inc index 0ebd2bb87ab0..7f1f347e4524 100644 --- a/mysql-test/suite/component_keyring_kmip/inc/setup_component_customized.inc +++ b/mysql-test/suite/component_keyring_kmip/inc/setup_component_customized.inc @@ -13,11 +13,25 @@ --source include/keyring_tests/helper/binary_create_customized_manifest.inc # Create global keyring config ---let KEYRING_KMIP_PATH = `SELECT CONCAT( '$MYSQLTEST_VARDIR', '/keyring_kmip')` ---let KEYRING_CONFIG_CONTENT = `SELECT CONCAT('{ \"path\": \"', '$KEYRING_KMIP_PATH','\", \"server_addr\": \"127.0.0.1\", \"server_port\": \"5696\", \"client_ca\": \"/tmp/certs/mysql-client-cert.pem\", \"client_key\":\"/tmp/certs/mysql-client-key.pem\", \"server_ca\":\"/tmp/certs/vault-kmip-ca.pem\" }')` ---source include/keyring_tests/helper/global_keyring_create_customized_config.inc +--let KEYRING_KMIP_PATH = `SELECT CONCAT( '$MYSQLTEST_VARDIR', '/keyring_kmip_customized')` +--echo # Creating custom global configuration file for keyring component: $COMPONENT_NAME +--perl + use strict; + my $vardir = $ENV{'MYSQLTEST_VARDIR'}; + my $config_file = $ENV{'COMPONENT_DIR'} . "/" . $ENV{'COMPONENT_NAME'} . ".cnf"; + open my $uuid_fh, '<', '/proc/sys/kernel/random/uuid' or die "Cannot open /proc/sys/kernel/random/uuid: $!"; + my $uuid = <$uuid_fh>; + close $uuid_fh or die "Cannot close /proc/sys/kernel/random/uuid: $!"; + chomp $uuid; + $uuid =~ s/-//g; + $uuid = substr($uuid, 0, 16); + my $keyring_path = $vardir . '/keyring_kmip_customized_' . $uuid; + my $config_content = '{ "path": "' . $keyring_path . '", "server_addr": "127.0.0.1", "server_port": "5696", "client_ca": "/tmp/certs/mysql-client-cert.pem", "client_key": "/tmp/certs/mysql-client-key.pem", "server_ca": "/tmp/certs/vault-kmip-ca.pem", "object_group": "test-object-group-customized-' . $uuid . '" }'; + open my $cfh, '>', $config_file or die "Cannot write $config_file: $!"; + print $cfh $config_content or die "Cannot write $config_file: $!"; + close $cfh or die "Cannot close $config_file: $!"; +EOF # Restart server with manifest file --source include/keyring_tests/helper/start_server_with_manifest.inc --echo # ---------------------------------------------------------------------- - diff --git a/mysql-test/suite/component_keyring_kmip/inc/teardown_component.inc b/mysql-test/suite/component_keyring_kmip/inc/teardown_component.inc index ddd78d7e55bb..30c47255751c 100644 --- a/mysql-test/suite/component_keyring_kmip/inc/teardown_component.inc +++ b/mysql-test/suite/component_keyring_kmip/inc/teardown_component.inc @@ -2,6 +2,17 @@ --echo # ---------------------------------------------------------------------- --echo # Teardown + +# Tests that enable innodb_redo_log_encrypt leave the redo log encrypted. +# Restart without encryption first so the subsequent inter-test restart can +# start InnoDB without needing the keyring to recover the redo log. +if ($KEYRING_KMIP_USE_BINARY_MANIFEST) +{ + let $restart_parameters = restart: $PLUGIN_DIR_OPT; + --source include/restart_mysqld_no_echo.inc + --let $KEYRING_KMIP_USE_BINARY_MANIFEST= +} + # Remove local manifest file for current server instance --source include/keyring_tests/helper/instance_remove_manifest.inc @@ -11,7 +22,8 @@ # Remove local keyring config --source include/keyring_tests/helper/local_keyring_remove_config.inc -# Remove global keyring config +# Keep shared global keyring config in place to avoid parallel test races. +# Preserve the standard teardown output for existing .result files. --source include/keyring_tests/helper/global_keyring_remove_config.inc # Restart server without manifest file diff --git a/mysql-test/suite/component_keyring_kmip/inc/teardown_component_customized.inc b/mysql-test/suite/component_keyring_kmip/inc/teardown_component_customized.inc index e91ff69d5d90..b1418148a3a4 100644 --- a/mysql-test/suite/component_keyring_kmip/inc/teardown_component_customized.inc +++ b/mysql-test/suite/component_keyring_kmip/inc/teardown_component_customized.inc @@ -2,6 +2,7 @@ --echo # ---------------------------------------------------------------------- --echo # Teardown +# Remove keyring kmip --source include/keyring_tests/helper/local_keyring_kmip_remove.inc # Drop global keyring config @@ -10,6 +11,21 @@ # Remove manifest file for mysqld binary --source include/keyring_tests/helper/binary_remove_manifest.inc +# Restore the standard binary manifest so subsequent tests can read their +# instance manifests (MySQL only reads instance manifests when the binary +# manifest says read_local_manifest: true). +--perl + use strict; + use File::Basename; + my $content = '{ "read_local_manifest": true }'; + my $manifest_file_ext = ".my"; + my ($exename, $path, $suffix) = fileparse($ENV{'MYSQLD'}, qr/\.[^.]*/); + my $manifest_file_path = $path.$exename.$manifest_file_ext; + open(my $mh, "> $manifest_file_path") or die "Cannot write $manifest_file_path: $!"; + print $mh $content or die; + close($mh) or die; +EOF + # Restart server without manifest file --source include/keyring_tests/helper/cleanup_server_with_manifest.inc --echo # ---------------------------------------------------------------------- diff --git a/mysql-test/suite/component_keyring_kmip/t/clone_remote_encrypt.test b/mysql-test/suite/component_keyring_kmip/t/clone_remote_encrypt.test index 3584dc8527ba..d7af8a82cc1c 100644 --- a/mysql-test/suite/component_keyring_kmip/t/clone_remote_encrypt.test +++ b/mysql-test/suite/component_keyring_kmip/t/clone_remote_encrypt.test @@ -1,6 +1,7 @@ --source include/have_component_keyring_kmip.inc # Test remote clone with different table types with debug sync +--source include/not_parallel.inc --source include/have_innodb_max_16k.inc --let $HOST = 127.0.0.1 diff --git a/mysql-test/suite/component_keyring_kmip/t/encrypt_explicit.result b/mysql-test/suite/component_keyring_kmip/t/encrypt_explicit.result index a6c956d38c8d..05202a5a5700 100644 --- a/mysql-test/suite/component_keyring_kmip/t/encrypt_explicit.result +++ b/mysql-test/suite/component_keyring_kmip/t/encrypt_explicit.result @@ -17,8 +17,6 @@ CREATE table tab1(c1 int); SELECT @@datadir; @@datadir MYSQLD_DATADIR1/ -SHOW VARIABLES LIKE 'keyring_file%'; -Variable_name Value SELECT @@innodb_undo_log_encrypt; @@innodb_undo_log_encrypt 1 @@ -37,6 +35,7 @@ DROP UNDO TABLESPACE undo_004; DROP TABLE tab2; DROP DATABASE nath; # Stop the encrypt server +# Taking backup of global manifest file for MySQL server SELECT @@datadir; @@datadir MYSQLD_OLD_DATADIR @@ -47,6 +46,7 @@ DROP TABLE tab1; # # Stop the non-encrypted server # Run the bootstrap command of datadir1 +# Restore global manifest file for MySQL server from backup # Show that a master key with a blank Server UUID is not used. SELECT * FROM performance_schema.keyring_keys where KEY_ID='INNODBKey--1'; KEY_ID KEY_OWNER BACKEND_KEY_ID diff --git a/mysql-test/suite/component_keyring_kmip/t/encrypt_explicit.test b/mysql-test/suite/component_keyring_kmip/t/encrypt_explicit.test index 028ec78ea118..1ea893fb68e9 100644 --- a/mysql-test/suite/component_keyring_kmip/t/encrypt_explicit.test +++ b/mysql-test/suite/component_keyring_kmip/t/encrypt_explicit.test @@ -4,7 +4,6 @@ --source include/have_innodb_16k.inc # This test must be not be run in parallel with other tests because # it requires global manifest file to load keyring component directly - --source ../inc/setup_component_customized.inc --source include/keyring_tests/innodb/encrypt_explicit.inc --source ../inc/teardown_component_customized.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/keyring_component_status.result b/mysql-test/suite/component_keyring_kmip/t/keyring_component_status.result index 2d5641be1917..b6af55ff9623 100644 --- a/mysql-test/suite/component_keyring_kmip/t/keyring_component_status.result +++ b/mysql-test/suite/component_keyring_kmip/t/keyring_component_status.result @@ -24,6 +24,10 @@ Client_ca /tmp/certs/mysql-client-cert.pem Client_key /tmp/certs/mysql-client-key.pem Server_ca /tmp/certs/vault-kmip-ca.pem Object_group test-object-group +Max_objects 65535 +Kmip_timeout_ms 5000 +Tls_peer_verification false +Tls_hostname_verification false SELECT PRIO, ERROR_CODE, SUBSYSTEM, DATA FROM performance_schema.error_log WHERE ERROR_CODE='MY-013712'; PRIO ERROR_CODE SUBSYSTEM DATA # Restarting server without keyring component diff --git a/mysql-test/suite/component_keyring_kmip/t/keyring_component_status.test b/mysql-test/suite/component_keyring_kmip/t/keyring_component_status.test index a762de00cac2..8ed33ee9f3b6 100644 --- a/mysql-test/suite/component_keyring_kmip/t/keyring_component_status.test +++ b/mysql-test/suite/component_keyring_kmip/t/keyring_component_status.test @@ -1,5 +1,34 @@ --source include/have_component_keyring_kmip.inc --source ../inc/setup_component.inc ---source include/keyring_tests/mats/keyring_component_status.inc +# Tests related to performance_schema.keyring_component_status table + +CALL mtr.add_suppression("No suitable 'keyring_component_metadata_query' service implementation found to fulfill the request"); + +--echo # +--echo # Bug#32390719: QUERYING KEYRING_COMPONENT_STATUS TABLE +--echo # GENERATES AN ERROR IN THE SERVER LOG +--echo # + +# Should show metadata obtained from keyring component +--replace_result $MYSQLTEST_VARDIR MYSQLTEST_VARDIR +--replace_regex /test-object-group-[0-9a-f]+/test-object-group/ +SELECT * FROM performance_schema.keyring_component_status; + +# Should be empty +SELECT PRIO, ERROR_CODE, SUBSYSTEM, DATA FROM performance_schema.error_log WHERE ERROR_CODE='MY-013712'; + +--echo # Restarting server without keyring component +--source include/keyring_tests/helper/instance_backup_manifest.inc +let $restart_parameters = restart: $PLUGIN_DIR_OPT; +--source include/restart_mysqld_no_echo.inc + +# Should be empty +--replace_regex /test-object-group-[0-9a-f]+/test-object-group/ +SELECT * FROM performance_schema.keyring_component_status; + +# Should show one row +SELECT PRIO, ERROR_CODE, SUBSYSTEM, DATA FROM performance_schema.error_log WHERE ERROR_CODE='MY-013712'; + +--source include/keyring_tests/helper/instance_restore_manifest.inc --source ../inc/teardown_component.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/log_encrypt_1.test b/mysql-test/suite/component_keyring_kmip/t/log_encrypt_1.test index 9b003b6e42de..d3645b315618 100644 --- a/mysql-test/suite/component_keyring_kmip/t/log_encrypt_1.test +++ b/mysql-test/suite/component_keyring_kmip/t/log_encrypt_1.test @@ -4,7 +4,8 @@ --source include/no_valgrind_without_big.inc --source include/have_innodb_max_16k.inc - +--source include/not_parallel.inc +--let $KEYRING_KMIP_USE_BINARY_MANIFEST= 1 --source ../inc/setup_component.inc --source include/keyring_tests/mats/log_encrypt_1.inc --source ../inc/teardown_component.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/log_encrypt_5.result b/mysql-test/suite/component_keyring_kmip/t/log_encrypt_5.result index f5ded3ea3f7c..5222bde1bc25 100644 --- a/mysql-test/suite/component_keyring_kmip/t/log_encrypt_5.result +++ b/mysql-test/suite/component_keyring_kmip/t/log_encrypt_5.result @@ -14,7 +14,7 @@ SET GLOBAL innodb_redo_log_encrypt = 1; SET GLOBAL innodb_undo_log_encrypt = 1; SELECT @@global.innodb_redo_log_encrypt ; @@global.innodb_redo_log_encrypt -ON +1 CREATE TABLE tde_db.t1 (a BIGINT PRIMARY KEY, b LONGBLOB) ENGINE=InnoDB; INSERT INTO t1 (a, b) VALUES (1, REPEAT('a', 6*512*512)); SELECT a,LEFT(b,10) FROM tde_db.t1; @@ -30,7 +30,7 @@ SET GLOBAL innodb_redo_log_encrypt = 0; SET GLOBAL innodb_undo_log_encrypt = 0; SELECT @@global.innodb_redo_log_encrypt ; @@global.innodb_redo_log_encrypt -OFF +0 CREATE TABLE tde_db.t3 (a BIGINT PRIMARY KEY, b LONGBLOB) ENGINE=InnoDB; INSERT INTO t3 (a, b) VALUES (1, REPEAT('a', 6*512*512)); SELECT a,LEFT(b,10) FROM tde_db.t3; @@ -46,7 +46,7 @@ SET GLOBAL innodb_redo_log_encrypt = 1; SET GLOBAL innodb_undo_log_encrypt = 1; SELECT @@global.innodb_redo_log_encrypt ; @@global.innodb_redo_log_encrypt -ON +1 CREATE TABLE tde_db.t5 (a BIGINT PRIMARY KEY, b LONGBLOB) ENGINE=InnoDB; INSERT INTO t5 (a, b) VALUES (1, REPEAT('a', 6*512*512)); SELECT a,LEFT(b,10) FROM tde_db.t5; diff --git a/mysql-test/suite/component_keyring_kmip/t/log_encrypt_6.result b/mysql-test/suite/component_keyring_kmip/t/log_encrypt_6.result index 082a665e239f..9b6a811fac5c 100644 --- a/mysql-test/suite/component_keyring_kmip/t/log_encrypt_6.result +++ b/mysql-test/suite/component_keyring_kmip/t/log_encrypt_6.result @@ -18,13 +18,14 @@ status_key status_value Component_name component_keyring_kmip Implementation_name component_keyring_kmip Component_status Active +Tls_hostname_verification false SET GLOBAL innodb_redo_log_encrypt = 1; SELECT @@global.innodb_redo_log_encrypt ; @@global.innodb_redo_log_encrypt -ON +1 SELECT @@global.innodb_redo_log_encrypt ; @@global.innodb_redo_log_encrypt -ON +1 CREATE TABLE tde_db.t1 (a BIGINT PRIMARY KEY, b LONGBLOB) ENGINE=InnoDB; INSERT INTO t1 (a, b) VALUES (1, REPEAT('a', 6*512*512)); SELECT a,LEFT(b,10) FROM tde_db.t1; @@ -38,7 +39,7 @@ a LEFT(b,10) 1 aaaaaaaaaa SELECT @@global.innodb_redo_log_encrypt ; @@global.innodb_redo_log_encrypt -ON +1 CREATE TABLE tde_db.t3 (a BIGINT PRIMARY KEY, b LONGBLOB) ENGINE=InnoDB; INSERT INTO t3 (a, b) VALUES (1, REPEAT('a', 6*512*512)); SELECT a,LEFT(b,10) FROM tde_db.t3; @@ -52,7 +53,7 @@ a LEFT(b,10) 1 aaaaaaaaaa SELECT @@global.innodb_redo_log_encrypt ; @@global.innodb_redo_log_encrypt -ON +1 CREATE TABLE tde_db.t5 (a BIGINT PRIMARY KEY, b LONGBLOB) ENGINE=InnoDB; INSERT INTO t5 (a, b) VALUES (1, REPEAT('a', 6*512*512)); SELECT a,LEFT(b,10) FROM tde_db.t5; diff --git a/mysql-test/suite/component_keyring_kmip/t/log_encrypt_kill.result b/mysql-test/suite/component_keyring_kmip/t/log_encrypt_kill.result index caff8f878396..817e955cc0de 100644 --- a/mysql-test/suite/component_keyring_kmip/t/log_encrypt_kill.result +++ b/mysql-test/suite/component_keyring_kmip/t/log_encrypt_kill.result @@ -7,10 +7,8 @@ # Taking backup of global manifest file for MySQL server SELECT @@global.innodb_redo_log_encrypt; @@global.innodb_redo_log_encrypt -OFF +0 SET GLOBAL innodb_redo_log_encrypt = 1; -Warnings: -Warning 7013 InnoDB: Redo log cannot be encrypted if the keyring is not loaded. SET GLOBAL innodb_undo_log_encrypt = 1; Warnings: Warning 7014 InnoDB: Undo log can't be encrypted if the keyring is not loaded. @@ -27,7 +25,7 @@ c1 LEFT(c2,10) 100 cccccccccc DROP TABLE tne_1; # Stop the MTR default DB server -Pattern "Redo log cannot be encrypted if the keyring is not loaded." found +Pattern "Can\'t set redo log files to be encrypted" found # create bootstrap file # Restore global manifest file for MySQL server from backup # Prepare new datadir @@ -35,7 +33,7 @@ Pattern "Redo log cannot be encrypted if the keyring is not loaded." found # Starting server with keyring plugin SELECT @@global.innodb_redo_log_encrypt; @@global.innodb_redo_log_encrypt -OFF +0 SET GLOBAL innodb_redo_log_encrypt = 1; SELECT @@global.innodb_undo_log_encrypt; @@global.innodb_undo_log_encrypt @@ -51,10 +49,11 @@ status_key status_value Component_name component_keyring_kmip Implementation_name component_keyring_kmip Component_status Active +Tls_hostname_verification false SET GLOBAL innodb_redo_log_encrypt = 0; SELECT @@global.innodb_redo_log_encrypt; @@global.innodb_redo_log_encrypt -OFF +0 SET GLOBAL innodb_undo_log_encrypt = 0; SELECT @@global.innodb_undo_log_encrypt; @@global.innodb_undo_log_encrypt @@ -66,6 +65,7 @@ status_key status_value Component_name component_keyring_kmip Implementation_name component_keyring_kmip Component_status Active +Tls_hostname_verification false DROP TABLE IF EXISTS t1; DROP DATABASE IF EXISTS tde_db; CREATE DATABASE tde_db; diff --git a/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_1.test b/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_1.test index 086a0975fc73..5b1b99466bfd 100644 --- a/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_1.test +++ b/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_1.test @@ -19,6 +19,7 @@ --source include/have_debug.inc # Disable in valgrind because of timeout, cf. Bug#22760145 --source include/not_valgrind.inc +--source include/not_parallel.inc --source ../inc/setup_component.inc --source include/keyring_tests/mats/mysql_ts_alter_encrypt_1.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_2.result b/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_2.result index eb75fc4a4d7c..24f3af4f20d8 100644 --- a/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_2.result +++ b/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_2.result @@ -228,10 +228,6 @@ DROP TABLESPACE mysql; ERROR 42000: InnoDB: `mysql` is a reserved tablespace name. ALTER TABLESPACE mysql RENAME TO xyz; ERROR 42000: InnoDB: `mysql` is a reserved tablespace name. -ALTER TABLESPACE mysql ENGINE=myisam; -ERROR HY000: Engine 'myisam' does not match stored engine 'InnoDB' for tablespace 'mysql' -ALTER TABLESPACE mysql ENGINE=memory; -ERROR HY000: Engine 'memory' does not match stored engine 'InnoDB' for tablespace 'mysql' ######################################################################### # Restart with same keyring option # - tables in mysql ts should be accessible diff --git a/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_2.test b/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_2.test index 1a12498d4ac1..fdcdc6d0ca60 100644 --- a/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_2.test +++ b/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_2.test @@ -21,6 +21,7 @@ --source include/big_test.inc --source include/have_debug.inc +--source include/not_parallel.inc # Disable in valgrind because of timeout, cf. Bug#22760145 --source include/not_valgrind.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_2.test b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_2.test index 6a06f4a67598..ffeddd3d9614 100644 --- a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_2.test +++ b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_2.test @@ -2,6 +2,7 @@ # InnoDB transparent tablespace data encryption # This test case will test basic encryption support features. +--source include/not_parallel.inc --source include/no_valgrind_without_big.inc --source ../inc/setup_component.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.result b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.result index 5b032f1e04b3..61d0ba5b50be 100644 --- a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.result +++ b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.result @@ -824,7 +824,7 @@ CREATE PROCEDURE tde_db.rotate_master_key() begin declare i int default 1; declare has_error int default 0; -while (i <= 500) DO +while (i <= 100) DO ALTER INSTANCE ROTATE INNODB MASTER KEY; set i = i + 1; end while; diff --git a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.test b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.test index 2e1c4a6acc4d..c59309c624e8 100644 --- a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.test +++ b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.test @@ -2,6 +2,7 @@ # InnoDB transparent tablespace data encryption # This test case will test basic encryption support features. +--source include/not_parallel.inc --source include/no_valgrind_without_big.inc --source include/have_innodb_max_16k.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_4.test b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_4.test index a39cc24d8b6d..c9dd978eac67 100644 --- a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_4.test +++ b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_4.test @@ -4,6 +4,7 @@ --source include/no_valgrind_without_big.inc --source include/have_debug.inc +--source include/not_parallel.inc --source ../inc/setup_component.inc --source include/keyring_tests/innodb/table_encrypt_4.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_10.result b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_10.result index 80a96160117d..50b1aa6248ee 100644 --- a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_10.result +++ b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_10.result @@ -41,7 +41,7 @@ ERROR HY000: Can't find master key from keyring, please check in the server log SET SESSION debug= '+d,alter_encrypt_tablespace_page_10'; # Encrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='Y'; -# RESTART 2 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 2 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just before flushing page 0 at the end SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -75,7 +75,7 @@ NAME ENCRYPTION encrypt_ts Y # Unencrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='N'; -# RESTART 4 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 4 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just before flushing page 0 at the end SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -87,7 +87,8 @@ Pattern "Finished DECRYPTION for tablespace encrypt_ts" not found # Search the pattern in error log Pattern "Resuming DECRYPTION for tablespace encrypt_ts" found Pattern "Finished DECRYPTION for tablespace encrypt_ts" found -# RESTART 6 : WITHOUT KEYRING PLUGIN +# RESTART 6 : WITHOUT KEYRING COMPONENT +# Taking backup of local manifest file for MySQL server instance SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; NAME ENCRYPTION encrypt_ts N @@ -103,6 +104,7 @@ SOME VALUES SOME VALUES SOME VALUES SOME VALUES +# Restore local manifest file for MySQL server instance from backup #-------------------------- TEST 2 -------------------------------------# # RESTART 7 : WITH KEYRING ######################################################################## @@ -112,7 +114,7 @@ SOME VALUES SET SESSION debug= '+d,alter_encrypt_tablespace_page_10'; # Encrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='Y'; -# RESTART 8 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 8 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just after flushing page 0 at the end SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -146,7 +148,7 @@ NAME ENCRYPTION encrypt_ts Y # Unencrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='N'; -# RESTART 10 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 10 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just after flushing page 0 at the end SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -158,7 +160,8 @@ Pattern "Finished DECRYPTION for tablespace encrypt_ts" not found # Search the pattern in error log Pattern "Resuming DECRYPTION for tablespace encrypt_ts" found Pattern "Finished DECRYPTION for tablespace encrypt_ts" found -# RESTART 12 : WITHOUT KEYRING PLUGIN +# RESTART 12 : WITHOUT KEYRING COMPONENT +# Taking backup of local manifest file for MySQL server instance SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; NAME ENCRYPTION encrypt_ts N @@ -174,6 +177,7 @@ SOME VALUES SOME VALUES SOME VALUES SOME VALUES +# Restore local manifest file for MySQL server instance from backup #-------------------------- TEST 3 -------------------------------------# # RESTART 13 : WITH KEYRING ######################################################################## @@ -183,7 +187,7 @@ SOME VALUES SET SESSION debug= '+d,alter_encrypt_tablespace_page_10'; # Encrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='Y'; -# RESTART 14 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 14 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just before resetting_progress on page 0 SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -217,7 +221,7 @@ NAME ENCRYPTION encrypt_ts Y # Unencrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='N'; -# RESTART 16 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 16 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just before resetting_progress on page 0 SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -229,7 +233,8 @@ Pattern "Finished DECRYPTION for tablespace encrypt_ts" not found # Search the pattern in error log Pattern "Resuming DECRYPTION for tablespace encrypt_ts" found Pattern "Finished DECRYPTION for tablespace encrypt_ts" found -# RESTART 18 : WITHOUT KEYRING PLUGIN +# RESTART 18 : WITHOUT KEYRING COMPONENT +# Taking backup of local manifest file for MySQL server instance SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; NAME ENCRYPTION encrypt_ts N @@ -245,6 +250,7 @@ SOME VALUES SOME VALUES SOME VALUES SOME VALUES +# Restore local manifest file for MySQL server instance from backup #-------------------------- TEST 4 -------------------------------------# # RESTART 19 : WITH KEYRING # Encrypt the tablespace. @@ -259,7 +265,7 @@ NAME ENCRYPTION encrypt_ts Y # Unencrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='N'; -# RESTART 20 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 20 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just before updating flags SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -271,7 +277,8 @@ Pattern "Finished DECRYPTION for tablespace encrypt_ts" not found # Search the pattern in error log Pattern "Resuming DECRYPTION for tablespace encrypt_ts" found Pattern "Finished DECRYPTION for tablespace encrypt_ts" found -# RESTART 22 : WITHOUT KEYRING PLUGIN +# RESTART 22 : WITHOUT KEYRING COMPONENT +# Taking backup of local manifest file for MySQL server instance SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; NAME ENCRYPTION encrypt_ts N @@ -287,6 +294,7 @@ SOME VALUES SOME VALUES SOME VALUES SOME VALUES +# Restore local manifest file for MySQL server instance from backup #-------------------------- TEST 5 -------------------------------------# # RESTART 23 : WITH KEYRING ######################################################################## @@ -296,7 +304,7 @@ SOME VALUES SET SESSION debug= '+d,alter_encrypt_tablespace_page_10'; # Encrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='Y'; -# RESTART 24 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 24 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just before encryption processing is started SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -330,7 +338,7 @@ NAME ENCRYPTION encrypt_ts Y # Unencrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='N'; -# RESTART 26 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 26 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just before encryption processing is started SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -342,7 +350,8 @@ Pattern "Finished DECRYPTION for tablespace encrypt_ts" not found # Search the pattern in error log Pattern "Resuming DECRYPTION for tablespace encrypt_ts" found Pattern "Finished DECRYPTION for tablespace encrypt_ts" found -# RESTART 28 : WITHOUT KEYRING PLUGIN +# RESTART 28 : WITHOUT KEYRING COMPONENT +# Taking backup of local manifest file for MySQL server instance SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; NAME ENCRYPTION encrypt_ts N @@ -358,6 +367,7 @@ SOME VALUES SOME VALUES SOME VALUES SOME VALUES +# Restore local manifest file for MySQL server instance from backup #-------------------------- TEST 6 -------------------------------------# # RESTART 29 : WITH KEYRING ######################################################################## @@ -367,7 +377,7 @@ SOME VALUES SET SESSION debug= '+d,alter_encrypt_tablespace_page_10'; # Encrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='Y'; -# RESTART 30 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 30 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just after encryption processing is finished SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -401,7 +411,7 @@ NAME ENCRYPTION encrypt_ts Y # Unencrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='N'; -# RESTART 32 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 32 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just after encryption processing is finished SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -413,7 +423,8 @@ Pattern "Finished DECRYPTION for tablespace encrypt_ts" not found # Search the pattern in error log Pattern "Resuming DECRYPTION for tablespace encrypt_ts" found Pattern "Finished DECRYPTION for tablespace encrypt_ts" found -# RESTART 34 : WITHOUT KEYRING PLUGIN +# RESTART 34 : WITHOUT KEYRING COMPONENT +# Taking backup of local manifest file for MySQL server instance SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; NAME ENCRYPTION encrypt_ts N @@ -429,6 +440,7 @@ SOME VALUES SOME VALUES SOME VALUES SOME VALUES +# Restore local manifest file for MySQL server instance from backup #-------------------------- TEST 7 -------------------------------------# # RESTART 35 : WITH KEYRING ######################################################################## @@ -438,7 +450,7 @@ SOME VALUES SET SESSION debug= '+d,alter_encrypt_tablespace_page_10'; # Encrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='Y'; -# RESTART 36 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 36 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just after inserting DDL Log Entry SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -472,7 +484,7 @@ NAME ENCRYPTION encrypt_ts Y # Unencrypt the tablespace. It will cause crash. ALTER TABLESPACE encrypt_ts ENCRYPTION='N'; -# RESTART 38 : WITH KEYRING PLUGIN after crash and cause resume operation +# RESTART 38 : WITH KEYRING COMPONENT after crash and cause resume operation # to crash just after inserting DDL Log Entry SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; Got one of the listed errors @@ -484,7 +496,8 @@ Pattern "Finished DECRYPTION for tablespace encrypt_ts" not found # Search the pattern in error log Pattern "Resuming DECRYPTION for tablespace encrypt_ts" found Pattern "Finished DECRYPTION for tablespace encrypt_ts" found -# RESTART 40 : WITHOUT KEYRING PLUGIN +# RESTART 40 : WITHOUT KEYRING COMPONENT +# Taking backup of local manifest file for MySQL server instance SELECT NAME, ENCRYPTION FROM INFORMATION_SCHEMA.INNODB_TABLESPACES WHERE NAME='encrypt_ts'; NAME ENCRYPTION encrypt_ts N @@ -500,6 +513,7 @@ SOME VALUES SOME VALUES SOME VALUES SOME VALUES +# Restore local manifest file for MySQL server instance from backup ########### # Cleanup # ########### diff --git a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_10.test b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_10.test index 7419ac764edf..8ea50077a6a2 100644 --- a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_10.test +++ b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_10.test @@ -1,6 +1,7 @@ --source include/have_component_keyring_kmip.inc --source include/have_debug.inc --source include/big_test.inc +--source include/not_parallel.inc --source include/no_valgrind_without_big.inc # Disable in valgrind because of timeout, cf. Bug#22760145 --source include/not_valgrind.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_11.test b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_11.test index da7c1bc9f439..81c74e8af471 100644 --- a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_11.test +++ b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_11.test @@ -4,6 +4,7 @@ --source include/no_valgrind_without_big.inc # Disable in valgrind because of timeout, cf. Bug#22760145 --source include/not_valgrind.inc +--source include/not_parallel.inc --source ../inc/setup_component.inc --source include/keyring_tests/innodb/tablespace_encrypt_11.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_3.test b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_3.test index a78047ef0521..13e5c9daa3c8 100644 --- a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_3.test +++ b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_3.test @@ -1,6 +1,7 @@ --source include/have_component_keyring_kmip.inc --source include/no_valgrind_without_big.inc --source include/have_debug.inc +--source include/not_parallel.inc --source ../inc/setup_component.inc --source include/keyring_tests/innodb/tablespace_encrypt_3.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_7.test b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_7.test index fdedd67834a7..c6c146563e97 100644 --- a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_7.test +++ b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_7.test @@ -1,6 +1,7 @@ --source include/have_component_keyring_kmip.inc --source include/big_test.inc --source include/have_debug.inc +--source include/not_parallel.inc # --source include/no_valgrind_without_big.inc # Disable in valgrind because of timeout, cf. Bug#22760145 --source include/not_valgrind.inc From 67b170e673485939fa999713748001b410c56167 Mon Sep 17 00:00:00 2001 From: Vadim Yalovets Date: Sun, 26 Jul 2026 18:39:21 +0300 Subject: [PATCH 13/32] PS-11238 debian package postinst inconsistency --- build-ps/debian/percona-server-server.postinst | 7 +++++++ 1 file changed, 7 insertions(+) diff --git a/build-ps/debian/percona-server-server.postinst b/build-ps/debian/percona-server-server.postinst index 4d9f33b339d8..e434ec419bdd 100755 --- a/build-ps/debian/percona-server-server.postinst +++ b/build-ps/debian/percona-server-server.postinst @@ -72,6 +72,13 @@ resolve_cnf_action() { # Already our alternative, nothing to ask/do. return 1 fi + if [ "${CNF_TARGET}" = "/etc/mysql/my.cnf.fallback" ]; then + # Just the low-priority fallback registered by + # percona-xtradb-cluster-common (which is configured + # before this package on a fresh install). It is not a + # real existing config, so take it over without asking. + return 0 + fi # A symlink managed by update-alternatives, but pointing elsewhere: # ask before taking it over. db_input high "${TEMPLATE_PREFIX}/existing_config_file" || true From c4a66a1ef2b6c6eb3ccdb537992aaf1374664c2a Mon Sep 17 00:00:00 2001 From: Oleksiy Lukin Date: Mon, 22 Jun 2026 14:03:16 +0300 Subject: [PATCH 14/32] PS-11143-[8.4] sql/auth: quiesce auth plugin callbacks before teardown Add shutdown protection on the level into the shared auth/plugin lifecycle so all authentication plugins are covered. - Add a global auth-plugin shutdown barrier API (`sql/auth/auth_plugin_shutdown.h` + implementation in `sql/auth/sql_authentication.cc`). - Guard auth plugin callback entry points with RAII operation tracking. - Start barrier shutdown/drain from `plugin_shutdown()` before plugin deinitialization (`sql/sql_plugin.cc`). - Remove plugin-local quiesce state/wait logic from `sql/auth/sha2_password.cc` now that teardown protection is centralized. - Bound auth-plugin shutdown barrier wait with a 2s timeout (`AUTH_PLUGIN_SHUTDOWN_TIMEOUT_MS`) to avoid indefinite shutdown hangs. - Keep normal shutdown flow progressing even when auth operations fail to drain in time. Regression coverage: - Add `percona.bug_ps11143_threadpool_auth_shutdown` to reproduce the threadpool auth-vs-shutdown race and verify clean shutdown. - Include expected result file and test cleanup to preserve MTR state. Verified: - percona.bug_ps11143_threadpool_auth_shutdown - percona.signal_handling_threadpool - percona.threadpool_stats - percona.kill_idle_trx_threadpool - percona.threadpool_debug --- ...ug_ps11143_threadpool_auth_shutdown.result | 11 +++ .../bug_ps11143_threadpool_auth_shutdown.test | 45 +++++++++ sql/auth/acl_table_user.cc | 21 ++++- sql/auth/auth_plugin_shutdown.h | 47 ++++++++++ sql/auth/sql_auth_cache.cc | 16 +++- sql/auth/sql_authentication.cc | 73 ++++++++++++++- sql/auth/sql_authentication.h | 2 + sql/auth/sql_mfa.cc | 17 +++- sql/auth/sql_user.cc | 91 ++++++++++++------- sql/sql_plugin.cc | 5 + 10 files changed, 281 insertions(+), 47 deletions(-) create mode 100644 mysql-test/suite/percona/r/bug_ps11143_threadpool_auth_shutdown.result create mode 100644 mysql-test/suite/percona/t/bug_ps11143_threadpool_auth_shutdown.test create mode 100644 sql/auth/auth_plugin_shutdown.h diff --git a/mysql-test/suite/percona/r/bug_ps11143_threadpool_auth_shutdown.result b/mysql-test/suite/percona/r/bug_ps11143_threadpool_auth_shutdown.result new file mode 100644 index 000000000000..4331e438c1c0 --- /dev/null +++ b/mysql-test/suite/percona/r/bug_ps11143_threadpool_auth_shutdown.result @@ -0,0 +1,11 @@ +# +# Verify auth callback drains before plugin deinit with threadpool +# +# restart:--thread-handling=pool-of-threads --thread-pool-size=2 --thread-pool-max-threads=4 +CREATE USER tp_user@127.0.0.1 IDENTIFIED WITH caching_sha2_password BY 'tp_pwd'; +SET GLOBAL debug = "+d,auth_plugin_before_callback_sync"; +1 +1 +SET DEBUG_SYNC='now WAIT_FOR auth_plugin_before_callback_entered TIMEOUT 30'; +# restart +DROP USER IF EXISTS tp_user@127.0.0.1; diff --git a/mysql-test/suite/percona/t/bug_ps11143_threadpool_auth_shutdown.test b/mysql-test/suite/percona/t/bug_ps11143_threadpool_auth_shutdown.test new file mode 100644 index 000000000000..1e78fd1f41c4 --- /dev/null +++ b/mysql-test/suite/percona/t/bug_ps11143_threadpool_auth_shutdown.test @@ -0,0 +1,45 @@ +--source include/not_windows.inc +--source include/have_debug.inc +--source include/have_debug_sync.inc + +--echo # +--echo # Verify auth callback drains before plugin deinit with threadpool +--echo # + +--let $restart_parameters=restart:--thread-handling=pool-of-threads --thread-pool-size=2 --thread-pool-max-threads=4 +--source include/restart_mysqld.inc +--source include/have_pool_of_threads.inc + +CREATE USER tp_user@127.0.0.1 IDENTIFIED WITH caching_sha2_password BY 'tp_pwd'; + +SET GLOBAL debug = "+d,auth_plugin_before_callback_sync"; + +--let $mysqld_pid_file=`SELECT @@GLOBAL.pid_file` + +--let $output_file= $MYSQLTEST_VARDIR/tmp/bug_ps11143_mysql_output +--let $pid_file= $MYSQLTEST_VARDIR/tmp/bug_ps11143_mysql_pid +--let $sql_file= $MYSQLTEST_VARDIR/tmp/bug_ps11143_query.sql +--write_file $sql_file +SELECT 1; +EOF +--let $command= $MYSQL +--let $command_opt= --protocol=tcp --host=127.0.0.1 --port=$MASTER_MYPORT --user=tp_user --password=tp_pwd < $sql_file +--let $redirect_stderr= 1 +--source include/start_proc_in_background.inc + +SET DEBUG_SYNC='now WAIT_FOR auth_plugin_before_callback_entered TIMEOUT 30'; + +--source include/expect_crash.inc +exec kill -TERM `cat $mysqld_pid_file`; + +--source include/wait_until_disconnected.inc +--source include/wait_proc_to_finish.inc + +--let $restart_parameters= +--source include/start_mysqld.inc + +DROP USER IF EXISTS tp_user@127.0.0.1; + +--remove_file $pid_file +--remove_file $output_file +--remove_file $sql_file diff --git a/sql/auth/acl_table_user.cc b/sql/auth/acl_table_user.cc index b3c3462159a2..2c401104c119 100644 --- a/sql/auth/acl_table_user.cc +++ b/sql/auth/acl_table_user.cc @@ -47,6 +47,7 @@ Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA */ #include "sql/auth/auth_acls.h" /* ACLs */ #include "sql/auth/auth_common.h" /* User_table_schema, ... */ #include "sql/auth/auth_internal.h" /* acl_print_ha_error */ +#include "sql/auth/auth_plugin_shutdown.h" #include "sql/auth/partial_revokes.h" #include "sql/auth/sql_auth_cache.h" /* global_acl_memory */ #include "sql/auth/sql_authentication.h" /* Cached_authentication_plugins */ @@ -1656,6 +1657,13 @@ bool Acl_table_user_reader::read_plugin_info( if (native_plugin) { const uint password_len = password ? strlen(password) : 0; st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(native_plugin)->info; + Auth_plugin_operation_guard op_guard; + if (!op_guard) { + LogErr(WARNING_LEVEL, ER_AUTHCACHE_USER_IGNORED_INVALID_PASSWORD, + user.user ? user.user : "", + user.host.get_host() ? user.host.get_host() : ""); + return true; + } if (auth->validate_authentication_string(password, password_len) == 0) { // auth_string takes precedence over password if (user.credentials[PRIMARY_CRED].m_auth_string.length == 0) { @@ -1704,10 +1712,11 @@ bool Acl_table_user_reader::read_plugin_info( my_plugin_lock_by_name(nullptr, user.plugin, MYSQL_AUTHENTICATION_PLUGIN); if (plugin) { st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(plugin)->info; - if (auth->validate_authentication_string( - const_cast( - user.credentials[PRIMARY_CRED].m_auth_string.str), - user.credentials[PRIMARY_CRED].m_auth_string.length)) { + Auth_plugin_operation_guard op_guard; + if (!op_guard || auth->validate_authentication_string( + const_cast( + user.credentials[PRIMARY_CRED].m_auth_string.str), + user.credentials[PRIMARY_CRED].m_auth_string.length)) { LogErr(WARNING_LEVEL, ER_AUTHCACHE_USER_IGNORED_INVALID_PASSWORD, user.user ? user.user : "", user.host.get_host() ? user.host.get_host() : ""); @@ -1921,7 +1930,9 @@ bool Acl_table_user_reader::read_user_attributes(ACL_USER &user) { if (plugin) { st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(plugin)->info; - if (auth->validate_authentication_string( + Auth_plugin_operation_guard op_guard; + if (!op_guard || + auth->validate_authentication_string( const_cast( user.credentials[SECOND_CRED].m_auth_string.str), user.credentials[SECOND_CRED].m_auth_string.length)) { diff --git a/sql/auth/auth_plugin_shutdown.h b/sql/auth/auth_plugin_shutdown.h new file mode 100644 index 000000000000..4a47f4003fe8 --- /dev/null +++ b/sql/auth/auth_plugin_shutdown.h @@ -0,0 +1,47 @@ +/* Copyright (c) 2026, Percona LLC and/or its affiliates. + + This program is free software; you can redistribute it and/or modify + it under the terms of the GNU General Public License, version 2.0, + as published by the Free Software Foundation. + + This program is designed to work with certain software (including + but not limited to OpenSSL) that is licensed under separate terms, + as designated in a particular file or component or in included license + documentation. The authors of MySQL hereby grant you an additional + permission to link the program and your derivative works with the + separately licensed software that they have either included with + the program or referenced in the documentation. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License, version 2.0, for more details. + + You should have received a copy of the GNU General Public License + along with this program; if not, write to the Free Software + Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA */ + +#ifndef AUTH_PLUGIN_SHUTDOWN_INCLUDED +#define AUTH_PLUGIN_SHUTDOWN_INCLUDED + +bool begin_auth_plugin_operation(); +void end_auth_plugin_operation(); +void start_auth_plugin_shutdown_and_wait(); + +class Auth_plugin_operation_guard { + public: + Auth_plugin_operation_guard() + : m_has_operation(begin_auth_plugin_operation()) {} + + ~Auth_plugin_operation_guard() { + if (m_has_operation) end_auth_plugin_operation(); + } + + bool has_operation() const { return m_has_operation; } + explicit operator bool() const { return m_has_operation; } + + private: + bool m_has_operation; +}; + +#endif // AUTH_PLUGIN_SHUTDOWN_INCLUDED diff --git a/sql/auth/sql_auth_cache.cc b/sql/auth/sql_auth_cache.cc index 0356203ccf8f..c6a490986ed2 100644 --- a/sql/auth/sql_auth_cache.cc +++ b/sql/auth/sql_auth_cache.cc @@ -50,6 +50,7 @@ #include "sql/auth/auth_acls.h" #include "sql/auth/auth_common.h" // ACL_internal_schema_access #include "sql/auth/auth_internal.h" // auth_plugin_is_built_in +#include "sql/auth/auth_plugin_shutdown.h" #include "sql/auth/auth_utility.h" #include "sql/auth/dynamic_privilege_table.h" #include "sql/auth/sql_authentication.h" // g_cached_authentication_plugins @@ -1973,12 +1974,17 @@ bool set_user_salt(ACL_USER *acl_user) { MYSQL_AUTHENTICATION_PLUGIN); if (plugin) { st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(plugin)->info; + Auth_plugin_operation_guard op_guard; - for (int i = 0; i < NUM_CREDENTIALS && !result; ++i) { - result = auth->set_salt(acl_user->credentials[i].m_auth_string.str, - acl_user->credentials[i].m_auth_string.length, - acl_user->credentials[i].m_salt, - &acl_user->credentials[i].m_salt_len); + if (!op_guard) { + result = true; + } else { + for (int i = 0; i < NUM_CREDENTIALS && !result; ++i) { + result = auth->set_salt(acl_user->credentials[i].m_auth_string.str, + acl_user->credentials[i].m_auth_string.length, + acl_user->credentials[i].m_salt, + &acl_user->credentials[i].m_salt_len); + } } plugin_unlock(nullptr, plugin); } diff --git a/sql/auth/sql_authentication.cc b/sql/auth/sql_authentication.cc index e19ccedff6e5..913104335f47 100644 --- a/sql/auth/sql_authentication.cc +++ b/sql/auth/sql_authentication.cc @@ -38,6 +38,9 @@ #include #include #include +#include +#include +#include #include /* std::string */ #include #include /* std::vector */ @@ -77,6 +80,7 @@ #include "sql/auth/auth_acls.h" #include "sql/auth/auth_common.h" #include "sql/auth/auth_internal.h" // optimize_plugin_compare_by_pointer +#include "sql/auth/auth_plugin_shutdown.h" #include "sql/auth/partial_revokes.h" #include "sql/auth/sql_auth_cache.h" // acl_cache #include "sql/auth/sql_security_ctx.h" @@ -1286,6 +1290,54 @@ int security_level(void) { external_roles_t g_external_roles; Cached_authentication_plugins *g_cached_authentication_plugins = nullptr; +namespace { + +class Auth_plugin_shutdown_state { + public: + bool begin_operation() { + std::lock_guard lock(m_mutex); + if (m_shutting_down) return false; + ++m_active_operations; + return true; + } + + void end_operation() { + std::lock_guard lock(m_mutex); + assert(m_active_operations > 0); + if (--m_active_operations == 0) m_cv.notify_all(); + } + + void start_shutdown_and_wait() { + std::unique_lock lock(m_mutex); + m_shutting_down = true; + m_cv.wait_for(lock, + std::chrono::milliseconds(AUTH_PLUGIN_SHUTDOWN_TIMEOUT_MS), + [&] { return m_active_operations == 0; }); + } + + private: + std::mutex m_mutex; + std::condition_variable m_cv; + size_t m_active_operations = 0; + bool m_shutting_down = false; +}; + +Auth_plugin_shutdown_state g_auth_plugin_shutdown_state; + +} // namespace + +bool begin_auth_plugin_operation() { + return g_auth_plugin_shutdown_state.begin_operation(); +} + +void end_auth_plugin_operation() { + g_auth_plugin_shutdown_state.end_operation(); +} + +void start_auth_plugin_shutdown_and_wait() { + g_auth_plugin_shutdown_state.start_shutdown_and_wait(); +} + bool disconnect_on_expired_password = true; extern bool initialized; @@ -3565,7 +3617,18 @@ static int do_auth_once(THD *thd, const LEX_CSTRING &auth_plugin_name, if (plugin) { st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(plugin)->info; - res = auth->authenticate_user(mpvio, &mpvio->auth_info); + Auth_plugin_operation_guard op_guard; + if (op_guard) { + DBUG_EXECUTE_IF("auth_plugin_before_callback_sync", { + const char act[] = "now SIGNAL auth_plugin_before_callback_entered"; + assert(!debug_sync_set_action(thd, STRING_WITH_LEN(act))); + my_sleep(2000000); + };); + res = auth->authenticate_user(mpvio, &mpvio->auth_info); + } else { + my_error(ER_SERVER_SHUTDOWN, MYF(0)); + res = CR_ERROR; + } if (unlock_plugin) plugin_unlock(thd, plugin); } else { @@ -3698,7 +3761,13 @@ static int do_multi_factor_auth(THD *thd, MPVIO_EXT *mpvio) { .auth_string_length; mpvio->status = MPVIO_EXT::START_MFA; st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(plugin)->info; - res = auth->authenticate_user(mpvio, &mpvio->auth_info); + Auth_plugin_operation_guard op_guard; + if (op_guard) { + res = auth->authenticate_user(mpvio, &mpvio->auth_info); + } else { + my_error(ER_SERVER_SHUTDOWN, MYF(0)); + res = CR_ERROR; + } if (res == CR_OK_AUTH_IN_SANDBOX_MODE) { /* Server allows user account to connect in case registration is diff --git a/sql/auth/sql_authentication.h b/sql/auth/sql_authentication.h index 847a17775ea9..97e3251fd639 100644 --- a/sql/auth/sql_authentication.h +++ b/sql/auth/sql_authentication.h @@ -250,6 +250,8 @@ using external_roles_t = std::map>; extern external_roles_t g_external_roles; extern Cached_authentication_plugins *g_cached_authentication_plugins; +constexpr unsigned int AUTH_PLUGIN_SHUTDOWN_TIMEOUT_MS = 2000; + ACL_USER *decoy_user(const LEX_CSTRING &username, const LEX_CSTRING &hostname, MEM_ROOT *mem, struct rand_struct *rand, bool is_initialized); diff --git a/sql/auth/sql_mfa.cc b/sql/auth/sql_mfa.cc index 888280728b48..cc745d6e8a5d 100644 --- a/sql/auth/sql_mfa.cc +++ b/sql/auth/sql_mfa.cc @@ -27,6 +27,7 @@ #include "mysql/components/services/log_builtins.h" #include "mysql/plugin_auth.h" +#include "sql/auth/auth_plugin_shutdown.h" #include "sql/auth/authentication_policy.h" #include "sql/auth/sql_mfa.h" #include "sql/derror.h" /* ER_THD */ @@ -643,8 +644,10 @@ bool Multi_factor_auth_info::validate_plugins_in_auth_chain( inbuflen = gen_password.length(); set_generated_password(gen_password.c_str(), gen_password.length()); } - if (auth->generate_authentication_string(outbuf, &buflen, inbuf, - inbuflen)) { + Auth_plugin_operation_guard op_guard; + if (!op_guard || auth->generate_authentication_string(outbuf, &buflen, + inbuf, inbuflen)) { + if (!op_guard) my_error(ER_SERVER_SHUTDOWN, MYF(0)); plugin_unlock(nullptr, plugin); return (true); } @@ -894,8 +897,11 @@ bool Multi_factor_auth_info::init_registration(THD *thd, uint nth_factor) { /* convert auth string to base64 to be stored in mysql.user table */ char outbuf[MAX_FIELD_WIDTH] = {0}; unsigned int outbuflen = MAX_FIELD_WIDTH; - if (auth->generate_authentication_string( + Auth_plugin_operation_guard op_guard; + if (!op_guard || + auth->generate_authentication_string( outbuf, &outbuflen, reinterpret_cast(buf), buflen)) { + if (!op_guard) my_error(ER_SERVER_SHUTDOWN, MYF(0)); if (buf) delete[] buf; plugin_unlock(nullptr, plugin); return (true); @@ -982,9 +988,12 @@ bool Multi_factor_auth_info::finish_registration(THD *thd, LEX_USER *user_name, /* convert auth string to base64 to be stored in mysql.user table */ char outbuf[MAX_FIELD_WIDTH] = {0}; unsigned int outbuflen = MAX_FIELD_WIDTH; - if (auth->generate_authentication_string( + Auth_plugin_operation_guard op_guard; + if (!op_guard || + auth->generate_authentication_string( outbuf, &outbuflen, reinterpret_cast(challenge_response), challenge_response_len)) { + if (!op_guard) my_error(ER_SERVER_SHUTDOWN, MYF(0)); plugin_unlock(nullptr, plugin); return (true); } diff --git a/sql/auth/sql_user.cc b/sql/auth/sql_user.cc index af25999cfcb5..ce025f6ed370 100644 --- a/sql/auth/sql_user.cc +++ b/sql/auth/sql_user.cc @@ -102,6 +102,7 @@ #include "prealloced_array.h" #include "sql/auth/auth_internal.h" +#include "sql/auth/auth_plugin_shutdown.h" #include "sql/auth/sql_auth_cache.h" #include "sql/auth/sql_authentication.h" #include "sql/auth/sql_mfa.h" @@ -632,20 +633,27 @@ static bool auth_verify_password_history( */ if (cleartext_length && cleartext && 0 == (what_to_set & DIFFERENT_PLUGIN_ATTR) && - (auth->authentication_flags & AUTH_FLAG_USES_INTERNAL_STORAGE) && - auth->validate_authentication_string && - !auth->validate_authentication_string(cred_val.c_ptr_safe(), - (unsigned)cred_val.length()) && - auth->compare_password_with_hash && - !auth->compare_password_with_hash( - cred_val.c_ptr_safe(), (unsigned long)cred_val.length(), - cleartext, (unsigned long)cleartext_length, &is_error) && - !is_error) { - my_error(ER_CREDENTIALS_CONTRADICT_TO_HISTORY, MYF(0), user->length, - user->str, host->length, host->str); - /* password found in history */ - result = true; - goto end; + (auth->authentication_flags & AUTH_FLAG_USES_INTERNAL_STORAGE)) { + Auth_plugin_operation_guard op_guard; + if (!op_guard) { + my_error(ER_SERVER_SHUTDOWN, MYF(0)); + result = true; + goto end; + } + if (auth->validate_authentication_string && + !auth->validate_authentication_string( + cred_val.c_ptr_safe(), (unsigned)cred_val.length()) && + auth->compare_password_with_hash && + !auth->compare_password_with_hash( + cred_val.c_ptr_safe(), (unsigned long)cred_val.length(), + cleartext, (unsigned long)cleartext_length, &is_error) && + !is_error) { + my_error(ER_CREDENTIALS_CONTRADICT_TO_HISTORY, MYF(0), user->length, + user->str, host->length, host->str); + /* password found in history */ + result = true; + goto end; + } } } @@ -893,16 +901,22 @@ static bool validate_password_require_current( current auth string. */ if ((auth->authentication_flags & AUTH_FLAG_USES_INTERNAL_STORAGE) && - auth->compare_password_with_hash && - auth->compare_password_with_hash( - acl_user->credentials[PRIMARY_CRED].m_auth_string.str, - (unsigned long)acl_user->credentials[PRIMARY_CRED] - .m_auth_string.length, - Str->current_auth.str, (unsigned long)Str->current_auth.length, - &is_error) && - !is_error) { - my_error(ER_INCORRECT_CURRENT_PASSWORD, MYF(0)); - return (true); + auth->compare_password_with_hash) { + Auth_plugin_operation_guard op_guard; + if (!op_guard) { + my_error(ER_SERVER_SHUTDOWN, MYF(0)); + return (true); + } + if (auth->compare_password_with_hash( + acl_user->credentials[PRIMARY_CRED].m_auth_string.str, + (unsigned long)acl_user->credentials[PRIMARY_CRED] + .m_auth_string.length, + Str->current_auth.str, (unsigned long)Str->current_auth.length, + &is_error) && + !is_error) { + my_error(ER_INCORRECT_CURRENT_PASSWORD, MYF(0)); + return (true); + } } { @@ -1371,8 +1385,10 @@ bool set_and_validate_user_attributes( if (Str->first_factor_auth_info.uses_identified_by_clause) { inbuf = Str->first_factor_auth_info.auth.str; inbuflen = (unsigned)Str->first_factor_auth_info.auth.length; - if (auth->generate_authentication_string(outbuf, &buflen, inbuf, - inbuflen)) { + Auth_plugin_operation_guard op_guard; + if (!op_guard || auth->generate_authentication_string(outbuf, &buflen, + inbuf, inbuflen)) { + if (!op_guard) my_error(ER_SERVER_SHUTDOWN, MYF(0)); plugin_unlock(nullptr, plugin); what_to_set.m_what = NONE_ATTR; /* @@ -1390,10 +1406,15 @@ bool set_and_validate_user_attributes( Str->first_factor_auth_info.auth = {password, buflen}; } else if (Str->first_factor_auth_info.uses_authentication_string_clause) { assert(!is_role); - if (auth->validate_authentication_string( + Auth_plugin_operation_guard op_guard; + if (!op_guard || + auth->validate_authentication_string( const_cast(Str->first_factor_auth_info.auth.str), (unsigned)Str->first_factor_auth_info.auth.length)) { - my_error(ER_PASSWORD_FORMAT, MYF(0)); + if (!op_guard) + my_error(ER_SERVER_SHUTDOWN, MYF(0)); + else + my_error(ER_PASSWORD_FORMAT, MYF(0)); plugin_unlock(nullptr, plugin); what_to_set.m_what = NONE_ATTR; return true; @@ -1749,13 +1770,16 @@ bool set_and_validate_user_attributes( std::string(Str->host.str), gen_password, 1}; generated_passwords.push_back(p); } - if (auth->generate_authentication_string(outbuf, &buflen, inbuf, + Auth_plugin_operation_guard op_guard; + if (!op_guard || + auth->generate_authentication_string(outbuf, &buflen, inbuf, inbuflen) || auth_verify_password_history(thd, &Str->user, &Str->host, Str->alter_status.password_history_length, Str->alter_status.password_reuse_interval, auth, inbuf, inbuflen, outbuf, buflen, history_table, what_to_set.m_what)) { + if (!op_guard) my_error(ER_SERVER_SHUTDOWN, MYF(0)); plugin_unlock(nullptr, plugin); what_to_set.m_what = NONE_ATTR; /* @@ -1817,10 +1841,15 @@ bool set_and_validate_user_attributes( interdependencies if mysql_create_user() is refactored. */ assert(!is_role); - if (auth->validate_authentication_string( + Auth_plugin_operation_guard op_guard; + if (!op_guard || + auth->validate_authentication_string( const_cast(Str->first_factor_auth_info.auth.str), (unsigned)Str->first_factor_auth_info.auth.length)) { - my_error(ER_PASSWORD_FORMAT, MYF(0)); + if (!op_guard) + my_error(ER_SERVER_SHUTDOWN, MYF(0)); + else + my_error(ER_PASSWORD_FORMAT, MYF(0)); plugin_unlock(nullptr, plugin); what_to_set.m_what = NONE_ATTR; return (true); diff --git a/sql/sql_plugin.cc b/sql/sql_plugin.cc index d5fe903e91b7..3b9b731a11b7 100644 --- a/sql/sql_plugin.cc +++ b/sql/sql_plugin.cc @@ -71,6 +71,7 @@ #include "prealloced_array.h" #include "sql/auth/auth_acls.h" #include "sql/auth/auth_common.h" // check_table_access +#include "sql/auth/auth_plugin_shutdown.h" #include "sql/auto_thd.h" // Auto_THD #include "sql/current_thd.h" #include "sql/dd/cache/dictionary_client.h" // dd::cache::Dictionary_client @@ -2054,6 +2055,10 @@ void plugin_shutdown() { if (initialized) { size_t count = plugin_array->size(); + + // Stop new auth plugin operations and drain in-flight callbacks first. + start_auth_plugin_shutdown_and_wait(); + mysql_mutex_lock(&LOCK_plugin); reap_needed = true; From 9226854f5ee0222ad9385abd5ef2e271744a6fd5 Mon Sep 17 00:00:00 2001 From: Oleksiy Lukin Date: Tue, 7 Jul 2026 12:42:10 +0300 Subject: [PATCH 15/32] PS-10102 [9.7] fix of crash when KMIP server is unreachable https://perconadev.atlassian.net/browse/PS-10102 When component_keyring_kmip is loaded but keyring initialization fails (e.g. KMIP server not running or misconfigured), g_keyring_operations remains a null unique_ptr. Any subsequent call to a keyring service method (generate, store, remove, read, encrypt, decrypt, iterator) dereferences this null pointer, triggering an assertion in unique_ptr::operator*() and aborting the server. Add a null guard at the start of every service method that dereferences g_keyring_operations, returning true (error) immediately when the keyring has not been successfully initialized. --- .../keyring_encryption_service_definition.cc | 2 ++ .../keyring_generator_service_definition.cc | 1 + .../keyring_keys_metadata_iterator_service_definition.cc | 6 ++++++ .../keyring_reader_service_definition.cc | 4 ++++ .../keyring_writer_service_definition.cc | 2 ++ 5 files changed, 15 insertions(+) diff --git a/components/keyrings/keyring_kmip/service_implementation/keyring_encryption_service_definition.cc b/components/keyrings/keyring_kmip/service_implementation/keyring_encryption_service_definition.cc index 089d1974f91f..33947b90bf6c 100644 --- a/components/keyrings/keyring_kmip/service_implementation/keyring_encryption_service_definition.cc +++ b/components/keyrings/keyring_kmip/service_implementation/keyring_encryption_service_definition.cc @@ -49,6 +49,7 @@ DEFINE_BOOL_METHOD(Keyring_aes_service_impl::encrypt, const unsigned char *data_buffer, size_t data_buffer_length, unsigned char *out_buffer, size_t out_buffer_length, size_t *out_length)) { + if (!g_keyring_operations) return true; return aes_encrypt_template< Keyring_kmip_backend, keyring_common::data::Data_extension>( @@ -63,6 +64,7 @@ DEFINE_BOOL_METHOD(Keyring_aes_service_impl::decrypt, const unsigned char *data_buffer, size_t data_buffer_length, unsigned char *out_buffer, size_t out_buffer_length, size_t *out_length)) { + if (!g_keyring_operations) return true; return aes_decrypt_template< Keyring_kmip_backend, keyring_common::data::Data_extension>( diff --git a/components/keyrings/keyring_kmip/service_implementation/keyring_generator_service_definition.cc b/components/keyrings/keyring_kmip/service_implementation/keyring_generator_service_definition.cc index 5a07ad18f482..7420fc3e5887 100644 --- a/components/keyrings/keyring_kmip/service_implementation/keyring_generator_service_definition.cc +++ b/components/keyrings/keyring_kmip/service_implementation/keyring_generator_service_definition.cc @@ -38,6 +38,7 @@ namespace service_definition { DEFINE_BOOL_METHOD(Keyring_generator_service_impl::generate, (const char *data_id, const char *auth_id, const char *data_type, size_t data_size)) { + if (!g_keyring_operations) return true; return generate_template( data_id, auth_id, data_type, data_size, *g_keyring_operations, *g_component_callbacks); diff --git a/components/keyrings/keyring_kmip/service_implementation/keyring_keys_metadata_iterator_service_definition.cc b/components/keyrings/keyring_kmip/service_implementation/keyring_keys_metadata_iterator_service_definition.cc index ebc19834f633..9e407310d22b 100644 --- a/components/keyrings/keyring_kmip/service_implementation/keyring_keys_metadata_iterator_service_definition.cc +++ b/components/keyrings/keyring_kmip/service_implementation/keyring_keys_metadata_iterator_service_definition.cc @@ -46,6 +46,7 @@ using keyring_kmip::IdExt; namespace service_definition { DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::init, (my_h_keyring_keys_metadata_iterator * forward_iterator)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; bool retval = init_keys_metadata_iterator_template( it, *g_keyring_operations, *g_component_callbacks); @@ -57,6 +58,7 @@ DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::init, DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::deinit, (my_h_keyring_keys_metadata_iterator forward_iterator)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset( reinterpret_cast> *>(forward_iterator)); @@ -66,6 +68,7 @@ DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::deinit, DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::is_valid, (my_h_keyring_keys_metadata_iterator forward_iterator)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset( reinterpret_cast> *>(forward_iterator)); @@ -78,6 +81,7 @@ DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::is_valid, DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::next, (my_h_keyring_keys_metadata_iterator forward_iterator)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset( reinterpret_cast> *>(forward_iterator)); @@ -91,6 +95,7 @@ DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::next, DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::get_length, (my_h_keyring_keys_metadata_iterator forward_iterator, size_t *data_id_length, size_t *auth_id_length)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset( reinterpret_cast> *>(forward_iterator)); @@ -106,6 +111,7 @@ DEFINE_BOOL_METHOD(Keyring_keys_metadata_iterator_service_impl::get, (my_h_keyring_keys_metadata_iterator forward_iterator, char *data_id, size_t data_id_length, char *auth_id, size_t auth_id_length)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset( reinterpret_cast> *>(forward_iterator)); diff --git a/components/keyrings/keyring_kmip/service_implementation/keyring_reader_service_definition.cc b/components/keyrings/keyring_kmip/service_implementation/keyring_reader_service_definition.cc index e74c22f35dd4..ebdd168f1341 100644 --- a/components/keyrings/keyring_kmip/service_implementation/keyring_reader_service_definition.cc +++ b/components/keyrings/keyring_kmip/service_implementation/keyring_reader_service_definition.cc @@ -44,6 +44,7 @@ namespace service_definition { DEFINE_BOOL_METHOD(Keyring_reader_service_impl::init, (const char *data_id, const char *auth_id, my_h_keyring_reader_object *reader_object)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; int retval = init_reader_template( data_id, auth_id, it, *g_keyring_operations, *g_component_callbacks); @@ -55,6 +56,7 @@ DEFINE_BOOL_METHOD(Keyring_reader_service_impl::init, DEFINE_BOOL_METHOD(Keyring_reader_service_impl::deinit, (my_h_keyring_reader_object reader_object)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset(reinterpret_cast> *>(reader_object)); return deinit_reader_template(it, *g_keyring_operations, @@ -64,6 +66,7 @@ DEFINE_BOOL_METHOD(Keyring_reader_service_impl::deinit, DEFINE_BOOL_METHOD(Keyring_reader_service_impl::fetch_length, (my_h_keyring_reader_object reader_object, size_t *data_size, size_t *data_type_size)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset(reinterpret_cast> *>(reader_object)); bool retval = fetch_length_template( @@ -79,6 +82,7 @@ DEFINE_BOOL_METHOD(Keyring_reader_service_impl::fetch, unsigned char *data_buffer, size_t data_buffer_length, size_t *data_size, char *data_type_buffer, size_t data_type_buffer_length, size_t *data_type_size)) { + if (!g_keyring_operations) return true; std::unique_ptr>> it; it.reset(reinterpret_cast> *>(reader_object)); bool retval = fetch_template( diff --git a/components/keyrings/keyring_kmip/service_implementation/keyring_writer_service_definition.cc b/components/keyrings/keyring_kmip/service_implementation/keyring_writer_service_definition.cc index 56bd409ed07b..0e2b73c985e1 100644 --- a/components/keyrings/keyring_kmip/service_implementation/keyring_writer_service_definition.cc +++ b/components/keyrings/keyring_kmip/service_implementation/keyring_writer_service_definition.cc @@ -39,6 +39,7 @@ DEFINE_BOOL_METHOD(Keyring_writer_service_impl::store, (const char *data_id, const char *auth_id, const unsigned char *data, size_t data_size, const char *data_type)) { + if (!g_keyring_operations) return true; return store_template(data_id, auth_id, data, data_size, data_type, *g_keyring_operations, *g_component_callbacks); @@ -46,6 +47,7 @@ DEFINE_BOOL_METHOD(Keyring_writer_service_impl::store, DEFINE_BOOL_METHOD(Keyring_writer_service_impl::remove, (const char *data_id, const char *auth_id)) { + if (!g_keyring_operations) return true; return remove_template( data_id, auth_id, *g_keyring_operations, *g_component_callbacks); } From 4789bc29423dd4893f177d8c5a1818372360f0d7 Mon Sep 17 00:00:00 2001 From: Oleksiy Lukin Date: Thu, 9 Jul 2026 14:33:18 +0300 Subject: [PATCH 16/32] PS-10102 [9.7] fix of component_keyring_kmip MTR tests https://perconadev.atlassian.net/browse/PS-10102 Various fixes of component_keyring_kmip MTR tests to run successfully on PyKMIP and CosmianKMS servers. Some tests are excluded from parallel run. The CosmianKMS KMIP server now is main for testing, but PyKMIP server is still supported --- .../inc/keyring_udf_test_kmip.inc | 17 +++++++-- .../inc/setup_component.inc | 35 +++++++++++++++++-- .../inc/setup_component_customized.inc | 22 +++++++++--- .../inc/teardown_component.inc | 14 +++++++- .../inc/teardown_component_customized.inc | 16 +++++++++ .../t/clone_remote_encrypt.test | 1 + .../t/encrypt_explicit.test | 1 - .../t/keyring_component_status.test | 31 +++++++++++++++- .../t/log_encrypt_1.test | 3 +- .../t/mysql_ts_alter_encrypt_1.test | 1 + .../t/mysql_ts_alter_encrypt_2.test | 1 + .../t/table_encrypt_2.test | 1 + .../t/table_encrypt_3.result | 2 +- .../t/table_encrypt_3.test | 1 + .../t/table_encrypt_4.test | 1 + .../t/tablespace_encrypt_10.test | 1 + .../t/tablespace_encrypt_11.test | 1 + .../t/tablespace_encrypt_3.test | 1 + .../t/tablespace_encrypt_7.test | 1 + 19 files changed, 136 insertions(+), 15 deletions(-) diff --git a/mysql-test/suite/component_keyring_kmip/inc/keyring_udf_test_kmip.inc b/mysql-test/suite/component_keyring_kmip/inc/keyring_udf_test_kmip.inc index 0a7b7e549694..e9b3e7470501 100644 --- a/mysql-test/suite/component_keyring_kmip/inc/keyring_udf_test_kmip.inc +++ b/mysql-test/suite/component_keyring_kmip/inc/keyring_udf_test_kmip.inc @@ -20,6 +20,20 @@ --echo # ---------------------------------------------------------------------- +--disable_result_log +--disable_query_log +--disable_abort_on_error +# Best-effort cleanup for strict KMIP servers that reject duplicate key names. +SELECT keyring_key_remove('AES_g1'); +SELECT keyring_key_remove('AES_g2'); +SELECT keyring_key_remove('AES_g3'); +SELECT keyring_key_remove('AES_s1'); +SELECT keyring_key_remove('AES_s2'); +SELECT keyring_key_remove('AES_s3'); +--enable_abort_on_error +--enable_query_log +--enable_result_log + --echo # Tests for AES key type --echo # keyring_key_generate tests SELECT keyring_key_generate('AES_g1', 'AES', 16); @@ -109,6 +123,3 @@ DROP FUNCTION keyring_key_type_fetch; DROP FUNCTION keyring_key_length_fetch; UNINSTALL PLUGIN keyring_udf; --echo # ---------------------------------------------------------------------- - - - diff --git a/mysql-test/suite/component_keyring_kmip/inc/setup_component.inc b/mysql-test/suite/component_keyring_kmip/inc/setup_component.inc index 5da074b5d224..56f4ec2b5a8d 100644 --- a/mysql-test/suite/component_keyring_kmip/inc/setup_component.inc +++ b/mysql-test/suite/component_keyring_kmip/inc/setup_component.inc @@ -10,6 +10,21 @@ --echo # ---------------------------------------------------------------------- --echo # Setup +# Ensure the standard binary manifest (read_local_manifest: true) is in place. +# A previous customized-setup test may have replaced or removed it; restoring +# it here guarantees that the instance manifest is always read at server start. +--perl + use strict; + use File::Basename; + my $content = '{ "read_local_manifest": true }'; + my $manifest_file_ext = ".my"; + my ($exename, $path, $suffix) = fileparse($ENV{'MYSQLD'}, qr/\.[^.]*/); + my $manifest_file_path = $path.$exename.$manifest_file_ext; + open(my $mh, "> $manifest_file_path") or die "Cannot write $manifest_file_path: $!"; + print $mh $content or die; + close($mh) or die; +EOF + --let PLUGIN_DIR_OPT = $KEYRING_KMIP_COMPONENT_OPT # Data directory location @@ -22,8 +37,23 @@ # Create local keyring config --let KEYRING_KMIP_PATH = `SELECT CONCAT( '$MYSQLTEST_VARDIR', '/keyring_kmip')` ---let KEYRING_CONFIG_CONTENT = `SELECT CONCAT('{ \"path\": \"', '$KEYRING_KMIP_PATH', '\", \"server_addr\": \"', '$KMIP_ADDR', '\", \"server_port\": \"', '$KMIP_PORT', '\", \"client_ca\": \"', '$KMIP_CLIENT_CA', '\", \"client_key\": \"', '$KMIP_CLIENT_KEY', '\", \"server_ca\": \"', '$KMIP_SERVER_CA', '\", \"object_group\": \"', 'test-object-group', '\" }')` ---source include/keyring_tests/helper/local_keyring_create_config.inc +--echo # Creating local configuration file for keyring component: $COMPONENT_NAME +--perl + use strict; + my $vardir = $ENV{'MYSQLTEST_VARDIR'}; + my $config_file = $ENV{'CURRENT_DATADIR'} . "/" . $ENV{'COMPONENT_NAME'} . ".cnf"; + open my $uuid_fh, '<', '/proc/sys/kernel/random/uuid' or die "Cannot open /proc/sys/kernel/random/uuid: $!"; + my $uuid = <$uuid_fh>; + close $uuid_fh or die "Cannot close /proc/sys/kernel/random/uuid: $!"; + chomp $uuid; + $uuid =~ s/-//g; + $uuid = substr($uuid, 0, 24); + my $keyring_path = $vardir . '/keyring_kmip_' . $uuid; + my $config_content = '{ "path": "' . $keyring_path . '", "server_addr": "127.0.0.1", "server_port": "5696", "client_ca": "/tmp/certs/mysql-client-cert.pem", "client_key": "/tmp/certs/mysql-client-key.pem", "server_ca": "/tmp/certs/vault-kmip-ca.pem", "object_group": "test-object-group-' . $uuid . '" }'; + open my $cfh, '>', $config_file or die "Cannot write $config_file: $!"; + print $cfh $config_content or die "Cannot write $config_file: $!"; + close $cfh or die "Cannot close $config_file: $!"; +EOF # Create local manifest file for current server instance --let LOCAL_MANIFEST_CONTENT = `SELECT CONCAT('{ \"components\": \"file://', '$COMPONENT_NAME', '\" }')` @@ -32,4 +62,3 @@ # Restart server with manifest file --source include/keyring_tests/helper/start_server_with_manifest.inc --echo # ---------------------------------------------------------------------- - diff --git a/mysql-test/suite/component_keyring_kmip/inc/setup_component_customized.inc b/mysql-test/suite/component_keyring_kmip/inc/setup_component_customized.inc index 0ebd2bb87ab0..7f1f347e4524 100644 --- a/mysql-test/suite/component_keyring_kmip/inc/setup_component_customized.inc +++ b/mysql-test/suite/component_keyring_kmip/inc/setup_component_customized.inc @@ -13,11 +13,25 @@ --source include/keyring_tests/helper/binary_create_customized_manifest.inc # Create global keyring config ---let KEYRING_KMIP_PATH = `SELECT CONCAT( '$MYSQLTEST_VARDIR', '/keyring_kmip')` ---let KEYRING_CONFIG_CONTENT = `SELECT CONCAT('{ \"path\": \"', '$KEYRING_KMIP_PATH','\", \"server_addr\": \"127.0.0.1\", \"server_port\": \"5696\", \"client_ca\": \"/tmp/certs/mysql-client-cert.pem\", \"client_key\":\"/tmp/certs/mysql-client-key.pem\", \"server_ca\":\"/tmp/certs/vault-kmip-ca.pem\" }')` ---source include/keyring_tests/helper/global_keyring_create_customized_config.inc +--let KEYRING_KMIP_PATH = `SELECT CONCAT( '$MYSQLTEST_VARDIR', '/keyring_kmip_customized')` +--echo # Creating custom global configuration file for keyring component: $COMPONENT_NAME +--perl + use strict; + my $vardir = $ENV{'MYSQLTEST_VARDIR'}; + my $config_file = $ENV{'COMPONENT_DIR'} . "/" . $ENV{'COMPONENT_NAME'} . ".cnf"; + open my $uuid_fh, '<', '/proc/sys/kernel/random/uuid' or die "Cannot open /proc/sys/kernel/random/uuid: $!"; + my $uuid = <$uuid_fh>; + close $uuid_fh or die "Cannot close /proc/sys/kernel/random/uuid: $!"; + chomp $uuid; + $uuid =~ s/-//g; + $uuid = substr($uuid, 0, 16); + my $keyring_path = $vardir . '/keyring_kmip_customized_' . $uuid; + my $config_content = '{ "path": "' . $keyring_path . '", "server_addr": "127.0.0.1", "server_port": "5696", "client_ca": "/tmp/certs/mysql-client-cert.pem", "client_key": "/tmp/certs/mysql-client-key.pem", "server_ca": "/tmp/certs/vault-kmip-ca.pem", "object_group": "test-object-group-customized-' . $uuid . '" }'; + open my $cfh, '>', $config_file or die "Cannot write $config_file: $!"; + print $cfh $config_content or die "Cannot write $config_file: $!"; + close $cfh or die "Cannot close $config_file: $!"; +EOF # Restart server with manifest file --source include/keyring_tests/helper/start_server_with_manifest.inc --echo # ---------------------------------------------------------------------- - diff --git a/mysql-test/suite/component_keyring_kmip/inc/teardown_component.inc b/mysql-test/suite/component_keyring_kmip/inc/teardown_component.inc index ddd78d7e55bb..30c47255751c 100644 --- a/mysql-test/suite/component_keyring_kmip/inc/teardown_component.inc +++ b/mysql-test/suite/component_keyring_kmip/inc/teardown_component.inc @@ -2,6 +2,17 @@ --echo # ---------------------------------------------------------------------- --echo # Teardown + +# Tests that enable innodb_redo_log_encrypt leave the redo log encrypted. +# Restart without encryption first so the subsequent inter-test restart can +# start InnoDB without needing the keyring to recover the redo log. +if ($KEYRING_KMIP_USE_BINARY_MANIFEST) +{ + let $restart_parameters = restart: $PLUGIN_DIR_OPT; + --source include/restart_mysqld_no_echo.inc + --let $KEYRING_KMIP_USE_BINARY_MANIFEST= +} + # Remove local manifest file for current server instance --source include/keyring_tests/helper/instance_remove_manifest.inc @@ -11,7 +22,8 @@ # Remove local keyring config --source include/keyring_tests/helper/local_keyring_remove_config.inc -# Remove global keyring config +# Keep shared global keyring config in place to avoid parallel test races. +# Preserve the standard teardown output for existing .result files. --source include/keyring_tests/helper/global_keyring_remove_config.inc # Restart server without manifest file diff --git a/mysql-test/suite/component_keyring_kmip/inc/teardown_component_customized.inc b/mysql-test/suite/component_keyring_kmip/inc/teardown_component_customized.inc index e91ff69d5d90..b1418148a3a4 100644 --- a/mysql-test/suite/component_keyring_kmip/inc/teardown_component_customized.inc +++ b/mysql-test/suite/component_keyring_kmip/inc/teardown_component_customized.inc @@ -2,6 +2,7 @@ --echo # ---------------------------------------------------------------------- --echo # Teardown +# Remove keyring kmip --source include/keyring_tests/helper/local_keyring_kmip_remove.inc # Drop global keyring config @@ -10,6 +11,21 @@ # Remove manifest file for mysqld binary --source include/keyring_tests/helper/binary_remove_manifest.inc +# Restore the standard binary manifest so subsequent tests can read their +# instance manifests (MySQL only reads instance manifests when the binary +# manifest says read_local_manifest: true). +--perl + use strict; + use File::Basename; + my $content = '{ "read_local_manifest": true }'; + my $manifest_file_ext = ".my"; + my ($exename, $path, $suffix) = fileparse($ENV{'MYSQLD'}, qr/\.[^.]*/); + my $manifest_file_path = $path.$exename.$manifest_file_ext; + open(my $mh, "> $manifest_file_path") or die "Cannot write $manifest_file_path: $!"; + print $mh $content or die; + close($mh) or die; +EOF + # Restart server without manifest file --source include/keyring_tests/helper/cleanup_server_with_manifest.inc --echo # ---------------------------------------------------------------------- diff --git a/mysql-test/suite/component_keyring_kmip/t/clone_remote_encrypt.test b/mysql-test/suite/component_keyring_kmip/t/clone_remote_encrypt.test index 3584dc8527ba..d7af8a82cc1c 100644 --- a/mysql-test/suite/component_keyring_kmip/t/clone_remote_encrypt.test +++ b/mysql-test/suite/component_keyring_kmip/t/clone_remote_encrypt.test @@ -1,6 +1,7 @@ --source include/have_component_keyring_kmip.inc # Test remote clone with different table types with debug sync +--source include/not_parallel.inc --source include/have_innodb_max_16k.inc --let $HOST = 127.0.0.1 diff --git a/mysql-test/suite/component_keyring_kmip/t/encrypt_explicit.test b/mysql-test/suite/component_keyring_kmip/t/encrypt_explicit.test index 028ec78ea118..1ea893fb68e9 100644 --- a/mysql-test/suite/component_keyring_kmip/t/encrypt_explicit.test +++ b/mysql-test/suite/component_keyring_kmip/t/encrypt_explicit.test @@ -4,7 +4,6 @@ --source include/have_innodb_16k.inc # This test must be not be run in parallel with other tests because # it requires global manifest file to load keyring component directly - --source ../inc/setup_component_customized.inc --source include/keyring_tests/innodb/encrypt_explicit.inc --source ../inc/teardown_component_customized.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/keyring_component_status.test b/mysql-test/suite/component_keyring_kmip/t/keyring_component_status.test index a762de00cac2..8ed33ee9f3b6 100644 --- a/mysql-test/suite/component_keyring_kmip/t/keyring_component_status.test +++ b/mysql-test/suite/component_keyring_kmip/t/keyring_component_status.test @@ -1,5 +1,34 @@ --source include/have_component_keyring_kmip.inc --source ../inc/setup_component.inc ---source include/keyring_tests/mats/keyring_component_status.inc +# Tests related to performance_schema.keyring_component_status table + +CALL mtr.add_suppression("No suitable 'keyring_component_metadata_query' service implementation found to fulfill the request"); + +--echo # +--echo # Bug#32390719: QUERYING KEYRING_COMPONENT_STATUS TABLE +--echo # GENERATES AN ERROR IN THE SERVER LOG +--echo # + +# Should show metadata obtained from keyring component +--replace_result $MYSQLTEST_VARDIR MYSQLTEST_VARDIR +--replace_regex /test-object-group-[0-9a-f]+/test-object-group/ +SELECT * FROM performance_schema.keyring_component_status; + +# Should be empty +SELECT PRIO, ERROR_CODE, SUBSYSTEM, DATA FROM performance_schema.error_log WHERE ERROR_CODE='MY-013712'; + +--echo # Restarting server without keyring component +--source include/keyring_tests/helper/instance_backup_manifest.inc +let $restart_parameters = restart: $PLUGIN_DIR_OPT; +--source include/restart_mysqld_no_echo.inc + +# Should be empty +--replace_regex /test-object-group-[0-9a-f]+/test-object-group/ +SELECT * FROM performance_schema.keyring_component_status; + +# Should show one row +SELECT PRIO, ERROR_CODE, SUBSYSTEM, DATA FROM performance_schema.error_log WHERE ERROR_CODE='MY-013712'; + +--source include/keyring_tests/helper/instance_restore_manifest.inc --source ../inc/teardown_component.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/log_encrypt_1.test b/mysql-test/suite/component_keyring_kmip/t/log_encrypt_1.test index 9b003b6e42de..d3645b315618 100644 --- a/mysql-test/suite/component_keyring_kmip/t/log_encrypt_1.test +++ b/mysql-test/suite/component_keyring_kmip/t/log_encrypt_1.test @@ -4,7 +4,8 @@ --source include/no_valgrind_without_big.inc --source include/have_innodb_max_16k.inc - +--source include/not_parallel.inc +--let $KEYRING_KMIP_USE_BINARY_MANIFEST= 1 --source ../inc/setup_component.inc --source include/keyring_tests/mats/log_encrypt_1.inc --source ../inc/teardown_component.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_1.test b/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_1.test index 086a0975fc73..5b1b99466bfd 100644 --- a/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_1.test +++ b/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_1.test @@ -19,6 +19,7 @@ --source include/have_debug.inc # Disable in valgrind because of timeout, cf. Bug#22760145 --source include/not_valgrind.inc +--source include/not_parallel.inc --source ../inc/setup_component.inc --source include/keyring_tests/mats/mysql_ts_alter_encrypt_1.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_2.test b/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_2.test index 1a12498d4ac1..fdcdc6d0ca60 100644 --- a/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_2.test +++ b/mysql-test/suite/component_keyring_kmip/t/mysql_ts_alter_encrypt_2.test @@ -21,6 +21,7 @@ --source include/big_test.inc --source include/have_debug.inc +--source include/not_parallel.inc # Disable in valgrind because of timeout, cf. Bug#22760145 --source include/not_valgrind.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_2.test b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_2.test index 6a06f4a67598..ffeddd3d9614 100644 --- a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_2.test +++ b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_2.test @@ -2,6 +2,7 @@ # InnoDB transparent tablespace data encryption # This test case will test basic encryption support features. +--source include/not_parallel.inc --source include/no_valgrind_without_big.inc --source ../inc/setup_component.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.result b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.result index 6564f1d79983..cd0f33a675f0 100644 --- a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.result +++ b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.result @@ -824,7 +824,7 @@ CREATE PROCEDURE tde_db.rotate_master_key() begin declare i int default 1; declare has_error int default 0; -while (i <= 500) DO +while (i <= 100) DO ALTER INSTANCE ROTATE INNODB MASTER KEY; set i = i + 1; end while; diff --git a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.test b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.test index 2e1c4a6acc4d..c59309c624e8 100644 --- a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.test +++ b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_3.test @@ -2,6 +2,7 @@ # InnoDB transparent tablespace data encryption # This test case will test basic encryption support features. +--source include/not_parallel.inc --source include/no_valgrind_without_big.inc --source include/have_innodb_max_16k.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_4.test b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_4.test index a39cc24d8b6d..c9dd978eac67 100644 --- a/mysql-test/suite/component_keyring_kmip/t/table_encrypt_4.test +++ b/mysql-test/suite/component_keyring_kmip/t/table_encrypt_4.test @@ -4,6 +4,7 @@ --source include/no_valgrind_without_big.inc --source include/have_debug.inc +--source include/not_parallel.inc --source ../inc/setup_component.inc --source include/keyring_tests/innodb/table_encrypt_4.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_10.test b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_10.test index 7419ac764edf..8ea50077a6a2 100644 --- a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_10.test +++ b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_10.test @@ -1,6 +1,7 @@ --source include/have_component_keyring_kmip.inc --source include/have_debug.inc --source include/big_test.inc +--source include/not_parallel.inc --source include/no_valgrind_without_big.inc # Disable in valgrind because of timeout, cf. Bug#22760145 --source include/not_valgrind.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_11.test b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_11.test index da7c1bc9f439..81c74e8af471 100644 --- a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_11.test +++ b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_11.test @@ -4,6 +4,7 @@ --source include/no_valgrind_without_big.inc # Disable in valgrind because of timeout, cf. Bug#22760145 --source include/not_valgrind.inc +--source include/not_parallel.inc --source ../inc/setup_component.inc --source include/keyring_tests/innodb/tablespace_encrypt_11.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_3.test b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_3.test index a78047ef0521..13e5c9daa3c8 100644 --- a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_3.test +++ b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_3.test @@ -1,6 +1,7 @@ --source include/have_component_keyring_kmip.inc --source include/no_valgrind_without_big.inc --source include/have_debug.inc +--source include/not_parallel.inc --source ../inc/setup_component.inc --source include/keyring_tests/innodb/tablespace_encrypt_3.inc diff --git a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_7.test b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_7.test index fdedd67834a7..c6c146563e97 100644 --- a/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_7.test +++ b/mysql-test/suite/component_keyring_kmip/t/tablespace_encrypt_7.test @@ -1,6 +1,7 @@ --source include/have_component_keyring_kmip.inc --source include/big_test.inc --source include/have_debug.inc +--source include/not_parallel.inc # --source include/no_valgrind_without_big.inc # Disable in valgrind because of timeout, cf. Bug#22760145 --source include/not_valgrind.inc From 37ea10d855e05df3c01e3ca1c947ed2e48df2bfd Mon Sep 17 00:00:00 2001 From: Oleksiy Lukin Date: Mon, 22 Jun 2026 14:03:16 +0300 Subject: [PATCH 17/32] PS-11143-[9.7] sql/auth: quiesce auth plugin callbacks before teardown Add shutdown protection on the level into the shared auth/plugin lifecycle so all authentication plugins are covered. - Add a global auth-plugin shutdown barrier API (`sql/auth/auth_plugin_shutdown.h` + implementation in `sql/auth/sql_authentication.cc`). - Guard auth plugin callback entry points with RAII operation tracking. - Start barrier shutdown/drain from `plugin_shutdown()` before plugin deinitialization (`sql/sql_plugin.cc`). - Remove plugin-local quiesce state/wait logic from `sql/auth/sha2_password.cc` now that teardown protection is centralized. - Bound auth-plugin shutdown barrier wait with a 2s timeout (`AUTH_PLUGIN_SHUTDOWN_TIMEOUT_MS`) to avoid indefinite shutdown hangs. - Keep normal shutdown flow progressing even when auth operations fail to drain in time. Regression coverage: - Add `percona.bug_ps11143_threadpool_auth_shutdown` to reproduce the threadpool auth-vs-shutdown race and verify clean shutdown. - Include expected result file and test cleanup to preserve MTR state. Verified: - percona.bug_ps11143_threadpool_auth_shutdown - percona.signal_handling_threadpool - percona.threadpool_stats - percona.kill_idle_trx_threadpool - percona.threadpool_debug --- ...ug_ps11143_threadpool_auth_shutdown.result | 11 +++ .../bug_ps11143_threadpool_auth_shutdown.test | 45 +++++++++ sql/auth/acl_table_user.cc | 64 ++++++++++++- sql/auth/auth_plugin_shutdown.h | 47 ++++++++++ sql/auth/sql_auth_cache.cc | 16 +++- sql/auth/sql_authentication.cc | 73 ++++++++++++++- sql/auth/sql_authentication.h | 2 + sql/auth/sql_mfa.cc | 17 +++- sql/auth/sql_user.cc | 91 ++++++++++++------- sql/sql_plugin.cc | 5 + 10 files changed, 324 insertions(+), 47 deletions(-) create mode 100644 mysql-test/suite/percona/r/bug_ps11143_threadpool_auth_shutdown.result create mode 100644 mysql-test/suite/percona/t/bug_ps11143_threadpool_auth_shutdown.test create mode 100644 sql/auth/auth_plugin_shutdown.h diff --git a/mysql-test/suite/percona/r/bug_ps11143_threadpool_auth_shutdown.result b/mysql-test/suite/percona/r/bug_ps11143_threadpool_auth_shutdown.result new file mode 100644 index 000000000000..4331e438c1c0 --- /dev/null +++ b/mysql-test/suite/percona/r/bug_ps11143_threadpool_auth_shutdown.result @@ -0,0 +1,11 @@ +# +# Verify auth callback drains before plugin deinit with threadpool +# +# restart:--thread-handling=pool-of-threads --thread-pool-size=2 --thread-pool-max-threads=4 +CREATE USER tp_user@127.0.0.1 IDENTIFIED WITH caching_sha2_password BY 'tp_pwd'; +SET GLOBAL debug = "+d,auth_plugin_before_callback_sync"; +1 +1 +SET DEBUG_SYNC='now WAIT_FOR auth_plugin_before_callback_entered TIMEOUT 30'; +# restart +DROP USER IF EXISTS tp_user@127.0.0.1; diff --git a/mysql-test/suite/percona/t/bug_ps11143_threadpool_auth_shutdown.test b/mysql-test/suite/percona/t/bug_ps11143_threadpool_auth_shutdown.test new file mode 100644 index 000000000000..1e78fd1f41c4 --- /dev/null +++ b/mysql-test/suite/percona/t/bug_ps11143_threadpool_auth_shutdown.test @@ -0,0 +1,45 @@ +--source include/not_windows.inc +--source include/have_debug.inc +--source include/have_debug_sync.inc + +--echo # +--echo # Verify auth callback drains before plugin deinit with threadpool +--echo # + +--let $restart_parameters=restart:--thread-handling=pool-of-threads --thread-pool-size=2 --thread-pool-max-threads=4 +--source include/restart_mysqld.inc +--source include/have_pool_of_threads.inc + +CREATE USER tp_user@127.0.0.1 IDENTIFIED WITH caching_sha2_password BY 'tp_pwd'; + +SET GLOBAL debug = "+d,auth_plugin_before_callback_sync"; + +--let $mysqld_pid_file=`SELECT @@GLOBAL.pid_file` + +--let $output_file= $MYSQLTEST_VARDIR/tmp/bug_ps11143_mysql_output +--let $pid_file= $MYSQLTEST_VARDIR/tmp/bug_ps11143_mysql_pid +--let $sql_file= $MYSQLTEST_VARDIR/tmp/bug_ps11143_query.sql +--write_file $sql_file +SELECT 1; +EOF +--let $command= $MYSQL +--let $command_opt= --protocol=tcp --host=127.0.0.1 --port=$MASTER_MYPORT --user=tp_user --password=tp_pwd < $sql_file +--let $redirect_stderr= 1 +--source include/start_proc_in_background.inc + +SET DEBUG_SYNC='now WAIT_FOR auth_plugin_before_callback_entered TIMEOUT 30'; + +--source include/expect_crash.inc +exec kill -TERM `cat $mysqld_pid_file`; + +--source include/wait_until_disconnected.inc +--source include/wait_proc_to_finish.inc + +--let $restart_parameters= +--source include/start_mysqld.inc + +DROP USER IF EXISTS tp_user@127.0.0.1; + +--remove_file $pid_file +--remove_file $output_file +--remove_file $sql_file diff --git a/sql/auth/acl_table_user.cc b/sql/auth/acl_table_user.cc index 967dc4228466..3fee641916c8 100644 --- a/sql/auth/acl_table_user.cc +++ b/sql/auth/acl_table_user.cc @@ -47,6 +47,7 @@ Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA */ #include "sql/auth/auth_acls.h" /* ACLs */ #include "sql/auth/auth_common.h" /* User_table_schema, ... */ #include "sql/auth/auth_internal.h" /* acl_print_ha_error */ +#include "sql/auth/auth_plugin_shutdown.h" #include "sql/auth/partial_revokes.h" #include "sql/auth/sql_auth_cache.h" /* global_acl_memory */ #include "sql/auth/sql_authentication.h" /* Cached_authentication_plugins */ @@ -1628,6 +1629,56 @@ bool Acl_table_user_reader::read_plugin_info( user.plugin.str = tmpstr ? tmpstr : ""; user.plugin.length = strlen(user.plugin.str); + /* + In case we are working with 5.6 db layout we need to make server + aware of Password field and that the plugin column can be null. + In case when plugin column is null we use native password plugin + if we can. + */ + if (is_old_db_layout && (user.plugin.length == 0 || + Cached_authentication_plugins::compare_plugin( + PLUGIN_MYSQL_NATIVE_PASSWORD, user.plugin))) { + char *password = get_field( + &m_mem_root, m_table->field[m_table_schema->password_idx()]); + + // We do not support pre 4.1 hashes + plugin_ref native_plugin = + g_cached_authentication_plugins->get_cached_plugin_ref( + PLUGIN_MYSQL_NATIVE_PASSWORD); + if (native_plugin) { + const uint password_len = password ? strlen(password) : 0; + st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(native_plugin)->info; + Auth_plugin_operation_guard op_guard; + if (!op_guard) { + LogErr(WARNING_LEVEL, ER_AUTHCACHE_USER_IGNORED_INVALID_PASSWORD, + user.user ? user.user : "", + user.host.get_host() ? user.host.get_host() : ""); + return true; + } + if (auth->validate_authentication_string(password, password_len) == 0) { + // auth_string takes precedence over password + if (user.credentials[PRIMARY_CRED].m_auth_string.length == 0) { + user.credentials[PRIMARY_CRED].m_auth_string.str = password; + user.credentials[PRIMARY_CRED].m_auth_string.length = password_len; + } + if (user.plugin.length == 0) { + user.plugin.str = Cached_authentication_plugins::get_plugin_name( + PLUGIN_MYSQL_NATIVE_PASSWORD); + user.plugin.length = strlen(user.plugin.str); + } + } else { + if ((user.access & SUPER_ACL) && !super_users_with_empty_plugin && + (user.plugin.length == 0)) + super_users_with_empty_plugin = true; + + LogErr(WARNING_LEVEL, ER_AUTHCACHE_USER_IGNORED_DEPRECATED_PASSWORD, + user.user ? user.user : "", + user.host.get_host() ? user.host.get_host() : ""); + return true; + } + } + } + /* Check if the plugin string is blank or null. If it is, the user will be skipped. @@ -1652,10 +1703,11 @@ bool Acl_table_user_reader::read_plugin_info( my_plugin_lock_by_name(nullptr, user.plugin, MYSQL_AUTHENTICATION_PLUGIN); if (plugin) { st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(plugin)->info; - if (auth->validate_authentication_string( - const_cast( - user.credentials[PRIMARY_CRED].m_auth_string.str), - user.credentials[PRIMARY_CRED].m_auth_string.length)) { + Auth_plugin_operation_guard op_guard; + if (!op_guard || auth->validate_authentication_string( + const_cast( + user.credentials[PRIMARY_CRED].m_auth_string.str), + user.credentials[PRIMARY_CRED].m_auth_string.length)) { LogErr(WARNING_LEVEL, ER_AUTHCACHE_USER_IGNORED_INVALID_PASSWORD, user.user ? user.user : "", user.host.get_host() ? user.host.get_host() : ""); @@ -1870,7 +1922,9 @@ bool Acl_table_user_reader::read_user_attributes(ACL_USER &user) { if (plugin) { st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(plugin)->info; - if (auth->validate_authentication_string( + Auth_plugin_operation_guard op_guard; + if (!op_guard || + auth->validate_authentication_string( const_cast( user.credentials[SECOND_CRED].m_auth_string.str), user.credentials[SECOND_CRED].m_auth_string.length)) { diff --git a/sql/auth/auth_plugin_shutdown.h b/sql/auth/auth_plugin_shutdown.h new file mode 100644 index 000000000000..4a47f4003fe8 --- /dev/null +++ b/sql/auth/auth_plugin_shutdown.h @@ -0,0 +1,47 @@ +/* Copyright (c) 2026, Percona LLC and/or its affiliates. + + This program is free software; you can redistribute it and/or modify + it under the terms of the GNU General Public License, version 2.0, + as published by the Free Software Foundation. + + This program is designed to work with certain software (including + but not limited to OpenSSL) that is licensed under separate terms, + as designated in a particular file or component or in included license + documentation. The authors of MySQL hereby grant you an additional + permission to link the program and your derivative works with the + separately licensed software that they have either included with + the program or referenced in the documentation. + + This program is distributed in the hope that it will be useful, + but WITHOUT ANY WARRANTY; without even the implied warranty of + MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the + GNU General Public License, version 2.0, for more details. + + You should have received a copy of the GNU General Public License + along with this program; if not, write to the Free Software + Foundation, Inc., 51 Franklin St, Fifth Floor, Boston, MA 02110-1301 USA */ + +#ifndef AUTH_PLUGIN_SHUTDOWN_INCLUDED +#define AUTH_PLUGIN_SHUTDOWN_INCLUDED + +bool begin_auth_plugin_operation(); +void end_auth_plugin_operation(); +void start_auth_plugin_shutdown_and_wait(); + +class Auth_plugin_operation_guard { + public: + Auth_plugin_operation_guard() + : m_has_operation(begin_auth_plugin_operation()) {} + + ~Auth_plugin_operation_guard() { + if (m_has_operation) end_auth_plugin_operation(); + } + + bool has_operation() const { return m_has_operation; } + explicit operator bool() const { return m_has_operation; } + + private: + bool m_has_operation; +}; + +#endif // AUTH_PLUGIN_SHUTDOWN_INCLUDED diff --git a/sql/auth/sql_auth_cache.cc b/sql/auth/sql_auth_cache.cc index 406fbad197a0..69fffa0f06f3 100644 --- a/sql/auth/sql_auth_cache.cc +++ b/sql/auth/sql_auth_cache.cc @@ -51,6 +51,7 @@ #include "sql/auth/auth_acls.h" #include "sql/auth/auth_common.h" // ACL_internal_schema_access #include "sql/auth/auth_internal.h" // auth_plugin_is_built_in +#include "sql/auth/auth_plugin_shutdown.h" #include "sql/auth/auth_utility.h" #include "sql/auth/dynamic_privilege_table.h" #include "sql/auth/sql_authentication.h" // g_cached_authentication_plugins @@ -1965,12 +1966,17 @@ bool set_user_salt(ACL_USER *acl_user) { MYSQL_AUTHENTICATION_PLUGIN); if (plugin) { st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(plugin)->info; + Auth_plugin_operation_guard op_guard; - for (int i = 0; i < NUM_CREDENTIALS && !result; ++i) { - result = auth->set_salt(acl_user->credentials[i].m_auth_string.str, - acl_user->credentials[i].m_auth_string.length, - acl_user->credentials[i].m_salt, - &acl_user->credentials[i].m_salt_len); + if (!op_guard) { + result = true; + } else { + for (int i = 0; i < NUM_CREDENTIALS && !result; ++i) { + result = auth->set_salt(acl_user->credentials[i].m_auth_string.str, + acl_user->credentials[i].m_auth_string.length, + acl_user->credentials[i].m_salt, + &acl_user->credentials[i].m_salt_len); + } } plugin_unlock(nullptr, plugin); } diff --git a/sql/auth/sql_authentication.cc b/sql/auth/sql_authentication.cc index 59bd48ce2a34..0367daf22c7e 100644 --- a/sql/auth/sql_authentication.cc +++ b/sql/auth/sql_authentication.cc @@ -38,6 +38,9 @@ #include #include #include +#include +#include +#include #include /* std::string */ #include #include /* std::vector */ @@ -76,6 +79,7 @@ #include "sql/auth/auth_acls.h" #include "sql/auth/auth_common.h" #include "sql/auth/auth_internal.h" // optimize_plugin_compare_by_pointer +#include "sql/auth/auth_plugin_shutdown.h" #include "sql/auth/partial_revokes.h" #include "sql/auth/sql_auth_cache.h" // acl_cache #include "sql/auth/sql_security_ctx.h" @@ -1324,6 +1328,54 @@ int security_level(void) { external_roles_t g_external_roles; Cached_authentication_plugins *g_cached_authentication_plugins = nullptr; +namespace { + +class Auth_plugin_shutdown_state { + public: + bool begin_operation() { + std::lock_guard lock(m_mutex); + if (m_shutting_down) return false; + ++m_active_operations; + return true; + } + + void end_operation() { + std::lock_guard lock(m_mutex); + assert(m_active_operations > 0); + if (--m_active_operations == 0) m_cv.notify_all(); + } + + void start_shutdown_and_wait() { + std::unique_lock lock(m_mutex); + m_shutting_down = true; + m_cv.wait_for(lock, + std::chrono::milliseconds(AUTH_PLUGIN_SHUTDOWN_TIMEOUT_MS), + [&] { return m_active_operations == 0; }); + } + + private: + std::mutex m_mutex; + std::condition_variable m_cv; + size_t m_active_operations = 0; + bool m_shutting_down = false; +}; + +Auth_plugin_shutdown_state g_auth_plugin_shutdown_state; + +} // namespace + +bool begin_auth_plugin_operation() { + return g_auth_plugin_shutdown_state.begin_operation(); +} + +void end_auth_plugin_operation() { + g_auth_plugin_shutdown_state.end_operation(); +} + +void start_auth_plugin_shutdown_and_wait() { + g_auth_plugin_shutdown_state.start_shutdown_and_wait(); +} + bool disconnect_on_expired_password = true; extern bool initialized; @@ -3592,7 +3644,18 @@ static int do_auth_once(THD *thd, const LEX_CSTRING &auth_plugin_name, if (plugin) { st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(plugin)->info; - res = auth->authenticate_user(mpvio, &mpvio->auth_info); + Auth_plugin_operation_guard op_guard; + if (op_guard) { + DBUG_EXECUTE_IF("auth_plugin_before_callback_sync", { + const char act[] = "now SIGNAL auth_plugin_before_callback_entered"; + assert(!debug_sync_set_action(thd, STRING_WITH_LEN(act))); + my_sleep(2000000); + };); + res = auth->authenticate_user(mpvio, &mpvio->auth_info); + } else { + my_error(ER_SERVER_SHUTDOWN, MYF(0)); + res = CR_ERROR; + } if (unlock_plugin) plugin_unlock(thd, plugin); } else { @@ -3725,7 +3788,13 @@ static int do_multi_factor_auth(THD *thd, MPVIO_EXT *mpvio) { .auth_string_length; mpvio->status = MPVIO_EXT::START_MFA; st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(plugin)->info; - res = auth->authenticate_user(mpvio, &mpvio->auth_info); + Auth_plugin_operation_guard op_guard; + if (op_guard) { + res = auth->authenticate_user(mpvio, &mpvio->auth_info); + } else { + my_error(ER_SERVER_SHUTDOWN, MYF(0)); + res = CR_ERROR; + } if (res == CR_OK_AUTH_IN_SANDBOX_MODE) { /* Server allows user account to connect in case registration is diff --git a/sql/auth/sql_authentication.h b/sql/auth/sql_authentication.h index b34f3206d721..ece831c7f56f 100644 --- a/sql/auth/sql_authentication.h +++ b/sql/auth/sql_authentication.h @@ -249,6 +249,8 @@ using external_roles_t = std::map>; extern external_roles_t g_external_roles; extern Cached_authentication_plugins *g_cached_authentication_plugins; +constexpr unsigned int AUTH_PLUGIN_SHUTDOWN_TIMEOUT_MS = 2000; + ACL_USER *decoy_user(const LEX_CSTRING &username, const LEX_CSTRING &hostname, MEM_ROOT *mem, struct rand_struct *rand, bool is_initialized); diff --git a/sql/auth/sql_mfa.cc b/sql/auth/sql_mfa.cc index 0fbf43520b61..7fb3097aac17 100644 --- a/sql/auth/sql_mfa.cc +++ b/sql/auth/sql_mfa.cc @@ -27,6 +27,7 @@ #include "mysql/components/services/log_builtins.h" #include "mysql/plugin_auth.h" +#include "sql/auth/auth_plugin_shutdown.h" #include "sql/auth/authentication_policy.h" #include "sql/auth/sql_mfa.h" #include "sql/derror.h" /* ER_THD */ @@ -643,8 +644,10 @@ bool Multi_factor_auth_info::validate_plugins_in_auth_chain( inbuflen = gen_password.length(); set_generated_password(gen_password.c_str(), gen_password.length()); } - if (auth->generate_authentication_string(outbuf, &buflen, inbuf, - inbuflen)) { + Auth_plugin_operation_guard op_guard; + if (!op_guard || auth->generate_authentication_string(outbuf, &buflen, + inbuf, inbuflen)) { + if (!op_guard) my_error(ER_SERVER_SHUTDOWN, MYF(0)); plugin_unlock(nullptr, plugin); return (true); } @@ -900,8 +903,11 @@ bool Multi_factor_auth_info::init_registration(THD *thd, uint nth_factor) { /* convert auth string to base64 to be stored in mysql.user table */ char outbuf[MAX_FIELD_WIDTH] = {0}; unsigned int outbuflen = MAX_FIELD_WIDTH; - if (auth->generate_authentication_string( + Auth_plugin_operation_guard op_guard; + if (!op_guard || + auth->generate_authentication_string( outbuf, &outbuflen, reinterpret_cast(buf), buflen)) { + if (!op_guard) my_error(ER_SERVER_SHUTDOWN, MYF(0)); if (buf) delete[] buf; plugin_unlock(nullptr, plugin); return (true); @@ -988,9 +994,12 @@ bool Multi_factor_auth_info::finish_registration(THD *thd, LEX_USER *user_name, /* convert auth string to base64 to be stored in mysql.user table */ char outbuf[MAX_FIELD_WIDTH] = {0}; unsigned int outbuflen = MAX_FIELD_WIDTH; - if (auth->generate_authentication_string( + Auth_plugin_operation_guard op_guard; + if (!op_guard || + auth->generate_authentication_string( outbuf, &outbuflen, reinterpret_cast(challenge_response), challenge_response_len)) { + if (!op_guard) my_error(ER_SERVER_SHUTDOWN, MYF(0)); plugin_unlock(nullptr, plugin); return (true); } diff --git a/sql/auth/sql_user.cc b/sql/auth/sql_user.cc index 4c13d7f9a4d8..da4b9f1ca104 100644 --- a/sql/auth/sql_user.cc +++ b/sql/auth/sql_user.cc @@ -102,6 +102,7 @@ #include "prealloced_array.h" #include "sql/auth/auth_internal.h" +#include "sql/auth/auth_plugin_shutdown.h" #include "sql/auth/sql_auth_cache.h" #include "sql/auth/sql_authentication.h" #include "sql/auth/sql_mfa.h" @@ -641,20 +642,27 @@ static bool auth_verify_password_history( */ if (cleartext_length && cleartext && 0 == (what_to_set & DIFFERENT_PLUGIN_ATTR) && - (auth->authentication_flags & AUTH_FLAG_USES_INTERNAL_STORAGE) && - auth->validate_authentication_string && - !auth->validate_authentication_string(cred_val.c_ptr_safe(), - (unsigned)cred_val.length()) && - auth->compare_password_with_hash && - !auth->compare_password_with_hash( - cred_val.c_ptr_safe(), (unsigned long)cred_val.length(), - cleartext, (unsigned long)cleartext_length, &is_error) && - !is_error) { - my_error(ER_CREDENTIALS_CONTRADICT_TO_HISTORY, MYF(0), user->length, - user->str, host->length, host->str); - /* password found in history */ - result = true; - goto end; + (auth->authentication_flags & AUTH_FLAG_USES_INTERNAL_STORAGE)) { + Auth_plugin_operation_guard op_guard; + if (!op_guard) { + my_error(ER_SERVER_SHUTDOWN, MYF(0)); + result = true; + goto end; + } + if (auth->validate_authentication_string && + !auth->validate_authentication_string( + cred_val.c_ptr_safe(), (unsigned)cred_val.length()) && + auth->compare_password_with_hash && + !auth->compare_password_with_hash( + cred_val.c_ptr_safe(), (unsigned long)cred_val.length(), + cleartext, (unsigned long)cleartext_length, &is_error) && + !is_error) { + my_error(ER_CREDENTIALS_CONTRADICT_TO_HISTORY, MYF(0), user->length, + user->str, host->length, host->str); + /* password found in history */ + result = true; + goto end; + } } } @@ -902,16 +910,22 @@ static bool validate_password_require_current( current auth string. */ if ((auth->authentication_flags & AUTH_FLAG_USES_INTERNAL_STORAGE) && - auth->compare_password_with_hash && - auth->compare_password_with_hash( - acl_user->credentials[PRIMARY_CRED].m_auth_string.str, - (unsigned long)acl_user->credentials[PRIMARY_CRED] - .m_auth_string.length, - Str->current_auth.str, (unsigned long)Str->current_auth.length, - &is_error) && - !is_error) { - my_error(ER_INCORRECT_CURRENT_PASSWORD, MYF(0)); - return (true); + auth->compare_password_with_hash) { + Auth_plugin_operation_guard op_guard; + if (!op_guard) { + my_error(ER_SERVER_SHUTDOWN, MYF(0)); + return (true); + } + if (auth->compare_password_with_hash( + acl_user->credentials[PRIMARY_CRED].m_auth_string.str, + (unsigned long)acl_user->credentials[PRIMARY_CRED] + .m_auth_string.length, + Str->current_auth.str, (unsigned long)Str->current_auth.length, + &is_error) && + !is_error) { + my_error(ER_INCORRECT_CURRENT_PASSWORD, MYF(0)); + return (true); + } } { @@ -1446,8 +1460,10 @@ bool set_and_validate_user_attributes( if (Str->first_factor_auth_info.uses_identified_by_clause) { inbuf = Str->first_factor_auth_info.auth.str; inbuflen = (unsigned)Str->first_factor_auth_info.auth.length; - if (auth->generate_authentication_string(outbuf, &buflen, inbuf, - inbuflen)) { + Auth_plugin_operation_guard op_guard; + if (!op_guard || auth->generate_authentication_string(outbuf, &buflen, + inbuf, inbuflen)) { + if (!op_guard) my_error(ER_SERVER_SHUTDOWN, MYF(0)); plugin_unlock(nullptr, plugin); what_to_set.m_what = NONE_ATTR; /* @@ -1468,10 +1484,15 @@ bool set_and_validate_user_attributes( Str->first_factor_auth_info.auth = {password, buflen}; } else if (Str->first_factor_auth_info.uses_authentication_string_clause) { assert(!is_role); - if (auth->validate_authentication_string( + Auth_plugin_operation_guard op_guard; + if (!op_guard || + auth->validate_authentication_string( const_cast(Str->first_factor_auth_info.auth.str), (unsigned)Str->first_factor_auth_info.auth.length)) { - my_error(ER_PASSWORD_FORMAT, MYF(0)); + if (!op_guard) + my_error(ER_SERVER_SHUTDOWN, MYF(0)); + else + my_error(ER_PASSWORD_FORMAT, MYF(0)); plugin_unlock(nullptr, plugin); what_to_set.m_what = NONE_ATTR; return true; @@ -1794,13 +1815,16 @@ bool set_and_validate_user_attributes( std::string(Str->host.str), gen_password, 1}; generated_passwords.push_back(p); } - if (auth->generate_authentication_string(outbuf, &buflen, inbuf, + Auth_plugin_operation_guard op_guard; + if (!op_guard || + auth->generate_authentication_string(outbuf, &buflen, inbuf, inbuflen) || auth_verify_password_history(thd, &Str->user, &Str->host, Str->alter_status.password_history_length, Str->alter_status.password_reuse_interval, auth, inbuf, inbuflen, outbuf, buflen, history_table, what_to_set.m_what)) { + if (!op_guard) my_error(ER_SERVER_SHUTDOWN, MYF(0)); plugin_unlock(nullptr, plugin); what_to_set.m_what = NONE_ATTR; /* @@ -1862,10 +1886,15 @@ bool set_and_validate_user_attributes( interdependencies if mysql_create_user() is refactored. */ assert(!is_role); - if (auth->validate_authentication_string( + Auth_plugin_operation_guard op_guard; + if (!op_guard || + auth->validate_authentication_string( const_cast(Str->first_factor_auth_info.auth.str), (unsigned)Str->first_factor_auth_info.auth.length)) { - my_error(ER_PASSWORD_FORMAT, MYF(0)); + if (!op_guard) + my_error(ER_SERVER_SHUTDOWN, MYF(0)); + else + my_error(ER_PASSWORD_FORMAT, MYF(0)); plugin_unlock(nullptr, plugin); what_to_set.m_what = NONE_ATTR; return (true); diff --git a/sql/sql_plugin.cc b/sql/sql_plugin.cc index 4abc4bc0704a..9bb8624f11a2 100644 --- a/sql/sql_plugin.cc +++ b/sql/sql_plugin.cc @@ -71,6 +71,7 @@ #include "prealloced_array.h" #include "sql/auth/auth_acls.h" #include "sql/auth/auth_common.h" // check_table_access +#include "sql/auth/auth_plugin_shutdown.h" #include "sql/auto_thd.h" // Auto_THD #include "sql/current_thd.h" #include "sql/dd/cache/dictionary_client.h" // dd::cache::Dictionary_client @@ -2062,6 +2063,10 @@ void plugin_shutdown() { if (initialized) { size_t count = plugin_array->size(); + + // Stop new auth plugin operations and drain in-flight callbacks first. + start_auth_plugin_shutdown_and_wait(); + mysql_mutex_lock(&LOCK_plugin); reap_needed = true; From 6ef7a8c983b89cd8a22694ec2bf9cdf8ca802611 Mon Sep 17 00:00:00 2001 From: Ando Nogueira Date: Mon, 27 Jul 2026 13:05:57 +0200 Subject: [PATCH 18/32] PS-11078 [9.7]: Skip orphan-sweep runs on forks (#6088) - Job-level repository guard on both jobs so forks with Actions enabled stop inheriting the hourly cron, which fails without the canonical repo's secrets and OIDC trust and emails the fork owner every hour. --- .github/workflows/orphan-sweep.yml | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/.github/workflows/orphan-sweep.yml b/.github/workflows/orphan-sweep.yml index 8d68a746304b..76a97e4d8731 100644 --- a/.github/workflows/orphan-sweep.yml +++ b/.github/workflows/orphan-sweep.yml @@ -2,6 +2,10 @@ # GHA runners. Companion to builds.yml; catches orphans from control-plane # crashes, cancelled workflows, and probe-step leaks. # +# Runs only in percona/percona-server: forks that enable Actions would +# otherwise inherit the hourly cron and fail every run (no secrets, and +# the AWS OIDC trust rejects fork subjects), spamming the fork owner. +# # PS-11179 (2026-05-28): EC2 sweep added for the AWS fallback path. name: orphan-sweep @@ -30,6 +34,7 @@ concurrency: jobs: sweep: + if: github.repository == 'percona/percona-server' runs-on: ubuntu-latest permissions: {} env: @@ -143,6 +148,7 @@ jobs: # OIDC -> STS via the same AWS_ROLE_ARN used by builds.yml. Region pinned # to eu-central-1 (matches create-runner-aws). sweep-ec2: + if: github.repository == 'percona/percona-server' runs-on: ubuntu-latest permissions: contents: read From b6d91cff4da21e1e4d9446ad24b1be1599821704 Mon Sep 17 00:00:00 2001 From: Ando Nogueira Date: Mon, 27 Jul 2026 13:06:05 +0200 Subject: [PATCH 19/32] PS-11078 [8.4]: Skip orphan-sweep runs on forks (#6087) - Job-level repository guard on both jobs so forks with Actions enabled stop inheriting the hourly cron, which fails without the canonical repo's secrets and OIDC trust and emails the fork owner every hour. --- .github/workflows/orphan-sweep.yml | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/.github/workflows/orphan-sweep.yml b/.github/workflows/orphan-sweep.yml index 8d68a746304b..76a97e4d8731 100644 --- a/.github/workflows/orphan-sweep.yml +++ b/.github/workflows/orphan-sweep.yml @@ -2,6 +2,10 @@ # GHA runners. Companion to builds.yml; catches orphans from control-plane # crashes, cancelled workflows, and probe-step leaks. # +# Runs only in percona/percona-server: forks that enable Actions would +# otherwise inherit the hourly cron and fail every run (no secrets, and +# the AWS OIDC trust rejects fork subjects), spamming the fork owner. +# # PS-11179 (2026-05-28): EC2 sweep added for the AWS fallback path. name: orphan-sweep @@ -30,6 +34,7 @@ concurrency: jobs: sweep: + if: github.repository == 'percona/percona-server' runs-on: ubuntu-latest permissions: {} env: @@ -143,6 +148,7 @@ jobs: # OIDC -> STS via the same AWS_ROLE_ARN used by builds.yml. Region pinned # to eu-central-1 (matches create-runner-aws). sweep-ec2: + if: github.repository == 'percona/percona-server' runs-on: ubuntu-latest permissions: contents: read From 04bac30af399212c810f106656739a9f0b7340ad Mon Sep 17 00:00:00 2001 From: Oleksiy Lukin Date: Tue, 28 Jul 2026 10:34:17 +0300 Subject: [PATCH 20/32] PS-11143 [10.x] Fix compilation errors in acl_table_user.cc - proper port from 8.4 Remove native password handling code that references PLUGIN_MYSQL_NATIVE_PASSWORD and PLUGIN_SHA256_PASSWORD constants which are not available in 10.x+. This code block was ported from 8.4 branch during cherrypick but the referenced constants and native password plugin were removed in 10.x architecture. The code was handling legacy 5.6 database layout upgrades, which is no longer needed or supported in 10.x+. Fixes compilation errors: - 'is_old_db_layout' was not declared in this scope - 'PLUGIN_MYSQL_NATIVE_PASSWORD' was not declared in this scope; did you mean 'PLUGIN_SHA256_PASSWORD'? The cherrypick properly removed the parameter from function signatures in 10.x, but this code block that used it was incorrectly left behind. Please squash it with the prev. PS-11143 commit --- sql/auth/acl_table_user.cc | 50 -------------------------------------- 1 file changed, 50 deletions(-) diff --git a/sql/auth/acl_table_user.cc b/sql/auth/acl_table_user.cc index 3fee641916c8..325285af0b3f 100644 --- a/sql/auth/acl_table_user.cc +++ b/sql/auth/acl_table_user.cc @@ -1629,56 +1629,6 @@ bool Acl_table_user_reader::read_plugin_info( user.plugin.str = tmpstr ? tmpstr : ""; user.plugin.length = strlen(user.plugin.str); - /* - In case we are working with 5.6 db layout we need to make server - aware of Password field and that the plugin column can be null. - In case when plugin column is null we use native password plugin - if we can. - */ - if (is_old_db_layout && (user.plugin.length == 0 || - Cached_authentication_plugins::compare_plugin( - PLUGIN_MYSQL_NATIVE_PASSWORD, user.plugin))) { - char *password = get_field( - &m_mem_root, m_table->field[m_table_schema->password_idx()]); - - // We do not support pre 4.1 hashes - plugin_ref native_plugin = - g_cached_authentication_plugins->get_cached_plugin_ref( - PLUGIN_MYSQL_NATIVE_PASSWORD); - if (native_plugin) { - const uint password_len = password ? strlen(password) : 0; - st_mysql_auth *auth = (st_mysql_auth *)plugin_decl(native_plugin)->info; - Auth_plugin_operation_guard op_guard; - if (!op_guard) { - LogErr(WARNING_LEVEL, ER_AUTHCACHE_USER_IGNORED_INVALID_PASSWORD, - user.user ? user.user : "", - user.host.get_host() ? user.host.get_host() : ""); - return true; - } - if (auth->validate_authentication_string(password, password_len) == 0) { - // auth_string takes precedence over password - if (user.credentials[PRIMARY_CRED].m_auth_string.length == 0) { - user.credentials[PRIMARY_CRED].m_auth_string.str = password; - user.credentials[PRIMARY_CRED].m_auth_string.length = password_len; - } - if (user.plugin.length == 0) { - user.plugin.str = Cached_authentication_plugins::get_plugin_name( - PLUGIN_MYSQL_NATIVE_PASSWORD); - user.plugin.length = strlen(user.plugin.str); - } - } else { - if ((user.access & SUPER_ACL) && !super_users_with_empty_plugin && - (user.plugin.length == 0)) - super_users_with_empty_plugin = true; - - LogErr(WARNING_LEVEL, ER_AUTHCACHE_USER_IGNORED_DEPRECATED_PASSWORD, - user.user ? user.user : "", - user.host.get_host() ? user.host.get_host() : ""); - return true; - } - } - } - /* Check if the plugin string is blank or null. If it is, the user will be skipped. From 0236284344dc2f273b71aff0367aed7c2368cc67 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Pawe=C5=82=20Olchawa?= Date: Tue, 21 Jul 2026 14:56:19 +0200 Subject: [PATCH 21/32] PS-11446 [trunk]: Fix inverted old/young check in LRUItr::start() that always restarted the LRU scan https://perconadev.atlassian.net/browse/PS-11446 LRUItr::start() is meant to leave the scan hand pointer (m_hp) in place while it is still within the "old" (cold) sublist of the LRU list, and only rewind it to the tail of the LRU list once the pointer has advanced past the old/young boundary into the young (hot) sublist. The boundary check was inverted: `m_hp->old` is true while the pointer is still in the old region, so the scan was rewound on every call instead of only at the old -> young transition. This defeated the intended optimization by repeatedly restarting the scan from the tail rather than letting it continue. Fix the condition to `!m_hp->old`, so the pointer is only reset once it has crossed into the young sublist. --- storage/innobase/buf/buf0buf.cc | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/storage/innobase/buf/buf0buf.cc b/storage/innobase/buf/buf0buf.cc index 278ea4a1a505..5299e3b2569b 100644 --- a/storage/innobase/buf/buf0buf.cc +++ b/storage/innobase/buf/buf0buf.cc @@ -3106,7 +3106,7 @@ the LRU list it resets the value to the tail of the LRU list. buf_page_t *LRUItr::start() { ut_ad(mutex_own(m_mutex)); - if (!m_hp || m_hp->old) { + if (!m_hp || !m_hp->old) { m_hp = UT_LIST_GET_LAST(m_buf_pool->LRU); } From a95842c128139231b3310207424a8dfd6c18fe30 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Pawe=C5=82=20Olchawa?= Date: Wed, 29 Jul 2026 21:35:00 +0200 Subject: [PATCH 22/32] PS-10595 [trunk]: Change the default of innodb_buffer_pool_populate to ON Post-push fix for PS-10595. innodb_buffer_pool_populate defaulted to OFF, deferring buffer pool page faults to first access at runtime. Testing showed this default causes a measurable performance regression as pages get faulted in on demand during normal operation instead of being pre-populated at startup. Flip the default to ON so pre-population happens automatically on startup, avoiding the regression without requiring users to set the variable explicitly. https://perconadev.atlassian.net/browse/PS-10595 --- .../r/innodb_buffer_pool_populate_basic.result | 14 +++++++------- storage/innobase/handler/ha_innodb.cc | 7 ++++--- storage/innobase/srv/srv0srv.cc | 2 +- 3 files changed, 12 insertions(+), 11 deletions(-) diff --git a/mysql-test/suite/sys_vars/r/innodb_buffer_pool_populate_basic.result b/mysql-test/suite/sys_vars/r/innodb_buffer_pool_populate_basic.result index 35dc99586024..e24616fd701f 100644 --- a/mysql-test/suite/sys_vars/r/innodb_buffer_pool_populate_basic.result +++ b/mysql-test/suite/sys_vars/r/innodb_buffer_pool_populate_basic.result @@ -1,28 +1,28 @@ SET @start_global_value = @@global.innodb_buffer_pool_populate; SELECT @start_global_value; @start_global_value -0 +1 Valid values are 'ON' and 'OFF' select @@global.innodb_buffer_pool_populate in (0, 1); @@global.innodb_buffer_pool_populate in (0, 1) 1 select @@global.innodb_buffer_pool_populate; @@global.innodb_buffer_pool_populate -0 +1 select @@session.innodb_buffer_pool_populate; ERROR HY000: Variable 'innodb_buffer_pool_populate' is a GLOBAL variable show global variables like 'innodb_buffer_pool_populate'; Variable_name Value -innodb_buffer_pool_populate OFF +innodb_buffer_pool_populate ON show session variables like 'innodb_buffer_pool_populate'; Variable_name Value -innodb_buffer_pool_populate OFF +innodb_buffer_pool_populate ON select * from performance_schema.global_variables where variable_name='innodb_buffer_pool_populate'; VARIABLE_NAME VARIABLE_VALUE -innodb_buffer_pool_populate OFF +innodb_buffer_pool_populate ON select * from performance_schema.session_variables where variable_name='innodb_buffer_pool_populate'; VARIABLE_NAME VARIABLE_VALUE -innodb_buffer_pool_populate OFF +innodb_buffer_pool_populate ON set global innodb_buffer_pool_populate='ON'; select @@global.innodb_buffer_pool_populate; @@global.innodb_buffer_pool_populate @@ -59,4 +59,4 @@ ERROR 42000: Variable 'innodb_buffer_pool_populate' can't be set to the value of SET @@global.innodb_buffer_pool_populate = @start_global_value; SELECT @@global.innodb_buffer_pool_populate; @@global.innodb_buffer_pool_populate -0 +1 diff --git a/storage/innobase/handler/ha_innodb.cc b/storage/innobase/handler/ha_innodb.cc index e84e74c2bc48..3af8d0995eff 100644 --- a/storage/innobase/handler/ha_innodb.cc +++ b/storage/innobase/handler/ha_innodb.cc @@ -24184,11 +24184,12 @@ static MYSQL_SYSVAR_BOOL( static MYSQL_SYSVAR_BOOL( buffer_pool_populate, srv_buf_pool_populate, PLUGIN_VAR_NOCMDARG, "Enforce page faults for InnoDB buffer pool allocations at allocation time" - " (pre-populate pages). When OFF (default), the pre-population is skipped" - " to reduce startup time and page-faults happen on first page accesses." + " (pre-populate pages). When ON (default), pre-population happens at" + " allocation time. When OFF, the pre-population is skipped to reduce" + " startup time and page-faults happen on first page accesses." " Note: it is recommended to turn on this variable when using large pages" " on systems with multiple NUMA nodes.", - nullptr, nullptr, false); + nullptr, nullptr, true); static MYSQL_SYSVAR_BOOL( api_enable_binlog, ib_binlog_enabled, diff --git a/storage/innobase/srv/srv0srv.cc b/storage/innobase/srv/srv0srv.cc index 1b0c0d997a0f..a8f35425d35d 100644 --- a/storage/innobase/srv/srv0srv.cc +++ b/storage/innobase/srv/srv0srv.cc @@ -231,7 +231,7 @@ bool srv_buf_pool_lazy_latch_init = false; /** Whether buffer pool allocations (huge and regular pages) are pre-populated (MAP_POPULATE and the explicit prefault step) at allocation time. Exposed as the innodb_buffer_pool_populate system variable. */ -bool srv_buf_pool_populate = false; +bool srv_buf_pool_populate = true; #ifdef UNIV_DEBUG /** Force all user tables to use page compression. */ From b995fe90b29e44449ea310c0c1305551a86613c0 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Pawe=C5=82=20Olchawa?= Date: Tue, 21 Jul 2026 09:34:23 +0200 Subject: [PATCH 23/32] PS-11444 [trunk]: Narrow the LRU list mutex scope in buf_page_init_for_read() https://perconadev.atlassian.net/browse/PS-11444 On IO-bound workloads every physical read serialized on the pool-wide LRU list mutex, because buf_page_init_for_read() held it across the page-hash latch acquisition, the descriptor initialization and the 16 KB frame memset. This narrows that scope and adjusts the surrounding paths accordingly. 1. Narrow the LRU list mutex in buf_page_init_for_read() to cover only the LRU list insert. The page-hash insert, the read io-fix, the frame X-lock and the memset now run under the page-hash cell X-latch alone. Page-hash membership is from now on serialized by the hash cell latch, not by the LRU list mutex. 2. This exposes a short window in which a page is reachable through the page hash but not yet linked into the LRU list. It is made safe by the read io-fix, which is published before the page becomes hash-reachable and cleared only after the LRU-add: threads that find the page through the hash back off on the io-fix, readers wait on the frame X-lock, and eviction cannot see it. In particular buf_page_make_young_if_needed() skips such a page, so promotion never manipulates a not-yet-linked node. 3. Introduce a latching rule: no thread may wait for a block frame rw-lock while holding a buffer pool LRU list mutex. The paths that latch a frame under the LRU list mutex use the rw_lock_*_nowait() variants; the rule is enforced in debug builds by rw_lock_assert_wait_allowed(). 4. Keep the page-hash cell X-latch across the delete + re-insert of the keep-zip path of buf_LRU_free_page() (new keep_hash_lock parameter), so a page id is never observably absent from the page hash while it is still logically in the buffer pool. Without this a concurrent read could insert a duplicate descriptor for the same page id. This also closes the PS-9837 window at the root. 5. Drop the LRU list mutex from buf_pool_watch_set(). It was taken only to avoid the keep-zip gap of point 4; with that gap now closed by the continuous hash cell latch, the hash cell latches purge already holds are sufficient, and purge no longer contends on the LRU list mutex. 6. Add debug assertions guarding the new invariants and a stress test (innodb_zip.lru_mutex_narrow_stress_debug) that widens the windows via debug sync points and drives the attacking compressed-table workload. --- .../r/lru_mutex_narrow_stress_debug.result | 16 ++ .../lru_mutex_narrow_stress_debug-master.opt | 3 + .../t/lru_mutex_narrow_stress_debug.test | 109 +++++++ storage/innobase/btr/btr0sea.cc | 13 +- storage/innobase/buf/buf0buddy.cc | 25 ++ storage/innobase/buf/buf0buf.cc | 265 ++++++++++++++---- storage/innobase/buf/buf0lru.cc | 116 ++++++-- storage/innobase/buf/buf0rea.cc | 15 +- storage/innobase/include/buf0buf.h | 31 +- storage/innobase/include/buf0buf.ic | 7 +- storage/innobase/sync/sync0debug.cc | 12 +- storage/innobase/sync/sync0rw.cc | 44 +++ 12 files changed, 568 insertions(+), 88 deletions(-) create mode 100644 mysql-test/suite/innodb_zip/r/lru_mutex_narrow_stress_debug.result create mode 100644 mysql-test/suite/innodb_zip/t/lru_mutex_narrow_stress_debug-master.opt create mode 100644 mysql-test/suite/innodb_zip/t/lru_mutex_narrow_stress_debug.test diff --git a/mysql-test/suite/innodb_zip/r/lru_mutex_narrow_stress_debug.result b/mysql-test/suite/innodb_zip/r/lru_mutex_narrow_stress_debug.result new file mode 100644 index 000000000000..81f2172016b8 --- /dev/null +++ b/mysql-test/suite/innodb_zip/r/lru_mutex_narrow_stress_debug.result @@ -0,0 +1,16 @@ +SET GLOBAL DEBUG="+d,buf_lru_free_page_delay_zip_reinsert,buf_page_init_for_read_delay_lru_add"; +CREATE TABLE t1 ( +id INT PRIMARY KEY, +c LONGBLOB +) ENGINE=InnoDB ROW_FORMAT=COMPRESSED KEY_BLOCK_SIZE=8; +CALL scan_t1(6); +CALL scan_t1(6); +CALL update_t1(12); +SELECT COUNT(*), MIN(LENGTH(c)) > 0 FROM t1; +COUNT(*) MIN(LENGTH(c)) > 0 +2000 1 +SET GLOBAL DEBUG="-d,buf_lru_free_page_delay_zip_reinsert,buf_page_init_for_read_delay_lru_add"; +DROP PROCEDURE populate_t1; +DROP PROCEDURE scan_t1; +DROP PROCEDURE update_t1; +DROP TABLE t1; diff --git a/mysql-test/suite/innodb_zip/t/lru_mutex_narrow_stress_debug-master.opt b/mysql-test/suite/innodb_zip/t/lru_mutex_narrow_stress_debug-master.opt new file mode 100644 index 000000000000..451844199700 --- /dev/null +++ b/mysql-test/suite/innodb_zip/t/lru_mutex_narrow_stress_debug-master.opt @@ -0,0 +1,3 @@ +--innodb-buffer-pool-size=24M +--innodb-buffer-pool-chunk-size=2M +--innodb-buffer-pool-instances=1 diff --git a/mysql-test/suite/innodb_zip/t/lru_mutex_narrow_stress_debug.test b/mysql-test/suite/innodb_zip/t/lru_mutex_narrow_stress_debug.test new file mode 100644 index 000000000000..8e120feb2013 --- /dev/null +++ b/mysql-test/suite/innodb_zip/t/lru_mutex_narrow_stress_debug.test @@ -0,0 +1,109 @@ +# PS-11141: buf_page_init_for_read() inserts into the page hash without +# holding the LRU list mutex. This test stresses the two windows that the +# narrowed latching opens, with debug sync points widening them: +# +# 1) buf_lru_free_page_delay_zip_reinsert widens the window between the +# page hash delete and the re-insert of the compressed-only descriptor +# in the keep-zip path of buf_LRU_free_page(). The hash cell X-latch is +# held across that window, so a concurrent read of the page being +# unzip-evicted must block on the cell latch instead of inserting a +# second descriptor for the same page id (which would fail +# ut_a(!buf_page_hash_get_low(buf_pool, b->id)) at the re-insert). +# +# 2) buf_page_init_for_read_delay_lru_add widens the window in which a +# page is hash-visible but not yet linked into the LRU list. Threads +# finding such a page via the page hash (buf_page_get_zip() -> +# buf_block_try_discard_uncompressed() -> buf_LRU_free_page(), +# buf_buddy_relocate(), buf_page_make_young_if_needed()) must back off +# from it because it is io-fixed for read; buf_page_can_relocate() +# accepts it (io-fixed, not in the LRU list) without asserting. +# +# The workload keeps a compressed table with externally stored BLOBs +# (read through buf_page_get_zip()) larger than the buffer pool, so that +# unzip-LRU eviction, keep-zip frees, physical re-reads and buddy +# alloc/free churn all run concurrently. + +--source include/have_debug.inc +--source include/have_innodb_16k.inc +--source include/big_test.inc + +SET GLOBAL DEBUG="+d,buf_lru_free_page_delay_zip_reinsert,buf_page_init_for_read_delay_lru_add"; + +CREATE TABLE t1 ( + id INT PRIMARY KEY, + c LONGBLOB +) ENGINE=InnoDB ROW_FORMAT=COMPRESSED KEY_BLOCK_SIZE=8; + +--disable_query_log +DELIMITER |; +CREATE PROCEDURE populate_t1() +BEGIN + DECLARE i INT DEFAULT 1; + WHILE i <= 2000 DO + INSERT INTO t1 VALUES (i, REPEAT(CONCAT('row', i, '-'), 2048)); + IF i % 200 = 0 THEN + COMMIT; + END IF; + SET i = i + 1; + END WHILE; + COMMIT; +END| +CREATE PROCEDURE scan_t1(IN iterations INT) +BEGIN + DECLARE i INT DEFAULT 1; + DECLARE dummy BIGINT; + WHILE i <= iterations DO + SELECT SUM(LENGTH(c)) INTO dummy FROM t1; + SET i = i + 1; + END WHILE; +END| +CREATE PROCEDURE update_t1(IN iterations INT) +BEGIN + DECLARE i INT DEFAULT 1; + WHILE i <= iterations DO + UPDATE t1 SET c = REPEAT(CONCAT('mod', i, '-'), 2048) + WHERE id % 500 = i % 500; + COMMIT; + SET i = i + 1; + END WHILE; +END| +DELIMITER ;| +SET autocommit = 0; +CALL populate_t1(); +SET autocommit = 1; +--enable_query_log + +# Two connections scanning the BLOB column (physical re-reads + +# buf_page_get_zip() -> buf_block_try_discard_uncompressed()), one +# connection updating (creates dirty compressed pages, so the keep-zip +# eviction path sees both ZIP_PAGE and ZIP_DIRTY re-inserts). +--connect(con_scan1, localhost, root,,) +--send CALL scan_t1(6) + +--connect(con_scan2, localhost, root,,) +--send CALL scan_t1(6) + +--connect(con_upd, localhost, root,,) +--send CALL update_t1(12) + +--connection con_scan1 +--reap +--disconnect con_scan1 + +--connection con_scan2 +--reap +--disconnect con_scan2 + +--connection con_upd +--reap +--disconnect con_upd + +--connection default +SELECT COUNT(*), MIN(LENGTH(c)) > 0 FROM t1; + +SET GLOBAL DEBUG="-d,buf_lru_free_page_delay_zip_reinsert,buf_page_init_for_read_delay_lru_add"; + +DROP PROCEDURE populate_t1; +DROP PROCEDURE scan_t1; +DROP PROCEDURE update_t1; +DROP TABLE t1; diff --git a/storage/innobase/btr/btr0sea.cc b/storage/innobase/btr/btr0sea.cc index e9ea3deac4b9..6c9422100bda 100644 --- a/storage/innobase/btr/btr0sea.cc +++ b/storage/innobase/btr/btr0sea.cc @@ -1961,12 +1961,13 @@ static bool btr_search_hash_table_validate(ulint part_id) { /* When a block is being freed, buf_LRU_free_page() first removes the block from - buf_pool->page_hash by calling - buf_LRU_block_remove_hashed_page(). - After that, it invokes - buf_LRU_block_remove_hashed() to - remove the block from - btr_search_sys->hash_tables[i]. */ + buf_pool->page_hash and sets it to + BUF_BLOCK_REMOVE_HASH by calling + buf_LRU_block_remove_hashed(). + After that, it removes the block + from btr_search_sys->parts[i] (the + AHI) by calling + btr_search_drop_page_hash_index(). */ ut_a(buf_block_get_state(block) == BUF_BLOCK_REMOVE_HASH); } diff --git a/storage/innobase/buf/buf0buddy.cc b/storage/innobase/buf/buf0buddy.cc index 5d52916f1800..70da21ecfa1e 100644 --- a/storage/innobase/buf/buf0buddy.cc +++ b/storage/innobase/buf/buf0buddy.cc @@ -40,6 +40,9 @@ this program; if not, write to the Free Software Foundation, Inc., #include "page0zip.h" +#include "sync0debug.h" +#include "sync0types.h" + /** When freeing a buf we attempt to coalesce by looking at its buddy and deciding whether it is free or not. To ascertain if the buddy is free we look for BUF_BUDDY_STAMP_FREE at BUF_BUDDY_STAMP_OFFSET @@ -416,6 +419,25 @@ static void *buf_buddy_alloc_from(buf_pool_t *buf_pool, void *buf, ulint i, return (buf); } +#ifdef UNIV_DEBUG +/** Asserts that the calling thread holds no buffer pool page hash cell +latch (S or X). The buddy allocator must never be entered while one is +held: buf_buddy_free() may recombine and call buf_buddy_relocate(), +which acquires the cell X-latch of whatever page id is stamped in the +buddy frame - possibly the very cell the caller holds, and in any case +an unordered same-level acquisition (page-hash cell -> zip_free_mutex -> +page-hash cell) which can form a deadlock cycle with other threads. +This rule is what makes it safe for the keep-zip path of +buf_LRU_free_page() to keep a cell X-latch across +buf_LRU_block_remove_hashed() (keep_hash_lock): that path never reaches +the buddy allocator. */ +namespace { +void buf_buddy_no_page_hash_latch_validate() { + ut_ad(sync_check_find(SYNC_BUF_PAGE_HASH) == nullptr); +} +} // namespace +#endif /* UNIV_DEBUG */ + /** Allocate a block. @param[in,out] buf_pool buffer pool instance @param[in] i index of buf_pool->zip_free[] @@ -425,6 +447,7 @@ void *buf_buddy_alloc_low(buf_pool_t *buf_pool, ulint i) { buf_block_t *block; ut_ad(!mutex_own(&buf_pool->zip_mutex)); + ut_d(buf_buddy_no_page_hash_latch_validate()); ut_ad(i >= buf_buddy_get_slot(UNIV_ZIP_SIZE_MIN)); if (i < BUF_BUDDY_SIZES) { @@ -476,6 +499,7 @@ static bool buf_buddy_relocate(buf_pool_t *buf_pool, void *src, void *dst, ut_ad(mutex_own(&buf_pool->zip_free_mutex)); ut_ad(!mutex_own(&buf_pool->zip_mutex)); + ut_d(buf_buddy_no_page_hash_latch_validate()); ut_ad(!ut_align_offset(src, size)); ut_ad(!ut_align_offset(dst, size)); ut_ad(i >= buf_buddy_get_slot(UNIV_ZIP_SIZE_MIN)); @@ -611,6 +635,7 @@ void buf_buddy_free_low(buf_pool_t *buf_pool, void *buf, ulint i, buf_buddy_free_t *buddy; ut_ad(!mutex_own(&buf_pool->zip_mutex)); + ut_d(buf_buddy_no_page_hash_latch_validate()); ut_ad(i <= BUF_BUDDY_SIZES); ut_ad(i >= buf_buddy_get_slot(UNIV_ZIP_SIZE_MIN)); diff --git a/storage/innobase/buf/buf0buf.cc b/storage/innobase/buf/buf0buf.cc index 5299e3b2569b..12262e42bd2c 100644 --- a/storage/innobase/buf/buf0buf.cc +++ b/storage/innobase/buf/buf0buf.cc @@ -3135,9 +3135,8 @@ bool buf_pool_watch_is_sentinel(const buf_pool_t *buf_pool, } /** Add watch for the given page to be read in. Caller must have -appropriate hash_lock for the bpage and hold the LRU list mutex to avoid a race -condition with buf_LRU_free_page inserting the same page into the page hash. -This function may release the hash_lock and reacquire it. +appropriate hash_lock for the bpage. This function may release the +hash_lock and reacquire it. @param[in] page_id page id @param[in,out] hash_lock hash_lock currently latched @return NULL if watch set, block if the page is in the buffer pool */ @@ -3169,15 +3168,25 @@ static buf_page_t *buf_pool_watch_set(const page_id_t &page_id, of latching. We acquire all the hash_locks. They are needed because we don't want to read any stale information in buf_pool->watch[]. However, it is not in the critical code path - as this function will be called only by the purge thread. */ + as this function will be called only by the purge thread. + + Holding all the hash cell X-latches also stabilizes page hash + membership: every insert into and delete from the page hash happens + under the affected cell's X-latch, including the delete + re-insert + transition of the keep-zip path of buf_LRU_free_page(), which keeps + the cell's X-latch continuously so that the page id is never + observably absent from the hash (see the page_hash membership-change + protocol at its declaration in buf0buf.h). Therefore the LRU list + mutex does not need to be (and is not) taken here: the recheck below + cannot miss a page which is logically in the buffer pool. */ /* To obey latching order first release the hash_lock. */ rw_lock_x_unlock(*hash_lock); - mutex_enter(&buf_pool->LRU_list_mutex); hash_lock_x_all(buf_pool->page_hash); - /* If not own LRU_list_mutex, page_hash can be changed. */ + /* page_hash could have been resized while we did not hold any + hash cell latch. */ *hash_lock = buf_page_hash_lock_get(buf_pool, page_id); /* We have to recheck that the page @@ -3188,7 +3197,6 @@ static buf_page_t *buf_pool_watch_set(const page_id_t &page_id, bpage = buf_page_hash_get_low(buf_pool, page_id); if (bpage) { - mutex_exit(&buf_pool->LRU_list_mutex); hash_unlock_x_all_but(buf_pool->page_hash, *hash_lock); goto page_found; } @@ -3219,8 +3227,6 @@ static buf_page_t *buf_pool_watch_set(const page_id_t &page_id, HASH_INSERT(buf_page_t, hash, buf_pool->page_hash, page_id.hash(), bpage); - mutex_exit(&buf_pool->LRU_list_mutex); - /* Once the sentinel is in the page_hash we can safely release all locks except just the relevant hash_lock */ @@ -3275,7 +3281,9 @@ void buf_pool_watch_unset(const page_id_t &page_id) { rw_lock_t *hash_lock = buf_page_hash_lock_get(buf_pool, page_id); rw_lock_x_lock(hash_lock, UT_LOCATION_HERE); - /* page_hash can be changed. */ + /* A concurrent buffer pool resize can rehash page_hash and remap this + page id to a different shard latch between computing hash_lock and + latching it; re-confirm and re-latch the correct shard. */ hash_lock = buf_page_hash_lock_x_confirm(hash_lock, buf_pool, page_id); /* The page must exist because buf_pool_watch_set() @@ -3302,7 +3310,9 @@ bool buf_pool_watch_occurred(const page_id_t &page_id) { rw_lock_s_lock(hash_lock, UT_LOCATION_HERE); - /* If not own buf_pool_mutex, page_hash can be changed. */ + /* A concurrent buffer pool resize can rehash page_hash and remap this + page id to a different shard latch between computing hash_lock and + latching it; re-confirm and re-latch the correct shard. */ hash_lock = buf_page_hash_lock_s_confirm(hash_lock, buf_pool, page_id); /* The page must exist because buf_pool_watch_set() @@ -3352,6 +3362,26 @@ static void buf_page_make_young_if_needed(buf_page_t *bpage) { ut_ad(bpage->buf_fix_count > 0); ut_a(buf_page_in_file(bpage)); + /* A page whose read IO is still in progress may not yet be linked into + the LRU list: buf_page_init_for_read() makes the page hash-visible before + it links it into the LRU list. Such a page must not be promoted - + buf_LRU_make_block_young() would unlink a node which is not linked, + corrupting the LRU list. + Reading the io-fix snapshot without the block mutex is correct here: the + LRU-add happens-before the read IO is dispatched, which happens-before + io_fix is reset to BUF_IO_NONE at IO completion. Thus observing + !was_io_fix_read() implies the LRU-add has already happened, and the + buf-fix held by our caller keeps the page in the LRU. Skipping the + promotion on a stale BUF_IO_READ snapshot is benign: the page was just + added at the head of the old sublist and a subsequent access will promote + it. + Both orderings this argument depends on (BUF_IO_READ is published before + the page is hash-reachable; the read io-fix is cleared only after the + LRU-add) are asserted at the transitions in buf_page_t::set_io_fix(). */ + if (bpage->was_io_fix_read()) { + return; + } + if (buf_page_peek_if_too_old(bpage)) { buf_page_make_young(bpage); } @@ -3986,7 +4016,9 @@ buf_block_t *Buf_fetch::lookup() { rw_lock_s_lock(m_hash_lock, UT_LOCATION_HERE); - /* If not own LRU_list_mutex, page_hash can be changed. */ + /* A concurrent buffer pool resize can rehash page_hash and remap this + page id to a different shard latch between computing m_hash_lock and + latching it; re-confirm and re-latch the correct shard. */ m_hash_lock = buf_page_hash_lock_s_confirm(m_hash_lock, m_buf_pool, m_page_id); @@ -4036,7 +4068,10 @@ buf_block_t *Buf_fetch::is_on_watch() { rw_lock_x_lock(m_hash_lock, UT_LOCATION_HERE); - /* If not own LRU_list_mutex, page_hash can be changed. */ + /* A concurrent buffer pool resize can rehash page_hash (buf_pool_resize() + changes the number of cells), remapping this page id to a different shard + latch after we computed m_hash_lock but before we latched it. Re-confirm + the shard the page id currently maps to and re-latch it if it moved. */ m_hash_lock = buf_page_hash_lock_x_confirm(m_hash_lock, m_buf_pool, m_page_id); @@ -4101,7 +4136,10 @@ dberr_t Buf_fetch::zip_page_handler(buf_block_t *&fix_block) { mutex_enter(&m_buf_pool->LRU_list_mutex); - /* If not own LRU_list_mutex, page_hash can be changed. */ + /* We hold the LRU list mutex, which blocks a concurrent buffer pool + resize (buf_pool_resize() takes it), so page_hash cannot be rehashed + here: the shard latch for this page id is stable and no _confirm is + needed. */ m_hash_lock = buf_page_hash_lock_get(m_buf_pool, m_page_id); rw_lock_x_lock(m_hash_lock, UT_LOCATION_HERE); @@ -4397,12 +4435,15 @@ dberr_t Buf_fetch::debug_check(buf_block_t *fix_block) { mutex_enter(fix_mutex); if (buf_LRU_free_page(&fix_block->page, true)) { - /* If not own LRU_list_mutex, page_hash can be changed. */ + /* buf_LRU_free_page() released the page hash latch; re-acquire the + shard latch for this page id before re-checking / re-watching it. */ m_hash_lock = buf_page_hash_lock_get(m_buf_pool, m_page_id); rw_lock_x_lock(m_hash_lock, UT_LOCATION_HERE); - /* If not own LRU_list_mutex, page_hash can be changed. */ + /* We held no shard latch across the lines above, so a concurrent + buffer pool resize may have rehashed page_hash and remapped this page + id to a different shard; re-confirm and re-latch the correct shard. */ m_hash_lock = buf_page_hash_lock_x_confirm(m_hash_lock, m_buf_pool, m_page_id); @@ -5135,8 +5176,6 @@ buf_page_t *buf_page_init_for_read(ulint mode, const page_id_t &page_id, data = buf_buddy_alloc(buf_pool, page_size.physical()); } - mutex_enter(&buf_pool->LRU_list_mutex); - hash_lock = buf_page_hash_lock_get(buf_pool, page_id); rw_lock_x_lock(hash_lock, UT_LOCATION_HERE); @@ -5150,8 +5189,6 @@ buf_page_t *buf_page_init_for_read(ulint mode, const page_id_t &page_id, /* The page is already in the buffer pool. */ watch_page = nullptr; - mutex_exit(&buf_pool->LRU_list_mutex); - rw_lock_x_unlock(hash_lock); if (bpage != nullptr) { @@ -5188,39 +5225,81 @@ buf_page_t *buf_page_init_for_read(ulint mode, const page_id_t &page_id, block->mark_for_read_io(); buf_page_set_io_fix(bpage, BUF_IO_READ); - /* The block must be put to the LRU list, to the old blocks */ - buf_LRU_add_block(bpage, true /* to old blocks */); - if (page_size.is_compressed()) { + /* Setting zip.data is still protected by the hash X-latch here: + the page is already in the page hash, but no other thread can look + it up until the latch is released below. */ block->page.zip.data = (page_zip_t *)data; - - /* To maintain the invariant - block->in_unzip_LRU_list - == buf_page_belongs_to_unzip_LRU(&block->page) - we have to add this block to unzip_LRU - after block->page.zip.data is set. */ - ut_ad(buf_page_belongs_to_unzip_LRU(&block->page)); - buf_unzip_LRU_add_block(block, true); } - mutex_exit(&buf_pool->LRU_list_mutex); - - /* We set a pass-type x-lock on the frame because then - the same thread which called for the read operation - (and is running now at this point of code) can wait - for the read to complete by waiting for the x-lock on - the frame; if the x-lock were recursive, the same - thread would illegally get the x-lock before the page - read is completed. The x-lock is cleared by the - io-handler thread. */ - ut_ad(block->latches_initialized); rw_lock_x_lock_gen(&block->lock, BUF_IO_READ, UT_LOCATION_HERE); rw_lock_x_unlock(hash_lock); buf_page_mutex_exit(block); + + /* The page is hash-visible already, but eviction cannot see it + (not on LRU yet) and readers are blocked on the frame X-lock, + so no other thread can race with the add. + + IMPORTANT: we strongly depend here on the fact that there is no + other thread that can try to acquire that frame's S-lock while + holding already the LRU list mutex (it would be deadlock cycle). + For existing use cases, for that thread to exist, the page would + need to be in the LRU list already. This latching rule is documented + at the LRU_list_mutex declaration in buf0buf.h and enforced in debug + builds by rw_lock_assert_wait_allowed() at the rw-lock wait entry + points in sync0rw.cc (it cannot be expressed via latch_level_t + ordering: block->lock is SYNC_LEVEL_VARYING, which LatchDebug + ignores). */ + + /* Widen the hash-visible-not-in-LRU window: the page is already + reachable through the page hash but not yet linked into the LRU list. + Threads finding it via the page hash (buf_page_get_zip() -> + buf_block_try_discard_uncompressed(), buf_buddy_relocate(), + buf_page_make_young_if_needed()) must back off from it, because it is + io-fixed for read. */ + DBUG_EXECUTE_IF( + "buf_page_init_for_read_delay_lru_add", + std::this_thread::sleep_for(std::chrono::microseconds(100));); + + mutex_enter(&buf_pool->LRU_list_mutex); + + /* For a compressed page zip.data was set above, before the page + became reachable through the page hash, so + buf_page_belongs_to_unzip_LRU() already holds and buf_LRU_add_block() + links the block into the unzip_LRU list as well, within this same + critical section: every observer of the LRU list sees the invariant + block->in_unzip_LRU_list == + buf_page_belongs_to_unzip_LRU(&block->page) hold. (This is unlike the + pre-narrowing code, which set zip.data only after buf_LRU_add_block() + and therefore had to add the block to the unzip_LRU list explicitly + afterwards; an explicit second add here would corrupt the list.) */ + buf_LRU_add_block(bpage, true /* to old blocks */); + + ut_ad(!page_size.is_compressed() || block->in_unzip_LRU_list); + + mutex_exit(&buf_pool->LRU_list_mutex); } else { + /* Compressed-only page: a bare BUF_BLOCK_ZIP_PAGE descriptor with no + uncompressed frame (and thus no frame rw-lock). It is initialized and + made hash-visible while the page hash X-latch and zip_mutex (this + descriptor's "block mutex") are held, and is linked into the LRU list + afterwards under a brief LRU_list_mutex hold - the same narrowed + latching order as for the block-backed pages above. + + Setting io_fix = BUF_IO_READ before the descriptor becomes reachable + through the page hash is what makes the hash-visible-but-not-in-LRU + window safe, exactly as for block-backed pages: every path which + could move or free the page based on finding it in the page hash + backs off from a read-io-fixed page (buf_page_make_young_if_needed() + skips it, buf_page_free_stale() bails out, buf_buddy relocation and + Buf_fetch::zip_page_handler() require io_fix == BUF_IO_NONE), and + readers (e.g. buf_page_get_zip()) wait for the read to complete, + which happens-after the LRU-add below, because the read IO is only + dispatched after this function returns. */ + /* Initialize the buf_pool pointer. */ bpage->buf_pool_index = buf_pool_index(buf_pool); @@ -5251,6 +5330,8 @@ buf_page_t *buf_page_init_for_read(ulint mode, const page_id_t &page_id, ut_d(bpage->in_free_list = false); ut_d(bpage->in_LRU_list = false); + buf_page_set_io_fix(bpage, BUF_IO_READ); + ut_d(bpage->in_page_hash = true); if (watch_page != nullptr) { @@ -5271,16 +5352,31 @@ buf_page_t *buf_page_init_for_read(ulint mode, const page_id_t &page_id, rw_lock_x_unlock(hash_lock); - /* The block must be put to the LRU list, to the old blocks. - The zip size is already set into the page zip */ + mutex_exit(&buf_pool->zip_mutex); + + /* Widen the hash-visible-not-in-LRU window, as in the block-backed + branch above. */ + DBUG_EXECUTE_IF( + "buf_page_init_for_read_delay_lru_add", + std::this_thread::sleep_for(std::chrono::microseconds(100));); + + /* The page is hash-visible already (io-fixed for read, see above), + but eviction cannot see it (not on the LRU list yet), so no other + thread can race with the add. The block must be put to the LRU list, + to the old blocks. The zip size is already set into the page zip. */ + mutex_enter(&buf_pool->LRU_list_mutex); +#if defined UNIV_DEBUG || defined UNIV_BUF_DEBUG + /* buf_LRU_insert_zip_clean() requires the zip_mutex; re-acquired + here under the LRU list mutex, which follows the registered + latch_level_t order (SYNC_BUF_LRU_LIST > SYNC_BUF_BLOCK). */ + mutex_enter(&buf_pool->zip_mutex); +#endif /* UNIV_DEBUG || UNIV_BUF_DEBUG */ buf_LRU_add_block(bpage, true /* to old blocks */); #if defined UNIV_DEBUG || defined UNIV_BUF_DEBUG buf_LRU_insert_zip_clean(bpage); + mutex_exit(&buf_pool->zip_mutex); #endif /* UNIV_DEBUG || UNIV_BUF_DEBUG */ mutex_exit(&buf_pool->LRU_list_mutex); - buf_page_set_io_fix(bpage, BUF_IO_READ); - - mutex_exit(&buf_pool->zip_mutex); } buf_pool->n_pend_reads.fetch_add(1); @@ -5378,18 +5474,27 @@ buf_block_t *buf_page_create(const page_id_t &page_id, /* Latch the page before releasing hash lock so that concurrent request for this page doesn't see half initialized page. ALTER tablespace for encryption and clone page copy can request page for any page id within tablespace - size limit. */ + size limit. + + The nowait variants must be used and cannot fail: the frame comes from + the free list, so its latch is unlocked, and the block is unreachable by + other threads until the page hash X-latch is released below. This keeps + the LRU_list_mutex latching rule (no waiting for a frame latch under the + LRU list mutex, see the LRU_list_mutex declaration) free of blocking + acquisitions - we hold the LRU list mutex here. */ mtr_memo_type_t mtr_latch_type; + bool latched [[maybe_unused]]; ut_ad(block->latches_initialized); if (rw_latch == RW_X_LATCH) { - rw_lock_x_lock(&block->lock, UT_LOCATION_HERE); + latched = rw_lock_x_lock_nowait(&block->lock, UT_LOCATION_HERE); mtr_latch_type = MTR_MEMO_PAGE_X_FIX; } else { - rw_lock_sx_lock(&block->lock, UT_LOCATION_HERE); + latched = rw_lock_sx_lock_nowait(&block->lock, 0, UT_LOCATION_HERE); mtr_latch_type = MTR_MEMO_PAGE_SX_FIX; } + ut_ad(latched); mtr_memo_push(mtr, block, mtr_latch_type); rw_lock_x_unlock(hash_lock); @@ -5934,6 +6039,30 @@ void buf_page_t::set_io_fix(buf_io_fix io_fix) { take_io_responsibility(); } Latching_rules_helpers::on_transition_to(*this, io_fix); + + if (io_fix == BUF_IO_READ) { + /* BUF_IO_READ may only be stored on a page that is not yet reachable + through the page hash: either it is not in the page hash at all, or + the storing thread still holds the hash cell's X-latch. Threads which + find a page through a page hash lookup therefore can never observe a + pre-read BUF_IO_NONE, which is what allows + buf_page_make_young_if_needed() to test was_io_fix_read() without the + block mutex to detect a page whose LRU-add is still pending. + (This cannot be expressed in buf_io_fix_latching_rules: its latch set + does not include the page hash latches.) */ + ut_ad(!in_page_hash || + buf_page_hash_lock_held_x(buf_pool_from_bpage(this), this)); + } + + if (old_io_fix == BUF_IO_READ && io_fix == BUF_IO_NONE) { + /* The read IO is dispatched only after buf_page_init_for_read() has + linked the page into the LRU list, so by the time the read io-fix is + cleared the page must be in the LRU list. + buf_page_make_young_if_needed() relies on this: observing + io_fix != BUF_IO_READ implies the LRU-add has completed and the page + may be promoted. */ + ut_ad(in_LRU_list); + } #endif this->io_fix.store(io_fix, std::memory_order_relaxed); #ifdef UNIV_DEBUG @@ -6409,6 +6538,7 @@ static void buf_pool_validate_instance(buf_pool_t *buf_pool) { ulint n_flush = 0; ulint n_free = 0; ulint n_zip = 0; + ulint n_lru_add_pending = 0; ut_ad(buf_pool); @@ -6457,7 +6587,26 @@ static void buf_pool_validate_instance(buf_pool_t *buf_pool) { } } +#ifdef UNIV_DEBUG + if (!block->page.in_LRU_list) { + /* buf_page_init_for_read() makes the page hash-visible before + linking it into the LRU list. Such a page is still io-fixed for + read. Reading in_LRU_list is stable here: it is only modified + under LRU_list_mutex, which we hold. */ + ut_a(block->page.was_io_fix_read()); + n_lru_add_pending++; + } else { + n_lru++; + } +#else /* UNIV_DEBUG */ + /* Without UNIV_DEBUG there is no in_LRU_list flag; count how many + FILE_PAGE blocks may legitimately be missing from the LRU list so + the length cross-check below can be relaxed by that amount. */ + if (block->page.was_io_fix_read()) { + n_lru_add_pending++; + } n_lru++; +#endif /* UNIV_DEBUG */ break; case BUF_BLOCK_NOT_USED: @@ -6483,10 +6632,10 @@ static void buf_pool_validate_instance(buf_pool_t *buf_pool) { /* All clean blocks should be I/O-unfixed. */ break; case BUF_IO_READ: - /* In buf_LRU_free_page(), we temporarily set - b->io_fix = BUF_IO_READ for a newly allocated - control block in order to prevent - buf_page_get_gen() from decompressing the block. */ + /* A clean compressed-only page can be io-fixed for read only + while its initial read from disk is pending. (buf_LRU_free_page() + pins the re-inserted compressed-only descriptor with + buf_page_set_sticky(), which is BUF_IO_PIN, not BUF_IO_READ.) */ break; default: ut_error; @@ -6558,7 +6707,15 @@ static void buf_pool_validate_instance(buf_pool_t *buf_pool) { << buf_pool->curr_size << " zip " << n_zip << ". Aborting..."; } +#ifdef UNIV_DEBUG + /* Pages whose read IO is in progress and which are not yet linked into + the LRU list were counted into n_lru_add_pending instead of n_lru. */ + (void)n_lru_add_pending; ut_a(UT_LIST_GET_LEN(buf_pool->LRU) == n_lru); +#else /* UNIV_DEBUG */ + ut_a(UT_LIST_GET_LEN(buf_pool->LRU) <= n_lru); + ut_a(n_lru <= UT_LIST_GET_LEN(buf_pool->LRU) + n_lru_add_pending); +#endif /* UNIV_DEBUG */ mutex_exit(&buf_pool->LRU_list_mutex); mutex_exit(&buf_pool->chunks_mutex); diff --git a/storage/innobase/buf/buf0lru.cc b/storage/innobase/buf/buf0lru.cc index 040d4f87804b..40ccd19a8fd6 100644 --- a/storage/innobase/buf/buf0lru.cc +++ b/storage/innobase/buf/buf0lru.cc @@ -154,13 +154,20 @@ If a compressed page is freed other compressed pages may be relocated. compressed page of an uncompressed page @param[in] ignore_content true if should ignore page content, since it could be not initialized +@param[in] keep_hash_lock true if the hash cell X-latch should be kept + by this function instead of being released; + only allowed when the caller will re-insert + a compressed-only descriptor for this page + id into the page hash (the keep-zip path of + buf_LRU_free_page()), so that the page id is + never observably absent from the page hash @retval true if BUF_BLOCK_FILE_PAGE was removed from page_hash. The caller needs to free the page to the free list @retval false if BUF_BLOCK_ZIP_PAGE was removed from page_hash. In this case the block is already returned to the buddy allocator. */ -[[nodiscard]] static bool buf_LRU_block_remove_hashed(buf_page_t *bpage, - bool zip, - bool ignore_content); +[[nodiscard]] static bool buf_LRU_block_remove_hashed( + buf_page_t *bpage, bool zip, bool ignore_content, + bool keep_hash_lock = false); /** Puts a file page whose has no hash index to the free list. @param[in,out] block Must contain a file page and be in a state @@ -1915,16 +1922,22 @@ bool buf_LRU_free_page(buf_page_t *bpage, bool zip) { auto block_mutex = buf_page_get_mutex(bpage); auto hash_lock = buf_page_hash_lock_get(buf_pool, bpage->id); - ut_ad(bpage->in_LRU_list); ut_ad(mutex_own(&buf_pool->LRU_list_mutex)); ut_ad(mutex_own(block_mutex)); - ut_ad(buf_page_in_file(bpage)); if (!buf_page_can_relocate(bpage)) { /* Do not free buffer fixed and I/O-fixed blocks. */ return (false); } + /* These assertions can only be checked after the io-fix check + above, because pages become visible in the page hash table before + they are linked into the LRU list (they are io-fixed for read + until after they are linked into the LRU list, so + buf_page_can_relocate() has returned false for them). */ + ut_ad(bpage->in_LRU_list); + ut_ad(buf_page_in_file(bpage)); + #ifdef UNIV_IBUF_COUNT_DEBUG ut_a(ibuf_count_get(bpage->id) == 0); #endif /* UNIV_IBUF_COUNT_DEBUG */ @@ -1945,10 +1958,17 @@ bool buf_LRU_free_page(buf_page_t *bpage, bool zip) { return (false); } else if (buf_page_get_state(bpage) == BUF_BLOCK_FILE_PAGE) { + /* This holds because of the "else" keyword above, + but we assert it for clarity. */ + ut_ad(!zip && bpage->zip.data != nullptr); b = buf_page_alloc_descriptor(); ut_a(b); } + /* Protect: buf_buddy_free must not be invoked in buf_LRU_block_remove_hashed, + when passing keep_hash_lock = true (b != nullptr). */ + ut_ad(b == nullptr || (!zip && bpage->zip.data != nullptr)); + ut_ad(buf_page_in_file(bpage)); ut_ad(bpage->in_LRU_list); ut_ad(bpage->in_flush_list == is_dirty); @@ -1994,7 +2014,20 @@ bool buf_LRU_free_page(buf_page_t *bpage, bool zip) { ut_ad(rw_lock_own(hash_lock, RW_LOCK_X)); ut_ad(buf_page_can_relocate(bpage)); - if (!buf_LRU_block_remove_hashed(bpage, zip, false)) { + /* When the compressed page is to be kept (b != nullptr), ask + buf_LRU_block_remove_hashed() to keep the hash cell X-latch, so that it + is held continuously from before the HASH_DELETE until after the + compressed-only descriptor is re-inserted below: the page id is never + observably absent from the page hash. Otherwise a concurrent + buf_page_init_for_read() (which inserts into the page hash without + holding the LRU list mutex) could insert a second descriptor for this + page id in the meantime. + + Protect: buf_buddy_free must not be invoked in buf_LRU_block_remove_hashed, + when passing keep_hash_lock = true (b != nullptr). */ + ut_ad(b == nullptr || (!zip && bpage->zip.data != nullptr)); + + if (!buf_LRU_block_remove_hashed(bpage, zip, false, b != nullptr)) { mutex_exit(&buf_pool->LRU_list_mutex); if (b != nullptr) { @@ -2004,9 +2037,11 @@ bool buf_LRU_free_page(buf_page_t *bpage, bool zip) { } ut_ad(!mutex_own(block_mutex)); - /* buf_LRU_block_remove_hashed() releases the hash_lock */ - ut_ad(!rw_lock_own(hash_lock, RW_LOCK_X) && - !rw_lock_own(hash_lock, RW_LOCK_S)); + /* buf_LRU_block_remove_hashed() releases the hash_lock, unless it was + asked to keep it for the re-insert of the compressed-only descriptor. */ + ut_ad(b != nullptr ? rw_lock_own(hash_lock, RW_LOCK_X) + : (!rw_lock_own(hash_lock, RW_LOCK_X) && + !rw_lock_own(hash_lock, RW_LOCK_S))); /* We have just freed a BUF_BLOCK_FILE_PAGE. If b != nullptr then it was a compressed page with an uncompressed frame and @@ -2015,10 +2050,24 @@ bool buf_LRU_free_page(buf_page_t *bpage, bool zip) { into the LRU and page_hash (and possibly flush_list). if b == nullptr then it was a regular page that has been freed */ + /* Widen the window between the page hash delete and the re-insert of + the compressed-only descriptor. The hash cell X-latch is held here + (keep_hash_lock), so a concurrent buf_page_init_for_read() of this page + must block on the cell latch instead of inserting a second descriptor + for the same page id into the gap. */ + DBUG_EXECUTE_IF( + "buf_lru_free_page_delay_zip_reinsert", if (b != nullptr) { + std::this_thread::sleep_for(std::chrono::microseconds(100)); + }); + if (b != nullptr) { auto prev_b = UT_LIST_GET_PREV(LRU, b); - rw_lock_x_lock(hash_lock, UT_LOCATION_HERE); + /* The hash cell X-latch has been held continuously since before the + HASH_DELETE in buf_LRU_block_remove_hashed() (see keep_hash_lock), + which is what makes the assertion below provable: no other thread can + have inserted a descriptor for this page id in the meantime. */ + ut_ad(rw_lock_own(hash_lock, RW_LOCK_X)); mutex_enter(block_mutex); @@ -2231,7 +2280,8 @@ the object will be freed. The caller must hold buf_pool->LRU_list_mutex, the buf_page_get_mutex() mutex and the appropriate hash_lock. This function will release the -buf_page_get_mutex() and the hash_lock. +buf_page_get_mutex() and the hash_lock (the latter is kept if keep_hash_lock +is passed). If a compressed page is freed other compressed pages may be relocated. @@ -2242,12 +2292,20 @@ If a compressed page is freed other compressed pages may be relocated. compressed page of an uncompressed page @param[in] ignore_content true if should ignore page content, since it could be not initialized +@param[in] keep_hash_lock true if the hash cell X-latch should be kept + by this function instead of being released; + only allowed when the caller will re-insert + a compressed-only descriptor for this page + id into the page hash (the keep-zip path of + buf_LRU_free_page()), so that the page id is + never observably absent from the page hash @retval true if BUF_BLOCK_FILE_PAGE was removed from page_hash. The caller needs to free the page to the free list @retval false if BUF_BLOCK_ZIP_PAGE was removed from page_hash. In this case the block is already returned to the buddy allocator. */ static bool buf_LRU_block_remove_hashed(buf_page_t *bpage, bool zip, - bool ignore_content) { + bool ignore_content, + bool keep_hash_lock) { const buf_page_t *hashed_bpage; buf_pool_t *buf_pool = buf_pool_from_bpage(bpage); rw_lock_t *hash_lock; @@ -2255,6 +2313,16 @@ static bool buf_LRU_block_remove_hashed(buf_page_t *bpage, bool zip, ut_ad(mutex_own(&buf_pool->LRU_list_mutex)); ut_ad(mutex_own(buf_page_get_mutex(bpage))); + /* keep_hash_lock is only supported for the keep-zip path of + buf_LRU_free_page(): an uncompressed frame being freed while its + compressed page is kept and will be re-inserted into the page hash. + In particular the buddy allocator must not be invoked while the hash + cell X-latch is kept (buf_buddy_free() may take page hash latches via + buf_buddy_relocate()), which is guaranteed by zip == false. */ + ut_ad(!keep_hash_lock || + (!zip && buf_page_get_state(bpage) == BUF_BLOCK_FILE_PAGE && + bpage->zip.data != nullptr)); + hash_lock = buf_page_hash_lock_get(buf_pool, bpage->id); ut_ad(rw_lock_own(hash_lock, RW_LOCK_X)); @@ -2432,14 +2500,24 @@ static bool buf_LRU_block_remove_hashed(buf_page_t *bpage, bool zip, avoid relocation during the scan. But that is not possible because we are holding LRU list mutex. - 2) Not possible because in buf_page_init_for_read() - we do a look up of page_hash while holding LRU list - mutex and since we are holding LRU list mutex here - and by the time we'll release it in the caller we'd - have inserted the compressed only descriptor in the - page_hash. */ + 2) When a compressed-only descriptor will be re-inserted + for this page id (the keep-zip path of buf_LRU_free_page(), + keep_hash_lock == true), the hash cell X-latch is kept from + before the HASH_DELETE above until after the re-insert in the + caller, so the page id is never observably absent from the + page hash: a concurrent buf_page_init_for_read() blocks on + the hash cell latch and then finds the compressed-only + descriptor. (It cannot be the LRU list mutex which protects + this transition: buf_page_init_for_read() inserts into the + page hash without holding the LRU list mutex.) + When nothing will be re-inserted (keep_hash_lock == false), + the page is leaving the buffer pool for good and a concurrent + thread reading it from the disk afresh is the normal cache + miss path. */ ut_ad(mutex_own(&buf_pool->LRU_list_mutex)); - rw_lock_x_unlock(hash_lock); + if (!keep_hash_lock) { + rw_lock_x_unlock(hash_lock); + } mutex_exit(&((buf_block_t *)bpage)->mutex); if (zip && bpage->zip.data) { diff --git a/storage/innobase/buf/buf0rea.cc b/storage/innobase/buf/buf0rea.cc index eb5a666de329..847a408d3bf9 100644 --- a/storage/innobase/buf/buf0rea.cc +++ b/storage/innobase/buf/buf0rea.cc @@ -88,10 +88,17 @@ ulint buf_read_page_low(dberr_t *err, bool sync, ulint type, ulint mode, sync = true; } - /* The following call will also check if the tablespace does not exist - or is being dropped; if we succeed in initing the page in the buffer - pool for read, then DISCARD cannot proceed until the read has - completed */ + /* buf_page_init_for_read() makes the page hash-visible, io-fixed for + read and linked into the LRU list before we dispatch the read IO below. + This does not stop a concurrent tablespace drop or truncation: + tablespace deletion does not scan the buffer pool (BUF_REMOVE_NONE); it + bumps the space version and relies on the pages becoming stale. If the + tablespace is dropped before the IO is dispatched, fil_io() refuses the + read (DB_TABLESPACE_DELETED, see Fil_shard::do_io()) and the page is + removed by buf_read_page_handle_error() below. If a read completes + against a space which was dropped meanwhile, the page is detected as + stale (buf_page_t::was_stale()) and freed lazily by + buf_page_free_stale(), which waits out the read io-fix. */ bpage = buf_page_init_for_read(mode, page_id, page_size, unzip); ut_a(bpage == nullptr || bpage->get_space()->id == page_id.space()); diff --git a/storage/innobase/include/buf0buf.h b/storage/innobase/include/buf0buf.h index 6c07516e33cf..59c95d890860 100644 --- a/storage/innobase/include/buf0buf.h +++ b/storage/innobase/include/buf0buf.h @@ -2349,7 +2349,24 @@ struct buf_pool_t { for all buf_pool_t-s */ BufListMutex chunks_mutex; - /** LRU list mutex */ + /** LRU list mutex. + Latching rule: no thread may WAIT for a block's frame rw-lock + (block->lock) while holding this mutex. buf_page_init_for_read() + acquires this mutex while holding the X-latch on the frame of the page + being read in, so waiting for a frame latch under this mutex would + create a deadlock cycle with that path. Consequently, a frame latch may + be taken under this mutex only with the rw_lock_*_nowait() variants: + flushing does so and handles the failure, and buf_page_create() does so + on a frame taken from the free list, asserting success (its latch is + unlocked and the block is unreachable by other threads while the page + hash X-latch is still held, so the attempt cannot fail). Compressed-only + pages (BUF_BLOCK_ZIP_PAGE descriptors) have no frame and no frame + rw-lock, so the paths handling them add no edge to this rule. + This rule cannot be expressed via latch_level_t ordering, because + block->lock is registered with SYNC_LEVEL_VARYING which LatchDebug + ignores; instead it is enforced in debug builds (with + --innodb-sync-debug) by rw_lock_assert_wait_allowed() at the rw-lock + wait entry points in sync0rw.cc. */ BufListMutex LRU_list_mutex; /** free and withdraw list mutex */ @@ -2406,7 +2423,17 @@ struct buf_pool_t { /** Hash table of buf_page_t or buf_block_t file pages, buf_page_in_file() == true, indexed by (space_id, offset). page_hash is protected by an array of - mutexes. */ + mutexes. + Membership-change protocol: a descriptor for a page id may be inserted + only after verifying the id's absence, with the cell's X-latch held + continuously from that verification until the insert. Conversely, a + remover which will re-insert a descriptor for the same page id (the + keep-zip path of buf_LRU_free_page()) must keep the cell's X-latch held + continuously from the delete until the re-insert, so that the page id is + never observably absent from the hash while the page is still logically + in the buffer pool. Note that buf_page_init_for_read() inserts while + holding only the cell's X-latch (not the LRU list mutex), so the LRU + list mutex does NOT stabilize page hash membership. */ hash_table_t *page_hash; /** Hash table of buf_block_t blocks whose frames are allocated to the zip diff --git a/storage/innobase/include/buf0buf.ic b/storage/innobase/include/buf0buf.ic index ca8d7612a137..4a16d156725d 100644 --- a/storage/innobase/include/buf0buf.ic +++ b/storage/innobase/include/buf0buf.ic @@ -512,7 +512,12 @@ static inline bool buf_page_can_relocate( { ut_ad(mutex_own(buf_page_get_mutex(bpage))); ut_ad(buf_page_in_file(bpage)); - ut_ad(bpage->in_LRU_list); + /* A page whose read IO is still in progress may be hash-visible but not + yet linked into the LRU list: buf_page_init_for_read() links it into the + LRU list only after making it hash-visible. Such a page is io-fixed for + read (and not relocatable), which is the only legitimate way to reach + this function for a page which is not in the LRU list. */ + ut_ad(bpage->in_LRU_list || buf_page_get_io_fix(bpage) == BUF_IO_READ); return (buf_page_get_io_fix(bpage) == BUF_IO_NONE && bpage->buf_fix_count == 0); diff --git a/storage/innobase/sync/sync0debug.cc b/storage/innobase/sync/sync0debug.cc index 07ba18a6ce78..151ec5edca7e 100644 --- a/storage/innobase/sync/sync0debug.cc +++ b/storage/innobase/sync/sync0debug.cc @@ -38,6 +38,7 @@ this program; if not, write to the Free Software Foundation, Inc., *******************************************************/ #include "sync0debug.h" +#include "univ.i" #include #include @@ -684,7 +685,12 @@ const latch_t *LatchDebug::find(const Latches *latches, @param[in] level The level to lookup @return latch if found or NULL */ const latch_t *LatchDebug::find(latch_level_t level) UNIV_NOTHROW { - return (find(thread_latches(), level)); + /* A thread which has not acquired any latch yet has no latch list + allocated (thread_latches() without the create flag returns nullptr) + and thus holds nothing at any level. */ + const Latches *latches = thread_latches(); + + return (latches != nullptr ? find(latches, level) : nullptr); } /** @@ -1269,7 +1275,9 @@ static void sync_latch_meta_init() UNIV_NOTHROW { LATCH_ADD_MUTEX(FTS_PLL_TOKENIZE, SYNC_FTS_TOKENIZE, fts_pll_tokenize_mutex_key); - LATCH_ADD_MUTEX(HASH_TABLE_MUTEX, SYNC_BUF_PAGE_HASH, hash_table_mutex_key); + /* Do not use SYNC_BUF_PAGE_HASH for HASH_TABLE_MUTEX. + @see buf_buddy_no_page_hash_latch_validate(). */ + LATCH_ADD_MUTEX(HASH_TABLE_MUTEX, SYNC_ANY_LATCH, hash_table_mutex_key); LATCH_ADD_MUTEX(IBUF_BITMAP, SYNC_IBUF_BITMAP_MUTEX, ibuf_bitmap_mutex_key); diff --git a/storage/innobase/sync/sync0rw.cc b/storage/innobase/sync/sync0rw.cc index 3520334da16a..7227ad123baa 100644 --- a/storage/innobase/sync/sync0rw.cc +++ b/storage/innobase/sync/sync0rw.cc @@ -307,6 +307,42 @@ rw_lock_t::~rw_lock_t() { ut_d(magic_n = 0); } +#ifdef UNIV_DEBUG +/** Asserts that the calling thread is allowed to start waiting for the +given rw-lock. + +This is the hook for latch-order rules which cannot be expressed through +latch_level_t ordering and which constrain only actual waits, not +non-blocking acquisitions (the rw_lock_*_nowait() variants never reach +this, by design: an acquisition that never waits cannot participate in a +deadlock cycle). It is called at the only spots where an rw-lock +acquisition starts to wait - every unbounded wait, spinning included, +funnels into a sync array reservation. + +Currently there is one such rule, for the buffer block frame locks +(buf_block_t::lock): a thread must not wait for a frame lock while +holding any buffer pool LRU list mutex. buf_page_init_for_read() acquires +buf_pool_t::LRU_list_mutex while holding the X-latch on the frame of the +page being read in, so such a wait would form a deadlock cycle with that +path. The rule cannot use latch levels because buf_block_t::lock is +registered with SYNC_LEVEL_VARYING (B-tree page latching order genuinely +varies), and LatchDebug ignores such latches entirely - both when they +are acquired and when they are held. The code paths which do latch a +frame under the LRU list mutex (flushing, and buf_page_create() on a +block freshly taken from the free list) use the nowait variants, so they +are exempt by construction. + +Note that this check is only effective when LatchDebug is enabled +(--innodb-sync-debug); an actual deadlock occurrence is additionally caught +by the sync array deadlock detector, which does track frame rw-locks. */ +namespace { +void rw_lock_assert_wait_allowed(const rw_lock_t *lock) { + ut_ad(lock->get_id() != LATCH_ID_BUF_BLOCK_LOCK || + sync_check_find(SYNC_BUF_LRU_LIST) == nullptr); +} +} // namespace +#endif /* UNIV_DEBUG */ + void rw_lock_s_lock_spin(rw_lock_t *lock, ulint pass, ut::Location location) { ulint i = 0; /* spin round count */ sync_array_t *sync_arr; @@ -347,6 +383,8 @@ void rw_lock_s_lock_spin(rw_lock_t *lock, ulint pass, ut::Location location) { ++count_os_wait; + ut_d(rw_lock_assert_wait_allowed(lock)); + sync_cell_t *cell; sync_arr = @@ -429,6 +467,8 @@ static inline void rw_lock_x_lock_wait_func(rw_lock_t *lock, } /* If there is still a reader, then go to sleep.*/ + ut_d(rw_lock_assert_wait_allowed(lock)); + sync_cell_t *cell; sync_arr = sync_array_get_and_reserve_cell(lock, RW_LOCK_X_WAIT, @@ -649,6 +689,8 @@ void rw_lock_x_lock_func(rw_lock_t *lock, ulint pass, ut::Location location) { } } + ut_d(rw_lock_assert_wait_allowed(lock)); + sync_cell_t *cell; sync_arr = sync_array_get_and_reserve_cell(lock, RW_LOCK_X, location, &cell); @@ -714,6 +756,8 @@ void rw_lock_sx_lock_func(rw_lock_t *lock, ulint pass, ut::Location location) { } } + ut_d(rw_lock_assert_wait_allowed(lock)); + sync_cell_t *cell; sync_arr = sync_array_get_and_reserve_cell(lock, RW_LOCK_SX, location, &cell); From a22f316523a3773cfac516506ed32230ee86de03 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Pawe=C5=82=20Olchawa?= Date: Tue, 21 Jul 2026 19:42:34 +0200 Subject: [PATCH 24/32] PS-11445 [trunk]: [1] Restore per-buffer-pool LRU manager thread https://perconadev.atlassian.net/browse/PS-11445 Minimal restore of the pre-removal per-buffer-pool LRU manager thread (reverted 97d7eba97d4, PS-9071: Merge MySQL 8.3.0 - remove multithreaded asynchronous LRU flusher), completed and reconciled with the current 8.4 codebase: - Removes the recv_writer thread and its mutex/PFS key/latch level. It was reintroduced by 97d7eba97d4 as part of adopting upstream's recovery-time LRU flushing model; with the dedicated per-instance LRU manager threads restored, recv_writer's role is redundant and its removal is required for the restore to be self-consistent (recv_sys_t's writer_mutex/flush_type fields don't survive the revert, since later 8.4 development touched this area too, so leaving recv_writer's now-dangling references in place would not compile). - Reuses buf_flush_page_cleaner_disabled_debug for the new thread instead of adding a new sysvar, keeping the patch smaller (the flush/evict/scan stats rework is a separate commit). - Completes the thread lifecycle: joins the LRU manager threads on page-cleaner shutdown (missing after a plain revert) and fixes the m_lru_managers array to be freed with the allocator that matches how it is allocated. - Skips the dblwr::force_flush() call in buf_flush_end() when a batch flushed no pages (e.g. an LRU batch that only evicted clean pages), since nothing was written to the doublewrite buffer for it to flush and there is no reason to take the dblwr instance mutex. - Drops dead per-slot LRU accounting from the page cleaner (superseded by the restored thread) and an unrelated dead-code leftover in Flush_observer (inc_estimate/m_estimate/m_lsn). - Updates MTR results/tests across affected suites for the restored thread's effect on thread lists and shutdown/flush ordering. --- .../flush_dirty_pages_and_stop_flushing.inc | 3 +- .../perfschema/r/dml_setup_threads.result | 3 +- .../suite/perfschema/t/dml_setup_threads.test | 2 +- storage/innobase/buf/buf0buf.cc | 31 +- storage/innobase/buf/buf0flu.cc | 349 +++++++++++------- storage/innobase/handler/ha_innodb.cc | 4 +- storage/innobase/include/buf0buf.h | 6 + storage/innobase/include/buf0flu.h | 7 +- storage/innobase/include/log0recv.h | 8 - storage/innobase/include/srv0srv.h | 13 +- storage/innobase/include/sync0sync.h | 1 - storage/innobase/include/sync0types.h | 6 +- storage/innobase/log/log0recv.cc | 123 +----- storage/innobase/srv/srv0mon.cc | 8 +- storage/innobase/srv/srv0srv.cc | 18 + storage/innobase/srv/srv0start.cc | 9 +- storage/innobase/sync/sync0debug.cc | 4 - storage/innobase/sync/sync0sync.cc | 1 - 18 files changed, 301 insertions(+), 295 deletions(-) diff --git a/mysql-test/suite/innodb/include/flush_dirty_pages_and_stop_flushing.inc b/mysql-test/suite/innodb/include/flush_dirty_pages_and_stop_flushing.inc index 5f5cf92506dd..c2157c40830d 100644 --- a/mysql-test/suite/innodb/include/flush_dirty_pages_and_stop_flushing.inc +++ b/mysql-test/suite/innodb/include/flush_dirty_pages_and_stop_flushing.inc @@ -16,5 +16,6 @@ SET GLOBAL innodb_log_checkpoint_now = ON; # Disable page cleaner threads, because we are going to fill # redo log and we want to collect a group of dirty pages for -# which the redo log records protect changes. +# which the redo log records protect changes. This also pauses the +# per-instance LRU manager threads, which reuse this same debug flag. SET GLOBAL innodb_page_cleaner_disabled_debug = ON; \ No newline at end of file diff --git a/mysql-test/suite/perfschema/r/dml_setup_threads.result b/mysql-test/suite/perfschema/r/dml_setup_threads.result index 4348f7406098..517723b6dca5 100644 --- a/mysql-test/suite/perfschema/r/dml_setup_threads.result +++ b/mysql-test/suite/perfschema/r/dml_setup_threads.result @@ -1,8 +1,9 @@ select * from performance_schema.setup_threads; select * from performance_schema.setup_threads -order by name limit 13; +order by name limit 14; NAME ENABLED HISTORY PROPERTIES VOLATILITY DOCUMENTATION thread/innodb/buf_dump_thread YES YES singleton 0 NULL +thread/innodb/buf_lru_manager_thread YES YES 0 NULL thread/innodb/buf_pool_create_thread YES YES singleton 0 NULL thread/innodb/buf_resize_thread YES YES singleton 0 NULL thread/innodb/bulk_alloc_thread YES YES singleton 0 NULL diff --git a/mysql-test/suite/perfschema/t/dml_setup_threads.test b/mysql-test/suite/perfschema/t/dml_setup_threads.test index 2d3685a1a661..fd7ace8e1db8 100644 --- a/mysql-test/suite/perfschema/t/dml_setup_threads.test +++ b/mysql-test/suite/perfschema/t/dml_setup_threads.test @@ -12,7 +12,7 @@ select * from performance_schema.setup_threads; --enable_result_log select * from performance_schema.setup_threads - order by name limit 13; + order by name limit 14; --disable_result_log select * from performance_schema.setup_threads diff --git a/storage/innobase/buf/buf0buf.cc b/storage/innobase/buf/buf0buf.cc index 12262e42bd2c..c6926c9ca279 100644 --- a/storage/innobase/buf/buf0buf.cc +++ b/storage/innobase/buf/buf0buf.cc @@ -58,6 +58,7 @@ this program; if not, write to the Free Software Foundation, Inc., #include "log0buf.h" #include "log0chkp.h" #include "page0page.h" +#include "scope_guard.h" #include "sync0rw.h" #include "trx0purge.h" #include "trx0undo.h" @@ -1521,6 +1522,9 @@ static void buf_pool_create(buf_pool_t *buf_pool, ulint buf_pool_size, os_event_set(buf_pool->no_flush[i]); } + buf_pool->run_lru = os_event_create(); + os_event_set(buf_pool->run_lru); + buf_pool->watch = (buf_page_t *)ut::zalloc_withkey( UT_NEW_THIS_FILE_PSI_KEY, sizeof(*buf_pool->watch) * BUF_POOL_WATCH_SIZE); for (i = 0; i < BUF_POOL_WATCH_SIZE; i++) { @@ -1619,6 +1623,8 @@ static void buf_pool_free_instance(buf_pool_t *buf_pool) { os_event_destroy(buf_pool->no_flush[i]); } + os_event_destroy(buf_pool->run_lru); + ut::free(buf_pool->chunks); mutex_exit(&buf_pool->chunks_mutex); mutex_free(&buf_pool->chunks_mutex); @@ -6475,20 +6481,21 @@ static void buf_refresh_io_stats(buf_pool_t *buf_pool) { /** Invalidates file pages in one buffer pool instance @param[in] buf_pool buffer pool instance */ static void buf_pool_invalidate_instance(buf_pool_t *buf_pool) { + ulint i; + ut_ad(!mutex_own(&buf_pool->LRU_list_mutex)); - for (size_t i = BUF_FLUSH_LRU; i < BUF_FLUSH_N_TYPES; i++) { - /* As this function is called during startup and during redo application - phase during recovery, a flush might be requested either by - recv_writer thread (which is not started yet, or paused by writer_mutex), or - by our own thread (in which case we wait for it to finish initialization). - No new write batch can be in initialization stage at this point. - This also explains why we don't need flush_state_mutex to assert this. */ - ut_ad(!buf_pool->init_flush[i]); - - /* However, it is possible that a write batch that has been posted earlier - is still not complete. For buffer pool invalidation to proceed we must - ensure there is NO write activity happening. */ + os_event_reset(buf_pool->run_lru); + auto guard = create_scope_guard([&]() { os_event_set(buf_pool->run_lru); }); + + for (i = BUF_FLUSH_LRU; i < BUF_FLUSH_N_TYPES; i++) { + /* Although this function is called during startup and + during redo application phase during recovery, Percona InnoDB + might be running several LRU manager threads at this stage. + Hence, a new write batch can be in initialization stage at this point. */ + + /* For buffer pool invalidation to proceed we must ensure there is NO + write activity happening. */ buf_flush_await_no_flushing(buf_pool, static_cast(i)); } diff --git a/storage/innobase/buf/buf0flu.cc b/storage/innobase/buf/buf0flu.cc index 2060fbd3901b..b4e139165880 100644 --- a/storage/innobase/buf/buf0flu.cc +++ b/storage/innobase/buf/buf0flu.cc @@ -95,6 +95,7 @@ lsn_t get_flush_sync_lsn() noexcept { return buf_flush_sync_lsn; } #ifdef UNIV_PFS_THREAD mysql_pfs_key_t page_flush_thread_key; mysql_pfs_key_t page_flush_coordinator_thread_key; +mysql_pfs_key_t buf_lru_manager_thread_key; #endif /* UNIV_PFS_THREAD */ /** Event to synchronise with the flushing. */ @@ -125,8 +126,8 @@ struct page_cleaner_slot_t { protected by page_cleaner_t::mutex if the worker thread got the slot and set to PAGE_CLEANER_STATE_FLUSHING, - n_flushed_lru and n_flushed_list can be - updated only by the worker thread */ + n_flushed_list can be updated only by + the worker thread */ /* This value is set during state==PAGE_CLEANER_STATE_NONE */ ulint n_pages_requested; /*!< number of requested pages @@ -134,22 +135,15 @@ struct page_cleaner_slot_t { /* These values are updated during state==PAGE_CLEANER_STATE_FLUSHING, and committed with state==PAGE_CLEANER_STATE_FINISHED. The consistency is protected by the 'state' */ - ulint n_flushed_lru; - /*!< number of flushed pages - by LRU scan flushing */ ulint n_flushed_list; /*!< number of flushed pages by flush_list flushing */ bool succeeded_list; /*!< true if flush_list flushing succeeded. */ - std::chrono::milliseconds flush_lru_time; - /*!< elapsed time for LRU flushing */ std::chrono::milliseconds flush_list_time; /*!< elapsed time for flush_list flushing */ - ulint flush_lru_pass; - /*!< count to attempt LRU flushing */ ulint flush_list_pass; /*!< count to attempt flush_list flushing */ @@ -224,6 +218,10 @@ static void buf_flush_page_coordinator_thread(); /** Worker thread of page_cleaner. */ static void buf_flush_page_cleaner_thread(); +/** LRU manager thread for performing LRU flushes for buffer pool free +list refill. One thread is created for each buffer pool instance. */ +static void buf_lru_manager_thread(size_t buf_pool_instance); + /** Increases flush_list size in bytes with the page size in inline function */ static inline void incr_flush_list_size_in_bytes( buf_block_t *block, /*!< in: control block */ @@ -1503,8 +1501,11 @@ it is a best effort attempt and it is not guaranteed that after a call to this function there will be 'max' blocks in the free list. @param[in] buf_pool buffer pool instance @param[in] max desired number for blocks in the free_list -@return number of blocks for which the write request was queued. */ -static ulint buf_flush_LRU_list_batch(buf_pool_t *buf_pool, ulint max) { +@return pair of numbers where first number is the blocks for which +flush request is queued and second is the number of blocks that were +clean and simply evicted from the LRU. */ +static std::pair buf_flush_LRU_list_batch(buf_pool_t *buf_pool, + ulint max) { buf_page_t *bpage; ulint scanned = 0; ulint evict_count = 0; @@ -1582,7 +1583,7 @@ static ulint buf_flush_LRU_list_batch(buf_pool_t *buf_pool, ulint max) { MONITOR_LRU_BATCH_SCANNED_PER_CALL, scanned); } - return (count); + return (std::make_pair(count, evict_count)); } /** Flush and move pages from LRU or unzip_LRU list to the free list. @@ -1592,20 +1593,26 @@ Whether LRU or unzip_LRU is used depends on the state of the system. @return number of blocks for which either the write request was queued or in case of unzip_LRU the number of blocks actually moved to the free list */ -static ulint buf_do_LRU_batch(buf_pool_t *buf_pool, ulint max) { +static std::pair buf_do_LRU_batch(buf_pool_t *buf_pool, + ulint max) { ulint count = 0; + std::pair res; ut_ad(mutex_own(&buf_pool->LRU_list_mutex)); if (buf_LRU_evict_from_unzip_LRU(buf_pool)) { - count += buf_free_from_unzip_LRU_list_batch(buf_pool, max); + count = buf_free_from_unzip_LRU_list_batch(buf_pool, max); } if (max > count) { - count += buf_flush_LRU_list_batch(buf_pool, max - count); + res = buf_flush_LRU_list_batch(buf_pool, max - count); } - return (count); + /* Add evicted pages from unzip_LRU to the evicted pages from the simple + LRU. */ + res.second += count; + + return (res); } /** This utility flushes dirty blocks from the end of the flush_list. @@ -1687,9 +1694,10 @@ not guaranteed that the actual number is that big, though) @param[in] lsn_limit in the case of BUF_FLUSH_LIST all blocks whose oldest_modification is smaller than this should be flushed (if their number does not exceed min_n), otherwise ignored -@return number of blocks for which the write request was queued */ -static ulint buf_flush_batch(buf_pool_t *buf_pool, buf_flush_t flush_type, - ulint min_n, lsn_t lsn_limit) { +@return pair of numbers of flushed and evicted blocks */ +static std::pair buf_flush_batch(buf_pool_t *buf_pool, + buf_flush_t flush_type, + ulint min_n, lsn_t lsn_limit) { ut_ad(flush_type == BUF_FLUSH_LRU || flush_type == BUF_FLUSH_LIST); #ifdef UNIV_DEBUG @@ -1700,27 +1708,29 @@ static ulint buf_flush_batch(buf_pool_t *buf_pool, buf_flush_t flush_type, } #endif /* UNIV_DEBUG */ - ulint count = 0; + std::pair res; /* Note: The buffer pool mutexes is released and reacquired within the flush functions. */ switch (flush_type) { case BUF_FLUSH_LRU: mutex_enter(&buf_pool->LRU_list_mutex); - count = buf_do_LRU_batch(buf_pool, min_n); + res = buf_do_LRU_batch(buf_pool, min_n); mutex_exit(&buf_pool->LRU_list_mutex); break; case BUF_FLUSH_LIST: - count = buf_do_flush_list_batch(buf_pool, min_n, lsn_limit); + res.first = buf_do_flush_list_batch(buf_pool, min_n, lsn_limit); + res.second = 0; break; default: ut_error; } - DBUG_PRINT("ib_buf", ("flush %u completed, %u pages", unsigned(flush_type), - unsigned(count))); + DBUG_PRINT("ib_buf", + ("flush %u completed, flushed %u pages, evicted %u pages", + unsigned(flush_type), unsigned(res.first), unsigned(res.second))); - return (count); + return (res); } /** Gather the aggregated stats for both flush list and LRU list flushing. @@ -1756,7 +1766,8 @@ static bool buf_flush_start(buf_pool_t *buf_pool, buf_flush_t flush_type) { /** End a buffer flush batch for LRU or flush list @param[in] buf_pool buffer pool instance @param[in] flush_type BUF_FLUSH_LRU or BUF_FLUSH_LIST */ -static void buf_flush_end(buf_pool_t *buf_pool, buf_flush_t flush_type) { +static void buf_flush_end(buf_pool_t *buf_pool, buf_flush_t flush_type, + ulint flushed_page_count) { buf_pool->change_flush_state(flush_type, [&]() { buf_pool->try_LRU_scan = true; buf_pool->init_flush[flush_type] = false; @@ -1764,7 +1775,12 @@ static void buf_flush_end(buf_pool_t *buf_pool, buf_flush_t flush_type) { if (!srv_read_only_mode) { if (dblwr::is_enabled()) { - dblwr::force_flush(flush_type, buf_pool_index(buf_pool)); + /* Nothing was written to the doublewrite buffer when the batch + flushed no pages (e.g. an LRU batch that only evicted clean pages), + so there is no reason to take the dblwr instance mutex. */ + if (flushed_page_count != 0) { + dblwr::force_flush(flush_type, buf_pool_index(buf_pool)); + } } else { buf_flush_sync_datafiles(); } @@ -1797,12 +1813,12 @@ bool buf_flush_do_batch(buf_pool_t *buf_pool, buf_flush_t type, ulint min_n, return (false); } - ulint page_count = buf_flush_batch(buf_pool, type, min_n, lsn_limit); + const auto res = buf_flush_batch(buf_pool, type, min_n, lsn_limit); - buf_flush_end(buf_pool, type); + buf_flush_end(buf_pool, type, res.first); if (n_processed != nullptr) { - *n_processed = page_count; + *n_processed = res.first + res.second; } return (true); @@ -1954,7 +1970,7 @@ Clears up tail of the LRU list of a given buffer pool instance: The depth to which we scan each buffer pool is controlled by dynamic config parameter innodb_LRU_scan_depth. @param buf_pool buffer pool instance -@return total pages flushed */ +@return total pages flushed and evicted */ static ulint buf_flush_LRU_list(buf_pool_t *buf_pool) { ulint scan_depth, withdraw_depth; ulint n_flushed = 0; @@ -2109,13 +2125,9 @@ void set_average() { slot = &page_cleaner->slots[i]; - lru_tm += slot->flush_lru_time.count(); - lru_pass += slot->flush_lru_pass; list_tm += slot->flush_list_time.count(); list_pass += slot->flush_list_pass; - slot->flush_lru_time = std::chrono::seconds::zero(); - slot->flush_lru_pass = 0; slot->flush_list_time = std::chrono::seconds::zero(); slot->flush_list_pass = 0; } @@ -2531,6 +2543,15 @@ bool buf_flush_page_cleaner_is_active() { return (srv_thread_is_active(srv_threads.m_page_cleaner_coordinator)); } +/** Returns the count of currently active LRU manager threads. */ +size_t buf_flush_active_lru_managers() { + size_t count = 0; + for (size_t i = 0; i < srv_threads.m_lru_managers_n; ++i) { + count += (srv_thread_is_active(srv_threads.m_lru_managers[i]) ? 1 : 0); + } + return count; +} + void buf_flush_page_cleaner_init() { ut_ad(page_cleaner == nullptr); @@ -2560,6 +2581,15 @@ void buf_flush_page_cleaner_init() { /* Make sure page cleaner is active. */ ut_a(buf_flush_page_cleaner_is_active()); + + /* One LRU manager thread per buf_pool instance. */ + for (size_t i = 0; i < srv_threads.m_lru_managers_n; ++i) { + srv_threads.m_lru_managers[i] = os_thread_create( + buf_lru_manager_thread_key, i, buf_lru_manager_thread, i); + srv_threads.m_lru_managers[i].start(); + } + + ut_a(buf_flush_active_lru_managers() == srv_buf_pool_instances); } /** @@ -2572,6 +2602,14 @@ static void buf_flush_page_cleaner_close(void) { srv_threads.m_page_cleaner_workers[i].wait(); } + /* Wait for all LRU manager threads to exit. They observe + srv_shutdown_state > SRV_SHUTDOWN_CLEANUP and break out of their + loops; the run_lru event is always set during normal shutdown so they + do not block. */ + for (size_t i = 0; i < srv_threads.m_lru_managers_n; ++i) { + srv_threads.m_lru_managers[i].wait(); + } + mutex_destroy(&page_cleaner->mutex); os_event_destroy(page_cleaner->is_finished); @@ -2636,9 +2674,7 @@ static void pc_request(ulint min_n, lsn_t lsn_limit) { Do flush for one slot. @return the number of the slots which has not been treated yet. */ static ulint pc_flush_slot(void) { - std::chrono::steady_clock::duration lru_time; std::chrono::steady_clock::duration flush_list_time{}; - int lru_pass = 0; int list_pass = 0; mutex_enter(&page_cleaner->mutex); @@ -2670,36 +2706,25 @@ static ulint pc_flush_slot(void) { } if (!page_cleaner->is_running) { - slot->n_flushed_lru = 0; slot->n_flushed_list = 0; } else { mutex_exit(&page_cleaner->mutex); - const auto lru_start = std::chrono::steady_clock::now(); + /* Flush pages from flush_list if required. LRU-tail flushing is the + responsibility of the per-pool buf_lru_manager_thread (one per + buf_pool instance); the page cleaner only flushes the flush_list. */ + if (page_cleaner->requested) { + const auto flush_list_start = std::chrono::steady_clock::now(); - /* Flush pages from end of LRU if required */ - slot->n_flushed_lru = buf_flush_LRU_list(buf_pool); + slot->succeeded_list = buf_flush_do_batch( + buf_pool, BUF_FLUSH_LIST, slot->n_pages_requested, + page_cleaner->lsn_limit, &slot->n_flushed_list); - lru_time = std::chrono::steady_clock::now() - lru_start; - lru_pass = 1; - - if (!page_cleaner->is_running) { - slot->n_flushed_list = 0; + flush_list_time = std::chrono::steady_clock::now() - flush_list_start; + list_pass = 1; } else { - /* Flush pages from flush_list if required */ - if (page_cleaner->requested) { - const auto flush_list_start = std::chrono::steady_clock::now(); - - slot->succeeded_list = buf_flush_do_batch( - buf_pool, BUF_FLUSH_LIST, slot->n_pages_requested, - page_cleaner->lsn_limit, &slot->n_flushed_list); - - flush_list_time = std::chrono::steady_clock::now() - flush_list_start; - list_pass = 1; - } else { - slot->n_flushed_list = 0; - slot->succeeded_list = true; - } + slot->n_flushed_list = 0; + slot->succeeded_list = true; } mutex_enter(&page_cleaner->mutex); } @@ -2707,11 +2732,8 @@ static ulint pc_flush_slot(void) { page_cleaner->n_slots_finished++; slot->state = PAGE_CLEANER_STATE_FINISHED; - slot->flush_lru_time += - std::chrono::duration_cast(lru_time); slot->flush_list_time += std::chrono::duration_cast(flush_list_time); - slot->flush_lru_pass += lru_pass; slot->flush_list_pass += list_pass; if (page_cleaner->n_slots_requested == 0 && @@ -2729,14 +2751,12 @@ static ulint pc_flush_slot(void) { /** Wait until all flush requests are finished. -@param n_flushed_lru number of pages flushed from the end of the LRU list. @param n_flushed_list number of pages flushed from the end of the flush_list. @return true if all flush_list flushing batch were success. */ -static bool pc_wait_finished(ulint *n_flushed_lru, ulint *n_flushed_list) { +static bool pc_wait_finished(ulint *n_flushed_list) { bool all_succeeded = true; - *n_flushed_lru = 0; *n_flushed_list = 0; os_event_wait(page_cleaner->is_finished); @@ -2752,7 +2772,6 @@ static bool pc_wait_finished(ulint *n_flushed_lru, ulint *n_flushed_list) { ut_ad(slot->state == PAGE_CLEANER_STATE_FINISHED); - *n_flushed_lru += slot->n_flushed_lru; *n_flushed_list += slot->n_flushed_list; all_succeeded &= slot->succeeded_list; @@ -2774,7 +2793,7 @@ static bool pc_wait_finished(ulint *n_flushed_lru, ulint *n_flushed_list) { #ifdef UNIV_LINUX /** -Set priority for page_cleaner threads. +Set priority for page_cleaner and LRU manager threads. @param[in] priority priority intended to set @return true if set as intended */ static bool buf_flush_page_cleaner_set_priority(int priority) { @@ -2784,15 +2803,15 @@ static bool buf_flush_page_cleaner_set_priority(int priority) { #endif /* UNIV_LINUX */ #ifdef UNIV_DEBUG -/** Loop used to disable page cleaner threads. */ +/** Loop used to disable page cleaner and LRU manager threads. */ static void buf_flush_page_cleaner_disabled_loop(void) { - ut_ad(page_cleaner != nullptr); - if (!innodb_page_cleaner_disabled_debug) { /* We return to avoid entering and exiting mutex. */ return; } + ut_ad(page_cleaner != nullptr); + mutex_enter(&page_cleaner->mutex); page_cleaner->n_disabled_debug++; mutex_exit(&page_cleaner->mutex); @@ -2834,7 +2853,7 @@ void buf_flush_page_cleaner_disabled_debug_update(THD *, SYS_VAR *, void *, innodb_page_cleaner_disabled_debug = false; - /* Enable page cleaner threads. */ + /* Enable page cleaner and LRU manager threads. */ while (srv_shutdown_state.load() < SRV_SHUTDOWN_CLEANUP) { mutex_enter(&page_cleaner->mutex); const ulint n = page_cleaner->n_disabled_debug; @@ -2867,9 +2886,11 @@ void buf_flush_page_cleaner_disabled_debug_update(THD *, SYS_VAR *, void *, mutex_enter(&page_cleaner->mutex); - ut_ad(page_cleaner->n_disabled_debug <= srv_n_page_cleaners); + const auto all_flushing_threads = + srv_n_page_cleaners + srv_buf_pool_instances; + ut_ad(page_cleaner->n_disabled_debug <= all_flushing_threads); - if (page_cleaner->n_disabled_debug == srv_n_page_cleaners) { + if (page_cleaner->n_disabled_debug == all_flushing_threads) { mutex_exit(&page_cleaner->mutex); break; } @@ -2899,8 +2920,9 @@ static void buf_flush_page_coordinator_thread() { << buf_flush_page_cleaner_priority; } else { ib::info(ER_IB_MSG_127) << "If the mysqld execution user is authorized," - " page cleaner thread priority can be changed." - " See the man page of setpriority()."; + " page cleaner and LRU manager thread priority" + " can be changed. See the man page of" + " setpriority()."; } #endif /* UNIV_LINUX */ @@ -2917,7 +2939,6 @@ static void buf_flush_page_coordinator_thread() { srv_shutdown_state.load() < SRV_SHUTDOWN_CLEANUP && recv_sys->spaces != nullptr) { /* treat flushing requests during recovery. */ - ulint n_flushed_lru = 0; ulint n_flushed_list = 0; os_event_wait(recv_sys->flush_start); @@ -2927,27 +2948,13 @@ static void buf_flush_page_coordinator_thread() { break; } - switch (recv_sys->flush_type) { - case BUF_FLUSH_LRU: - /* Flush pages from end of LRU if required */ - pc_request(0, LSN_MAX); - while (pc_flush_slot() > 0) { - } - pc_wait_finished(&n_flushed_lru, &n_flushed_list); - break; - - case BUF_FLUSH_LIST: - /* Flush all pages */ - do { - pc_request(ULINT_MAX, LSN_MAX); - while (pc_flush_slot() > 0) { - } - } while (!pc_wait_finished(&n_flushed_lru, &n_flushed_list)); - break; - - default: - ut_d(ut_error); - } + /* Flush all pages. LRU flushing during recovery is handled by the + per-pool LRU manager threads, which run alongside this loop. */ + do { + pc_request(ULINT_MAX, LSN_MAX); + while (pc_flush_slot() > 0) { + } + } while (!pc_wait_finished(&n_flushed_list)); os_event_reset(recv_sys->flush_start); os_event_set(recv_sys->flush_end); @@ -2956,7 +2963,6 @@ static void buf_flush_page_coordinator_thread() { os_event_wait(buf_flush_event); ulint ret_sleep = 0; - ulint n_evicted = 0; ulint n_flushed_last = 0; ulint warn_interval = 1; ulint warn_count = 0; @@ -3015,7 +3021,7 @@ static void buf_flush_page_coordinator_thread() { ib::info(ER_IB_MSG_128) << "Page cleaner took " << diff_ms.count() << "ms to flush " - << n_flushed_last << " and evict " << n_evicted << " pages"; + << n_flushed_last << " pages"; if (warn_interval > 300) { warn_interval = 600; @@ -3034,7 +3040,7 @@ static void buf_flush_page_coordinator_thread() { } loop_start_time = curr_time; - n_flushed_last = n_evicted = 0; + n_flushed_last = 0; was_server_active = srv_check_activity(last_activity); last_activity = srv_get_activity_count(); @@ -3110,34 +3116,27 @@ static void buf_flush_page_coordinator_thread() { page_cleaner->flush_pass++; /* Wait for all slots to be finished */ - ulint n_flushed_lru = 0; ulint n_flushed_list = 0; - pc_wait_finished(&n_flushed_lru, &n_flushed_list); + pc_wait_finished(&n_flushed_list); - if (n_flushed_list > 0 || n_flushed_lru > 0) { - buf_flush_stats(n_flushed_list, n_flushed_lru); + if (n_flushed_list > 0) { + buf_flush_stats(n_flushed_list, 0); } if (n_to_flush != 0) { last_pages = n_flushed_list; } - n_evicted += n_flushed_lru; n_flushed_last += n_flushed_list; - n_flushed = n_flushed_lru + n_flushed_list; + n_flushed = n_flushed_list; if (is_sync_flush) { - MONITOR_INC_VALUE_CUMULATIVE( - MONITOR_FLUSH_SYNC_TOTAL_PAGE, MONITOR_FLUSH_SYNC_COUNT, - MONITOR_FLUSH_SYNC_PAGES, n_flushed_lru + n_flushed_list); + MONITOR_INC_VALUE_CUMULATIVE(MONITOR_FLUSH_SYNC_TOTAL_PAGE, + MONITOR_FLUSH_SYNC_COUNT, + MONITOR_FLUSH_SYNC_PAGES, n_flushed_list); } else { - if (n_flushed_lru) { - MONITOR_INC_VALUE_CUMULATIVE( - MONITOR_LRU_BATCH_FLUSH_TOTAL_PAGE, MONITOR_LRU_BATCH_FLUSH_COUNT, - MONITOR_LRU_BATCH_FLUSH_PAGES, n_flushed_lru); - } if (n_flushed_list) { MONITOR_INC_VALUE_CUMULATIVE( MONITOR_FLUSH_ADAPTIVE_TOTAL_PAGE, MONITOR_FLUSH_ADAPTIVE_COUNT, @@ -3190,6 +3189,9 @@ static void buf_flush_page_coordinator_thread() { the buffer pool but can't be sure that no new pages are being dirtied until we enter SRV_SHUTDOWN_FLUSH_PHASE phase which is the last phase (meanwhile we visit SRV_SHUTDOWN_MASTER_STOP). + Because the LRU manager thread is also flushing at SRV_SHUTDOWN_CLEANUP + but not SRV_SHUTDOWN_FLUSH_PHASE, we only leave the + SRV_SHUTDOWN_CLEANUP loop when the LRU manager quits. Note, that if we are handling fatal error, we set the state directly to EXIT_THREADS in which case we also might exit the loop @@ -3203,17 +3205,17 @@ static void buf_flush_page_coordinator_thread() { while (pc_flush_slot() > 0) { } - ulint n_flushed_lru = 0; ulint n_flushed_list = 0; - pc_wait_finished(&n_flushed_lru, &n_flushed_list); + pc_wait_finished(&n_flushed_list); - n_flushed = n_flushed_lru + n_flushed_list; + n_flushed = n_flushed_list; /* We sleep only if there are no pages to flush */ if (n_flushed == 0) { std::this_thread::sleep_for(std::chrono::milliseconds(100)); } - } while (srv_shutdown_state.load() < SRV_SHUTDOWN_FLUSH_PHASE); + } while (srv_shutdown_state.load() < SRV_SHUTDOWN_FLUSH_PHASE || + buf_flush_active_lru_managers() > 0); /* At this point all threads including the master and the purge thread must have been closed, unless we are handling some error @@ -3247,7 +3249,7 @@ static void buf_flush_page_coordinator_thread() { sweep and we'll come out of the loop leaving behind dirty pages in the flush_list */ buf_flush_await_no_flushing(nullptr, BUF_FLUSH_LIST); - buf_flush_await_no_flushing(nullptr, BUF_FLUSH_LRU); + ut_ad(buf_flush_active_lru_managers() == 0); bool success; bool are_any_read_ios_still_underway; @@ -3268,14 +3270,12 @@ static void buf_flush_page_coordinator_thread() { while (pc_flush_slot() > 0) { } - ulint n_flushed_lru = 0; ulint n_flushed_list = 0; - success = pc_wait_finished(&n_flushed_lru, &n_flushed_list); + success = pc_wait_finished(&n_flushed_list); - n_flushed = n_flushed_lru + n_flushed_list; + n_flushed = n_flushed_list; buf_flush_await_no_flushing(nullptr, BUF_FLUSH_LIST); - buf_flush_await_no_flushing(nullptr, BUF_FLUSH_LRU); } while (!success || n_flushed > 0 || are_any_read_ios_still_underway || buf_get_flush_list_len(nullptr) > 0); @@ -3382,6 +3382,107 @@ void buf_flush_sync_all_buf_pools() { buf_flush_fsync(); } +/** Make a LRU manager thread sleep until the passed target time, if it's not +already in the past. +@param[in] next_loop_time desired wake up time */ +static void buf_lru_manager_sleep_if_needed( + std::chrono::steady_clock::time_point next_loop_time) { + /* If this is the server shutdown buffer pool flushing phase, skip the + sleep to quit this thread faster */ + if (srv_shutdown_state.load() == SRV_SHUTDOWN_FLUSH_PHASE) return; + + const auto cur_time = std::chrono::steady_clock::now(); + + if (next_loop_time > cur_time) { + const auto period = std::chrono::duration_cast( + next_loop_time - cur_time); + + std::this_thread::sleep_for( + std::min(std::chrono::milliseconds{1000L}, period)); + } +} + +/** Adjust the LRU manager thread sleep time based on the free list length and +the last flush result +@param[in] buf_pool buffer pool whom we are flushing +@param[in] lru_n_flushed last LRU flush page count +@param[in,out] lru_sleep_time LRU manager thread sleep time */ +static void buf_lru_manager_adapt_sleep_time( + const buf_pool_t *buf_pool, ulint lru_n_flushed, + std::chrono::milliseconds &lru_sleep_time) { + const auto free_len = UT_LIST_GET_LEN(buf_pool->free); + const auto max_free_len = + std::min(UT_LIST_GET_LEN(buf_pool->LRU), srv_LRU_scan_depth); + + if (free_len < max_free_len / 100 && lru_n_flushed) { + /* Free list filled less than 1% and the last iteration was + able to flush, no sleep */ + lru_sleep_time = std::chrono::milliseconds::zero(); + } else if (free_len > max_free_len / 5 || + (free_len < max_free_len / 100 && lru_n_flushed == 0)) { + /* Free list filled more than 20% or no pages flushed in the + previous batch, sleep a bit more */ + lru_sleep_time += std::chrono::milliseconds{1}; + if (lru_sleep_time > std::chrono::milliseconds{1000}) + lru_sleep_time = std::chrono::milliseconds{1000}; + } else if (free_len < max_free_len / 20 && + lru_sleep_time >= std::chrono::milliseconds{50}) { + /* Free list filled less than 5%, sleep a bit less */ + lru_sleep_time -= std::chrono::milliseconds{50}; + } else { + /* Free lists filled between 5% and 20%, no change */ + } +} +/** LRU manager thread for performing LRU flushed and evictions for buffer pool +free list refill. One thread is created for each buffer pool instace. +@param[in] buf_pool_instance buffer pool instance number for this thread +*/ +static void buf_lru_manager_thread(size_t buf_pool_instance) { +#ifdef UNIV_LINUX + /* linux might be able to set different setting for each thread + worth to try to set high priority for page cleaner threads */ + if (buf_flush_page_cleaner_set_priority(buf_flush_page_cleaner_priority)) { + ib::info() << "lru_manager worker priority: " + << buf_flush_page_cleaner_priority; + } +#endif /* UNIV_LINUX */ + + ut_ad(buf_pool_instance < srv_buf_pool_instances); + + buf_pool_t *const buf_pool = buf_pool_from_array(buf_pool_instance); + + std::chrono::milliseconds lru_sleep_time{1000}; + auto next_loop_time = std::chrono::steady_clock::now() + lru_sleep_time; + ulint lru_n_flushed = 1; + + /* On server shutdown, the LRU manager thread runs through cleanup + phase to provide free pages for the master and purge threads. */ + while (srv_shutdown_state.load() == SRV_SHUTDOWN_NONE || + srv_shutdown_state.load() == SRV_SHUTDOWN_CLEANUP) { + ut_d(buf_flush_page_cleaner_disabled_loop()); + + os_event_wait(buf_pool->run_lru); + + buf_lru_manager_sleep_if_needed(next_loop_time); + + buf_lru_manager_adapt_sleep_time(buf_pool, lru_n_flushed, lru_sleep_time); + + next_loop_time = std::chrono::steady_clock::now() + lru_sleep_time; + + lru_n_flushed = buf_flush_LRU_list(buf_pool); + + buf_flush_await_no_flushing(buf_pool, BUF_FLUSH_LRU); + + if (lru_n_flushed) { + srv_stats.buf_pool_flushed.add(lru_n_flushed); + + MONITOR_INC_VALUE_CUMULATIVE( + MONITOR_LRU_BATCH_FLUSH_TOTAL_PAGE, MONITOR_LRU_BATCH_FLUSH_COUNT, + MONITOR_LRU_BATCH_FLUSH_PAGES, lru_n_flushed); + } + } +} + #if defined UNIV_DEBUG || defined UNIV_BUF_DEBUG /** Functor to validate the flush list. */ diff --git a/storage/innobase/handler/ha_innodb.cc b/storage/innobase/handler/ha_innodb.cc index 3af8d0995eff..6cd12d389d5b 100644 --- a/storage/innobase/handler/ha_innodb.cc +++ b/storage/innobase/handler/ha_innodb.cc @@ -775,7 +775,6 @@ static PSI_mutex_info all_innodb_mutexes[] = { PSI_MUTEX_KEY(dblwr_mutex, 0, 0, PSI_DOCUMENT_ME), PSI_MUTEX_KEY(purge_sys_pq_mutex, 0, 0, PSI_DOCUMENT_ME), PSI_MUTEX_KEY(recv_sys_mutex, 0, 0, PSI_DOCUMENT_ME), - PSI_MUTEX_KEY(recv_writer_mutex, 0, 0, PSI_DOCUMENT_ME), PSI_MUTEX_KEY(temp_space_rseg_mutex, 0, 0, PSI_DOCUMENT_ME), PSI_MUTEX_KEY(undo_space_rseg_mutex, 0, 0, PSI_DOCUMENT_ME), #ifdef UNIV_DEBUG @@ -886,8 +885,7 @@ static PSI_thread_info all_innodb_threads[] = { PSI_FLAG_SINGLETON, 0, PSI_DOCUMENT_ME), PSI_THREAD_KEY(log_flush_notifier_thread, "ib_log_fl_notif", PSI_FLAG_SINGLETON, 0, PSI_DOCUMENT_ME), - PSI_THREAD_KEY(recv_writer_thread, "ib_recv_write", PSI_FLAG_SINGLETON, 0, - PSI_DOCUMENT_ME), + PSI_THREAD_KEY(buf_lru_manager_thread, "ib_buf_lru", 0, 0, PSI_DOCUMENT_ME), PSI_THREAD_KEY(srv_error_monitor_thread, "ib_srv_err", PSI_FLAG_SINGLETON, 0, PSI_DOCUMENT_ME), PSI_THREAD_KEY(srv_lock_timeout_thread, "ib_srv_lock_to", diff --git a/storage/innobase/include/buf0buf.h b/storage/innobase/include/buf0buf.h index 59c95d890860..9a05cc732517 100644 --- a/storage/innobase/include/buf0buf.h +++ b/storage/innobase/include/buf0buf.h @@ -2492,6 +2492,12 @@ struct buf_pool_t { running. Protected by flush_state_mutex. */ os_event_t no_flush[BUF_FLUSH_N_TYPES]; + /* This event is always set at startup, so LRU threads do not wait for this + event. Before invalidating bufferpool, this event is reset, so the next LRU + batch flushing will wait for the event. Bufferpool invalidation needs LRU + flushing to be stopped. */ + os_event_t run_lru; + /** A sequence number used to count the number of buffer blocks removed from the end of the LRU list; NOTE that this counter may wrap around at 4 billion! A thread is allowed to read this for heuristic purposes without diff --git a/storage/innobase/include/buf0flu.h b/storage/innobase/include/buf0flu.h index d681350272e1..9d8053aa8b7a 100644 --- a/storage/innobase/include/buf0flu.h +++ b/storage/innobase/include/buf0flu.h @@ -47,6 +47,9 @@ this program; if not, write to the Free Software Foundation, Inc., /** Checks if the page_cleaner is in active state. */ bool buf_flush_page_cleaner_is_active(); +/** Returns the count of currently active LRU manager threads. */ +size_t buf_flush_active_lru_managers(); + #ifdef UNIV_DEBUG /** Value of MySQL global variable used to disable page cleaner. */ @@ -186,8 +189,8 @@ bool buf_flush_ready_for_replace(const buf_page_t *bpage); #ifdef UNIV_DEBUG struct SYS_VAR; -/** Disables page cleaner threads (coordinator and workers). -It's used by: SET GLOBAL innodb_page_cleaner_disabled_debug = 1 (0). +/** Disables page cleaner threads (coordinator and workers) and LRU manager +threads. It's used by: SET GLOBAL innodb_page_cleaner_disabled_debug = 1 (0). @param[in] thd thread handle @param[in] var pointer to system variable @param[out] var_ptr where the formal string goes diff --git a/storage/innobase/include/log0recv.h b/storage/innobase/include/log0recv.h index 2e731fb8546d..4ba08b069c6c 100644 --- a/storage/innobase/include/log0recv.h +++ b/storage/innobase/include/log0recv.h @@ -536,20 +536,12 @@ struct recv_sys_t { n_pages_to_recover, and the state field in each recv_addr struct */ ib_mutex_t mutex; - /** mutex coordinating flushing between recv_writer_thread and - the recovery thread. */ - ib_mutex_t writer_mutex; - /** event to activate page cleaner threads */ os_event_t flush_start; /** event to signal that the page cleaner has finished the request */ os_event_t flush_end; - /** type of the flush request. BUF_FLUSH_LRU: flush end of LRU, - keeping free blocks. BUF_FLUSH_LIST: flush all of blocks. */ - buf_flush_t flush_type; - #else /* !UNIV_HOTBACKUP */ bool apply_file_operations; #endif /* !UNIV_HOTBACKUP */ diff --git a/storage/innobase/include/srv0srv.h b/storage/innobase/include/srv0srv.h index 48393092e734..1b936541b628 100644 --- a/storage/innobase/include/srv0srv.h +++ b/storage/innobase/include/srv0srv.h @@ -231,9 +231,6 @@ struct Srv_threads { /** Thread doing rollbacks during recovery. */ IB_thread m_trx_recovery_rollback; - /** Thread writing recovered pages during recovery. */ - IB_thread m_recv_writer; - /** Purge coordinator (also being a worker) */ IB_thread m_purge_coordinator; @@ -254,6 +251,14 @@ struct Srv_threads { same shared state as m_page_cleaner_coordinator. */ IB_thread *m_page_cleaner_workers; + /** Number of LRU manager threads and size of array below. One per + buf_pool instance. */ + size_t m_lru_managers_n; + + /** LRU manager threads — sole owners of buf_flush_LRU_list for + free-list refill. The page cleaner only flushes the flush_list. */ + IB_thread *m_lru_managers; + /** Archiver's log archiver (used by Clone). */ IB_thread m_log_archiver; @@ -889,6 +894,7 @@ extern mysql_pfs_key_t page_archiver_thread_key; extern mysql_pfs_key_t buf_pool_create_thread_key; extern mysql_pfs_key_t buf_dump_thread_key; extern mysql_pfs_key_t buf_resize_thread_key; +extern mysql_pfs_key_t buf_lru_manager_thread_key; extern mysql_pfs_key_t clone_ddl_thread_key; extern mysql_pfs_key_t clone_gtid_thread_key; extern mysql_pfs_key_t ddl_thread_key; @@ -907,7 +913,6 @@ extern mysql_pfs_key_t log_write_notifier_thread_key; extern mysql_pfs_key_t log_flush_notifier_thread_key; extern mysql_pfs_key_t page_flush_coordinator_thread_key; extern mysql_pfs_key_t page_flush_thread_key; -extern mysql_pfs_key_t recv_writer_thread_key; extern mysql_pfs_key_t srv_error_monitor_thread_key; extern mysql_pfs_key_t srv_lock_timeout_thread_key; extern mysql_pfs_key_t srv_master_thread_key; diff --git a/storage/innobase/include/sync0sync.h b/storage/innobase/include/sync0sync.h index 8d75b6b772d1..87bf49ce8438 100644 --- a/storage/innobase/include/sync0sync.h +++ b/storage/innobase/include/sync0sync.h @@ -135,7 +135,6 @@ extern mysql_pfs_key_t recalc_pool_mutex_key; extern mysql_pfs_key_t page_cleaner_mutex_key; extern mysql_pfs_key_t purge_sys_pq_mutex_key; extern mysql_pfs_key_t recv_sys_mutex_key; -extern mysql_pfs_key_t recv_writer_mutex_key; extern mysql_pfs_key_t rtr_active_mutex_key; extern mysql_pfs_key_t rtr_match_mutex_key; extern mysql_pfs_key_t rtr_path_mutex_key; diff --git a/storage/innobase/include/sync0types.h b/storage/innobase/include/sync0types.h index 758ff4868d49..7c564c051542 100644 --- a/storage/innobase/include/sync0types.h +++ b/storage/innobase/include/sync0types.h @@ -328,8 +328,6 @@ enum latch_level_t { SYNC_TRX_I_S_RWLOCK, - SYNC_RECV_WRITER, - /** Level is varying. Only used with buffer pool page locks, which do not have a fixed level, but instead have their level set after the page is locked; see e.g. ibuf_bitmap_get_map_page(). */ @@ -407,7 +405,6 @@ enum latch_id_t { LATCH_ID_PURGE_SYS_PQ, LATCH_ID_RECALC_POOL, LATCH_ID_RECV_SYS, - LATCH_ID_RECV_WRITER, LATCH_ID_TEMP_SPACE_RSEG, LATCH_ID_UNDO_SPACE_RSEG, LATCH_ID_RW_LOCK_DEBUG, @@ -1136,8 +1133,7 @@ struct dict_sync_check : public sync_check_functor_t { if (!m_dict_mutex_allowed || (level != SYNC_DICT && level != SYNC_UNDO_SPACES && level != SYNC_FTS_CACHE && level != SYNC_DICT_OPERATION && - /* This only happens in recv_apply_hashed_log_recs. */ - level != SYNC_RECV_WRITER && level != SYNC_NO_ORDER_CHECK)) { + level != SYNC_NO_ORDER_CHECK)) { m_result = true; #ifdef UNIV_NO_ERR_MSGS ib::error() diff --git a/storage/innobase/log/log0recv.cc b/storage/innobase/log/log0recv.cc index 4adb6f847c8c..286373195ab4 100644 --- a/storage/innobase/log/log0recv.cc +++ b/storage/innobase/log/log0recv.cc @@ -190,17 +190,6 @@ is bigger than the lsn we are able to scan up to, that is an indication that the recovery failed and the database may be corrupt. */ static lsn_t recv_max_page_lsn; -#ifndef UNIV_HOTBACKUP -#ifdef UNIV_PFS_THREAD -mysql_pfs_key_t recv_writer_thread_key; -#endif /* UNIV_PFS_THREAD */ - -static bool recv_writer_is_active() { - return srv_thread_is_active(srv_threads.m_recv_writer); -} - -#endif /* !UNIV_HOTBACKUP */ - /* prototypes */ #ifndef UNIV_HOTBACKUP @@ -331,7 +320,6 @@ void recv_sys_create() { ut::zalloc_withkey(UT_NEW_THIS_FILE_PSI_KEY, sizeof(*recv_sys))); ut_a(recv_sys->last_block_first_mtr_boundary == 0); mutex_create(LATCH_ID_RECV_SYS, &recv_sys->mutex); - mutex_create(LATCH_ID_RECV_WRITER, &recv_sys->writer_mutex); recv_sys->spaces = nullptr; } @@ -431,11 +419,6 @@ void recv_sys_close() { mutex_free(&recv_sys->mutex); -#ifndef UNIV_HOTBACKUP - ut_ad(!recv_writer_is_active()); -#endif /* !UNIV_HOTBACKUP */ - mutex_free(&recv_sys->writer_mutex); - ut::free(recv_sys); recv_sys = nullptr; } @@ -701,58 +684,6 @@ void MetadataRecover::store() { mutex_exit(&dict_persist->mutex); } -/** recv_writer thread tasked with flushing dirty pages from the buffer -pools. */ -static void recv_writer_thread() { - ut_ad(!srv_read_only_mode); - - /* The code flow is as follows: - Step 1: In recv_recovery_from_checkpoint_start(). - Step 2: This recv_writer thread is started. - Step 3: In recv_recovery_from_checkpoint_finish(). - Step 4: Wait for recv_writer thread to complete. - Step 5: Assert that recv_writer thread is not active anymore. - - It is possible that the thread that is started in step 2, - becomes active only after step 4 and hence the assert in - step 5 fails. So mark this thread active only if necessary. */ - mutex_enter(&recv_sys->writer_mutex); - - if (!recv_recovery_on) { - mutex_exit(&recv_sys->writer_mutex); - return; - } - mutex_exit(&recv_sys->writer_mutex); - - while (srv_shutdown_state.load() == SRV_SHUTDOWN_NONE) { - ut_a(srv_shutdown_state_matches([](auto state) { - return state == SRV_SHUTDOWN_NONE || state == SRV_SHUTDOWN_EXIT_THREADS; - })); - - std::this_thread::sleep_for(std::chrono::milliseconds(100)); - - mutex_enter(&recv_sys->writer_mutex); - - if (!recv_recovery_on) { - mutex_exit(&recv_sys->writer_mutex); - break; - } - - if (log_test != nullptr) { - mutex_exit(&recv_sys->writer_mutex); - continue; - } - - /* Flush pages from end of LRU if required */ - os_event_reset(recv_sys->flush_end); - recv_sys->flush_type = BUF_FLUSH_LRU; - os_event_set(recv_sys->flush_start); - os_event_wait(recv_sys->flush_end); - - mutex_exit(&recv_sys->writer_mutex); - } -} - #endif /* !UNIV_HOTBACKUP */ /** Frees the recovery system. */ @@ -767,7 +698,6 @@ void recv_sys_free() { /* wake page cleaner up to progress */ if (!srv_read_only_mode) { ut_ad(!recv_recovery_on); - ut_ad(!recv_writer_is_active()); if (buf_flush_event != nullptr) { os_event_reset(buf_flush_event); } @@ -1261,16 +1191,6 @@ void recv_apply_hashed_log_recs(log_t &log) { mutex_exit(&recv_sys->mutex); - /* Stop the recv_writer thread from issuing any LRU - flush batches. */ - mutex_enter(&recv_sys->writer_mutex); - - /* Wait for any currently run batch to end. Note that BUF_FLUSH_LIST could - only be initiated by us in earlier call, but buf_pool_invalidate() waits for - all batches to finish, so only BUF_FLUSH_LRU can be running. - TBD: why is it important to wait for BUF_FLUSH_LRU to finish here? */ - buf_flush_await_no_flushing(nullptr, BUF_FLUSH_LRU); - os_event_reset(recv_sys->flush_end); /* We are about to request BUF_FLUSH_LIST, in hope to write all dirty pages @@ -1286,17 +1206,12 @@ void recv_apply_hashed_log_recs(log_t &log) { which has the same issue: skips over io-fixed pages. */ buf_pool_wait_for_no_pending_io(); - recv_sys->flush_type = BUF_FLUSH_LIST; - os_event_set(recv_sys->flush_start); os_event_wait(recv_sys->flush_end); buf_pool_invalidate(); - /* Allow batches from recv_writer thread. */ - mutex_exit(&recv_sys->writer_mutex); - ut_d(log.disable_redo_writes = false); mutex_enter(&recv_sys->mutex); @@ -3753,16 +3668,6 @@ static void recv_init_crash_recovery() { ib::info(ER_IB_MSG_727); recv_sys->dblwr->recover(); - - if (srv_force_recovery < SRV_FORCE_NO_LOG_REDO) { - /* Spawn the background thread to flush dirty pages - from the buffer pools. */ - - srv_threads.m_recv_writer = - os_thread_create(recv_writer_thread_key, 0, recv_writer_thread); - - srv_threads.m_recv_writer.start(); - } } dberr_t recv_recovery_from_checkpoint_start(log_t &log, lsn_t flush_lsn) { @@ -3950,39 +3855,17 @@ static void verify_page_type(page_id_t page_id, page_type_t type) { } MetadataRecover *recv_recovery_from_checkpoint_finish(bool aborting) { - /* Make sure that the recv_writer thread is done. This is - required because it grabs various mutexes and we want to - ensure that when we enable sync_order_checks there is no - mutex currently held by any thread. */ - mutex_enter(&recv_sys->writer_mutex); - /* Restore state. */ if (recv_sys->is_meb_db) dblwr::g_mode = recv_sys->dblwr_state; /* Free the resources of the recovery system */ recv_recovery_on = false; - /* By acquiring the mutex we ensure that the recv_writer thread won't trigger - any more LRU batches. Now wait for currently in progress batches to finish. - Note that BUF_FLUSH_LIST batches are awaited to finish before we get here. - TBD: Why is it important to wait for BUF_FLUSH_LRU to finish here? */ + /* Wait for LRU batches currently in progress (dispatched by the LRU + manager threads) to finish. Note that BUF_FLUSH_LIST batches are awaited + to finish before we get here. */ buf_flush_await_no_flushing(nullptr, BUF_FLUSH_LRU); - mutex_exit(&recv_sys->writer_mutex); - - uint32_t count = 0; - - while (recv_writer_is_active()) { - ++count; - - std::this_thread::sleep_for(std::chrono::milliseconds(100)); - - if (count >= 600) { - ib::info(ER_IB_MSG_738); - count = 0; - } - } - MetadataRecover *metadata{}; if (!aborting) { diff --git a/storage/innobase/srv/srv0mon.cc b/storage/innobase/srv/srv0mon.cc index c282f90db181..c4a6dc7822d6 100644 --- a/storage/innobase/srv/srv0mon.cc +++ b/storage/innobase/srv/srv0mon.cc @@ -404,7 +404,7 @@ static monitor_info_t innodb_counter_info[] = { "Avg time (ms) spent for adaptive flushing recently per slot.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_FLUSH_ADAPTIVE_AVG_TIME_SLOT}, - {"buffer_LRU_batch_flush_avg_time_slot", "buffer", + {"buffer_LRU_batch_flush_avg_time_slot", "buffer", // TODO: always zero "Avg time (ms) spent for LRU batch flushing recently per slot.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_LRU_BATCH_FLUSH_AVG_TIME_SLOT}, @@ -414,7 +414,7 @@ static monitor_info_t innodb_counter_info[] = { MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_FLUSH_ADAPTIVE_AVG_TIME_THREAD}, - {"buffer_LRU_batch_flush_avg_time_thread", "buffer", + {"buffer_LRU_batch_flush_avg_time_thread", "buffer", // TODO: always zero "Avg time (ms) spent for LRU batch flushing recently per thread.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_LRU_BATCH_FLUSH_AVG_TIME_THREAD}, @@ -423,7 +423,7 @@ static monitor_info_t innodb_counter_info[] = { "Estimated time (ms) spent for adaptive flushing recently.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_FLUSH_ADAPTIVE_AVG_TIME_EST}, - {"buffer_LRU_batch_flush_avg_time_est", "buffer", + {"buffer_LRU_batch_flush_avg_time_est", "buffer", // TODO: always zero "Estimated time (ms) spent for LRU batch flushing recently.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_LRU_BATCH_FLUSH_AVG_TIME_EST}, @@ -435,7 +435,7 @@ static monitor_info_t innodb_counter_info[] = { "Number of adaptive flushes passed during the recent Avg period.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_FLUSH_ADAPTIVE_AVG_PASS}, - {"buffer_LRU_batch_flush_avg_pass", "buffer", + {"buffer_LRU_batch_flush_avg_pass", "buffer", // TODO: always zero "Number of LRU batch flushes passed during the recent Avg period.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_LRU_BATCH_FLUSH_AVG_PASS}, diff --git a/storage/innobase/srv/srv0srv.cc b/storage/innobase/srv/srv0srv.cc index a8f35425d35d..98458b9f6eda 100644 --- a/storage/innobase/srv/srv0srv.cc +++ b/storage/innobase/srv/srv0srv.cc @@ -1209,6 +1209,12 @@ static void srv_init(void) { UT_NEW_THIS_FILE_PSI_KEY, ut::Count{srv_threads.m_page_cleaner_workers_n}); + /* One LRU manager thread per buf_pool instance. */ + srv_threads.m_lru_managers_n = srv_buf_pool_instances; + + srv_threads.m_lru_managers = ut::new_arr_withkey( + UT_NEW_THIS_FILE_PSI_KEY, ut::Count{srv_threads.m_lru_managers_n}); + srv_sys = static_cast( ut::zalloc_withkey(UT_NEW_THIS_FILE_PSI_KEY, srv_sys_sz)); @@ -1317,6 +1323,18 @@ void srv_free(void) { srv_threads.m_page_cleaner_workers = nullptr; } + if (srv_threads.m_lru_managers != nullptr) { + for (size_t i = 0; i < srv_threads.m_lru_managers_n; ++i) { + srv_threads.m_lru_managers[i] = {}; + } + /* Allocated with ut::new_arr_withkey(), so it must be released + with ut::delete_arr() (which runs the element destructors and uses the + matching array allocator), not ut::free(). This mirrors the + m_page_cleaner_workers / m_purge_workers teardown above. */ + ut::delete_arr(srv_threads.m_lru_managers); + srv_threads.m_lru_managers = nullptr; + } + if (srv_threads.m_purge_workers != nullptr) { for (size_t i = 0; i < srv_threads.m_purge_workers_n; ++i) { srv_threads.m_purge_workers[i] = {}; diff --git a/storage/innobase/srv/srv0start.cc b/storage/innobase/srv/srv0start.cc index 3dc5ada3a046..ab55cee0b4ee 100644 --- a/storage/innobase/srv/srv0start.cc +++ b/storage/innobase/srv/srv0start.cc @@ -2489,8 +2489,6 @@ void srv_pre_dd_shutdown() { /* Crash if some query threads are still alive. */ ut_a(srv_conc_get_active_threads() == 0); - ut_a(!srv_thread_is_active(srv_threads.m_recv_writer)); - /* Avoid fast shutdown, if redo logging is disabled. Otherwise, we won't be able to recover. */ if (mtr_t::s_logging.is_disabled() && srv_fast_shutdown == 2) { @@ -2699,7 +2697,9 @@ static void srv_shutdown_page_cleaners() { here to let it complete the flushing of the buffer pools before proceeding further. */ - for (uint32_t count = 0; buf_flush_page_cleaner_is_active(); ++count) { + for (uint32_t count = 0; buf_flush_page_cleaner_is_active() || + buf_flush_active_lru_managers() > 0; + ++count) { if (count >= SHUTDOWN_SLEEP_ROUNDS) { ib::info(ER_IB_MSG_1251); count = 0; @@ -2709,6 +2709,8 @@ static void srv_shutdown_page_cleaners() { std::chrono::microseconds(SHUTDOWN_SLEEP_TIME_US)); } + ut_ad(buf_flush_active_lru_managers() == 0); + ut_ad(buf_pool_pending_io_reads_count() == 0); ut_ad(buf_pool_pending_io_writes_count() == 0); } @@ -2842,7 +2844,6 @@ void srv_shutdown() { std::cref(srv_threads.m_purge_coordinator), std::cref(srv_threads.m_ts_alter_encrypt), std::cref(srv_threads.m_fts_optimize), - std::cref(srv_threads.m_recv_writer), std::cref(srv_threads.m_dict_stats)}; for (const auto &thread : threads_stopped_before_shutdown) { diff --git a/storage/innobase/sync/sync0debug.cc b/storage/innobase/sync/sync0debug.cc index 151ec5edca7e..81593c84bf38 100644 --- a/storage/innobase/sync/sync0debug.cc +++ b/storage/innobase/sync/sync0debug.cc @@ -473,7 +473,6 @@ LatchDebug::LatchDebug() { LEVEL_MAP_INSERT(SYNC_FTS_BG_THREADS); LEVEL_MAP_INSERT(SYNC_FTS_CACHE_INIT); LEVEL_MAP_INSERT(SYNC_RECV); - LEVEL_MAP_INSERT(SYNC_RECV_WRITER); LEVEL_MAP_INSERT(SYNC_LOG_ONLINE); LEVEL_MAP_INSERT(SYNC_LOG_SN); LEVEL_MAP_INSERT(SYNC_LOG_SN_MUTEX); @@ -724,7 +723,6 @@ Latches *LatchDebug::check_order(const latch_t *latch, case SYNC_LOCK_FREE_HASH: case SYNC_MONITOR_MUTEX: case SYNC_RECV: - case SYNC_RECV_WRITER: case SYNC_FTS_BG_THREADS: case SYNC_WORK_QUEUE: case SYNC_FTS_TOKENIZE: @@ -1330,8 +1328,6 @@ static void sync_latch_meta_init() UNIV_NOTHROW { LATCH_ADD_MUTEX(RECV_SYS, SYNC_RECV, recv_sys_mutex_key); - LATCH_ADD_MUTEX(RECV_WRITER, SYNC_RECV_WRITER, recv_writer_mutex_key); - LATCH_ADD_MUTEX(TEMP_SPACE_RSEG, SYNC_TEMP_SPACE_RSEG, temp_space_rseg_mutex_key); diff --git a/storage/innobase/sync/sync0sync.cc b/storage/innobase/sync/sync0sync.cc index 428731720270..b30e0560976e 100644 --- a/storage/innobase/sync/sync0sync.cc +++ b/storage/innobase/sync/sync0sync.cc @@ -102,7 +102,6 @@ mysql_pfs_key_t recalc_pool_mutex_key; mysql_pfs_key_t page_cleaner_mutex_key; mysql_pfs_key_t purge_sys_pq_mutex_key; mysql_pfs_key_t recv_sys_mutex_key; -mysql_pfs_key_t recv_writer_mutex_key; mysql_pfs_key_t temp_space_rseg_mutex_key; mysql_pfs_key_t undo_space_rseg_mutex_key; mysql_pfs_key_t page_zip_stat_per_index_mutex_key; From d34d6d0257caa06588bcce47ea9140e087400521 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Pawe=C5=82=20Olchawa?= Date: Tue, 21 Jul 2026 19:42:34 +0200 Subject: [PATCH 25/32] PS-11445 [trunk]: [2] LRU: Close the buf_pool_invalidate_instance race https://perconadev.atlassian.net/browse/PS-11445 The restored implementation pauses the LRU manager thread around buf_pool_invalidate_instance() purely by resetting the run_lru event. That's racy: resetting run_lru only makes the *next* os_event_wait() park, so a manager thread that had already returned from the wait (or was in its pre-batch sleep) when invalidation began could still start an LRU batch concurrently with the teardown. Close that window with a flushing_allowed flag: - buf_pool_invalidate_instance() clears it on teardown, - buf_flush_start() checks it when starting a new flush. Updates to the flag are protected by the flush_state_mutex. When starting a new flush, the flag is checked and the flush is started only if allowed and atomically with the check. Therefore after setting the flag to false, the only possible flushes are those that started before the flag was set, so it is then enough to wait for any pending flushes. --- storage/innobase/buf/buf0buf.cc | 27 ++++++++++++-------- storage/innobase/buf/buf0flu.cc | 40 ++++++++++++++++++------------ storage/innobase/include/buf0buf.h | 13 +++++++--- 3 files changed, 50 insertions(+), 30 deletions(-) diff --git a/storage/innobase/buf/buf0buf.cc b/storage/innobase/buf/buf0buf.cc index c6926c9ca279..8118e7faeefa 100644 --- a/storage/innobase/buf/buf0buf.cc +++ b/storage/innobase/buf/buf0buf.cc @@ -1525,6 +1525,8 @@ static void buf_pool_create(buf_pool_t *buf_pool, ulint buf_pool_size, buf_pool->run_lru = os_event_create(); os_event_set(buf_pool->run_lru); + buf_pool->flushing_allowed = true; + buf_pool->watch = (buf_page_t *)ut::zalloc_withkey( UT_NEW_THIS_FILE_PSI_KEY, sizeof(*buf_pool->watch) * BUF_POOL_WATCH_SIZE); for (i = 0; i < BUF_POOL_WATCH_SIZE; i++) { @@ -6481,21 +6483,26 @@ static void buf_refresh_io_stats(buf_pool_t *buf_pool) { /** Invalidates file pages in one buffer pool instance @param[in] buf_pool buffer pool instance */ static void buf_pool_invalidate_instance(buf_pool_t *buf_pool) { - ulint i; - ut_ad(!mutex_own(&buf_pool->LRU_list_mutex)); + /* Pause LRU threads on event. */ os_event_reset(buf_pool->run_lru); - auto guard = create_scope_guard([&]() { os_event_set(buf_pool->run_lru); }); - for (i = BUF_FLUSH_LRU; i < BUF_FLUSH_N_TYPES; i++) { - /* Although this function is called during startup and - during redo application phase during recovery, Percona InnoDB - might be running several LRU manager threads at this stage. - Hence, a new write batch can be in initialization stage at this point. */ + /* Prevent new flushes to start (buf_flush_start() checks this flag, + when starting a new flush). */ + mutex_enter(&buf_pool->flush_state_mutex); + buf_pool->flushing_allowed = false; + mutex_exit(&buf_pool->flush_state_mutex); + + auto guard = create_scope_guard([&]() { + mutex_enter(&buf_pool->flush_state_mutex); + buf_pool->flushing_allowed = true; + mutex_exit(&buf_pool->flush_state_mutex); + os_event_set(buf_pool->run_lru); + }); - /* For buffer pool invalidation to proceed we must ensure there is NO - write activity happening. */ + /* New flushing has been disallowed; wait for pending flushes. */ + for (size_t i = BUF_FLUSH_LRU; i < BUF_FLUSH_N_TYPES; i++) { buf_flush_await_no_flushing(buf_pool, static_cast(i)); } diff --git a/storage/innobase/buf/buf0flu.cc b/storage/innobase/buf/buf0flu.cc index b4e139165880..7a93d2b98c8c 100644 --- a/storage/innobase/buf/buf0flu.cc +++ b/storage/innobase/buf/buf0flu.cc @@ -1754,8 +1754,11 @@ static bool buf_flush_start(buf_pool_t *buf_pool, buf_flush_t flush_type) { bool started = false; buf_pool->change_flush_state(flush_type, [&]() { /* Can't start a new batch of the same type as one already running - - various synchronization mechanisms/counters would not work. */ - if (!buf_pool->is_flushing(flush_type)) { + various synchronization mechanisms/counters would not work. + + Also, don't start one while buf_pool_invalidate_instance() has closed + the gate (flushing_allowed == false) for the duration of a teardown. */ + if (!buf_pool->is_flushing(flush_type) && buf_pool->flushing_allowed) { buf_pool->init_flush[flush_type] = true; started = true; } @@ -3382,14 +3385,12 @@ void buf_flush_sync_all_buf_pools() { buf_flush_fsync(); } -/** Make a LRU manager thread sleep until the passed target time, if it's not -already in the past. -@param[in] next_loop_time desired wake up time */ +/** Sleep the LRU manager thread until next_loop_time, unless we are already +past it or shutdown is in the flush phase or later (in which case the manager +runs without sleeping so it can exit promptly). */ static void buf_lru_manager_sleep_if_needed( std::chrono::steady_clock::time_point next_loop_time) { - /* If this is the server shutdown buffer pool flushing phase, skip the - sleep to quit this thread faster */ - if (srv_shutdown_state.load() == SRV_SHUTDOWN_FLUSH_PHASE) return; + if (srv_shutdown_state.load() >= SRV_SHUTDOWN_FLUSH_PHASE) return; const auto cur_time = std::chrono::steady_clock::now(); @@ -3433,10 +3434,20 @@ static void buf_lru_manager_adapt_sleep_time( /* Free lists filled between 5% and 20%, no change */ } } -/** LRU manager thread for performing LRU flushed and evictions for buffer pool -free list refill. One thread is created for each buffer pool instace. -@param[in] buf_pool_instance buffer pool instance number for this thread -*/ + +/** LRU manager thread. One per buf_pool instance. Periodically calls +buf_flush_LRU_list to keep the free list topped up. The current +buf_LRU_get_free_block continues to do its own scan_and_free / +single_page_flush fall-back as needed; this thread is an additional source +of LRU-tail pressure that lets the user-thread fall-back stay rare on +healthy workloads. + +The thread runs until srv_shutdown_state reaches SRV_SHUTDOWN_FLUSH_PHASE, +mirroring the page cleaner coordinator's pre-flush loop, so that user +threads keep finding free pages throughout every earlier shutdown phase. +The buf_pool's run_lru event is set at startup; it is reset only inside +buf_pool_invalidate_instance() so the manager pauses while the pool is +torn down. */ static void buf_lru_manager_thread(size_t buf_pool_instance) { #ifdef UNIV_LINUX /* linux might be able to set different setting for each thread @@ -3455,10 +3466,7 @@ static void buf_lru_manager_thread(size_t buf_pool_instance) { auto next_loop_time = std::chrono::steady_clock::now() + lru_sleep_time; ulint lru_n_flushed = 1; - /* On server shutdown, the LRU manager thread runs through cleanup - phase to provide free pages for the master and purge threads. */ - while (srv_shutdown_state.load() == SRV_SHUTDOWN_NONE || - srv_shutdown_state.load() == SRV_SHUTDOWN_CLEANUP) { + while (srv_shutdown_state.load() < SRV_SHUTDOWN_FLUSH_PHASE) { ut_d(buf_flush_page_cleaner_disabled_loop()); os_event_wait(buf_pool->run_lru); diff --git a/storage/innobase/include/buf0buf.h b/storage/innobase/include/buf0buf.h index 9a05cc732517..7228f72cc7a0 100644 --- a/storage/innobase/include/buf0buf.h +++ b/storage/innobase/include/buf0buf.h @@ -2492,12 +2492,17 @@ struct buf_pool_t { running. Protected by flush_state_mutex. */ os_event_t no_flush[BUF_FLUSH_N_TYPES]; - /* This event is always set at startup, so LRU threads do not wait for this - event. Before invalidating bufferpool, this event is reset, so the next LRU - batch flushing will wait for the event. Bufferpool invalidation needs LRU - flushing to be stopped. */ + /** Always set at startup so the LRU manager thread does not have to wait. + Reset by buf_pool_invalidate_instance() so the manager pauses while the + buffer pool is being torn down / re-initialised; set again afterwards. */ os_event_t run_lru; + /** Run gate for flushes, checked by buf_flush_start() inside + the change_flush_state() critical section that sets init_flush[type]. + True in normal operation; set to false by buf_pool_invalidate_instance() + for the duration of a teardown. Protected by flush_state_mutex. */ + bool flushing_allowed; + /** A sequence number used to count the number of buffer blocks removed from the end of the LRU list; NOTE that this counter may wrap around at 4 billion! A thread is allowed to read this for heuristic purposes without From 91ecbd393fa9719930d1169bab272080d0ce3111 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Pawe=C5=82=20Olchawa?= Date: Thu, 23 Jul 2026 12:16:38 +0200 Subject: [PATCH 26/32] PS-11445 [trunk]: [3] LRU: Add innodb_lru_threads sysvar https://perconadev.atlassian.net/browse/PS-11445 1. Introduce a knob that allows to enable LRU manager threads. 2. Aggregate multi-instance flush/evict/scan stats for LRU flushing for both LRU threads and page cleaners. With a per-buffer-pool-instance LRU manager thread, several instances can call into the LRU batch path concurrently. We want to avoid the need to synchronize on stats update between them. Therefore each of them returns the statistics and the aggregation happens in a single thread - the buf_flush_page_coordinator thread. --- .../r/innodb_lru_threads_kill_recovery.result | 23 + .../t/innodb_lru_threads_kill_recovery.test | 48 ++ .../r/innodb_lru_threads_basic.result | 27 ++ .../sys_vars/t/innodb_lru_threads_basic.test | 29 ++ storage/innobase/buf/buf0flu.cc | 457 +++++++++++++----- storage/innobase/handler/ha_innodb.cc | 12 + storage/innobase/include/buf0buf.h | 32 ++ storage/innobase/include/buf0flu.h | 24 +- storage/innobase/include/log0recv.h | 8 + storage/innobase/include/srv0srv.h | 12 +- storage/innobase/include/sync0sync.h | 1 + storage/innobase/include/sync0types.h | 6 +- storage/innobase/log/log0recv.cc | 123 ++++- storage/innobase/srv/srv0mon.cc | 8 +- storage/innobase/srv/srv0srv.cc | 2 + storage/innobase/srv/srv0start.cc | 3 + storage/innobase/sync/sync0debug.cc | 4 + storage/innobase/sync/sync0sync.cc | 1 + 18 files changed, 675 insertions(+), 145 deletions(-) create mode 100644 mysql-test/suite/innodb/r/innodb_lru_threads_kill_recovery.result create mode 100644 mysql-test/suite/innodb/t/innodb_lru_threads_kill_recovery.test create mode 100644 mysql-test/suite/sys_vars/r/innodb_lru_threads_basic.result create mode 100644 mysql-test/suite/sys_vars/t/innodb_lru_threads_basic.test diff --git a/mysql-test/suite/innodb/r/innodb_lru_threads_kill_recovery.result b/mysql-test/suite/innodb/r/innodb_lru_threads_kill_recovery.result new file mode 100644 index 000000000000..23a90998a4c3 --- /dev/null +++ b/mysql-test/suite/innodb/r/innodb_lru_threads_kill_recovery.result @@ -0,0 +1,23 @@ +# +# PS-11445: with --innodb-lru-threads=ON, the LRU manager threads +# are created before buf_pool_invalidate() runs on every startup, so +# they are alive throughout crash recovery (including the +# buf_pool_invalidate_instance() pause/resume of run_lru). Restart +# after a crash with a non-trivial amount of redo to apply exercises +# that path; a normal restart afterward exercises the page-cleaner +# coordinator waiting for the LRU managers to stop on shutdown. +# +# restart:--innodb-lru-threads=ON --innodb-buffer-pool-size=8M --innodb-buffer-pool-instances=2 --innodb-lru-scan-depth=100 +CREATE TABLE t1 (a INT PRIMARY KEY, b VARCHAR(512)) ENGINE=InnoDB; +# Kill and restart:--innodb-lru-threads=ON --innodb-buffer-pool-size=8M --innodb-buffer-pool-instances=2 --innodb-lru-scan-depth=100 +SELECT COUNT(*) FROM t1; +COUNT(*) +20000 +# Clean shutdown must not hang: the page-cleaner coordinator has to +# observe every LRU manager thread stop. +# restart:--innodb-lru-threads=ON --innodb-buffer-pool-size=8M --innodb-buffer-pool-instances=2 --innodb-lru-scan-depth=100 +SELECT COUNT(*) FROM t1; +COUNT(*) +20000 +DROP TABLE t1; +# restart diff --git a/mysql-test/suite/innodb/t/innodb_lru_threads_kill_recovery.test b/mysql-test/suite/innodb/t/innodb_lru_threads_kill_recovery.test new file mode 100644 index 000000000000..b2e0a586f7b0 --- /dev/null +++ b/mysql-test/suite/innodb/t/innodb_lru_threads_kill_recovery.test @@ -0,0 +1,48 @@ +--source include/no_valgrind_without_big.inc + +--echo # +--echo # PS-11445: with --innodb-lru-threads=ON, the LRU manager threads +--echo # are created before buf_pool_invalidate() runs on every startup, so +--echo # they are alive throughout crash recovery (including the +--echo # buf_pool_invalidate_instance() pause/resume of run_lru). Restart +--echo # after a crash with a non-trivial amount of redo to apply exercises +--echo # that path; a normal restart afterward exercises the page-cleaner +--echo # coordinator waiting for the LRU managers to stop on shutdown. +--echo # + +--let $restart_parameters=restart:--innodb-lru-threads=ON --innodb-buffer-pool-size=8M --innodb-buffer-pool-instances=2 --innodb-lru-scan-depth=100 +--source include/restart_mysqld.inc + +CREATE TABLE t1 (a INT PRIMARY KEY, b VARCHAR(512)) ENGINE=InnoDB; + +--disable_query_log +DELIMITER |; +CREATE PROCEDURE fill_t1() +BEGIN + DECLARE i INT DEFAULT 0; + WHILE i < 20000 DO + INSERT INTO t1 VALUES (i, REPEAT('x', 512)); + SET i = i + 1; + END WHILE; +END| +DELIMITER ;| +CALL fill_t1(); +DROP PROCEDURE fill_t1; +--enable_query_log + +--let $wait_counter= 3000 +--source include/kill_and_restart_mysqld.inc + +SELECT COUNT(*) FROM t1; + +--echo # Clean shutdown must not hang: the page-cleaner coordinator has to +--echo # observe every LRU manager thread stop. +--let $restart_parameters=restart:--innodb-lru-threads=ON --innodb-buffer-pool-size=8M --innodb-buffer-pool-instances=2 --innodb-lru-scan-depth=100 +--source include/restart_mysqld.inc + +SELECT COUNT(*) FROM t1; + +DROP TABLE t1; + +--let $restart_parameters= +--source include/restart_mysqld.inc diff --git a/mysql-test/suite/sys_vars/r/innodb_lru_threads_basic.result b/mysql-test/suite/sys_vars/r/innodb_lru_threads_basic.result new file mode 100644 index 000000000000..f2a17f69b789 --- /dev/null +++ b/mysql-test/suite/sys_vars/r/innodb_lru_threads_basic.result @@ -0,0 +1,27 @@ +# +# Basic test for innodb_lru_threads (read-only) +# +SELECT @@GLOBAL.innodb_lru_threads; +@@GLOBAL.innodb_lru_threads +0 +SET GLOBAL innodb_lru_threads = OFF; +ERROR HY000: Variable 'innodb_lru_threads' is a read only variable +SET GLOBAL innodb_lru_threads = ON; +ERROR HY000: Variable 'innodb_lru_threads' is a read only variable +SELECT @@GLOBAL.innodb_lru_threads; +@@GLOBAL.innodb_lru_threads +0 +# +# Startup with innodb_lru_threads=ON +# +# restart:--innodb-lru-threads=ON +SELECT @@GLOBAL.innodb_lru_threads; +@@GLOBAL.innodb_lru_threads +1 +# +# Startup with innodb_lru_threads=OFF +# +# restart +SELECT @@GLOBAL.innodb_lru_threads; +@@GLOBAL.innodb_lru_threads +0 diff --git a/mysql-test/suite/sys_vars/t/innodb_lru_threads_basic.test b/mysql-test/suite/sys_vars/t/innodb_lru_threads_basic.test new file mode 100644 index 000000000000..c2de5a7f5c48 --- /dev/null +++ b/mysql-test/suite/sys_vars/t/innodb_lru_threads_basic.test @@ -0,0 +1,29 @@ +--echo # +--echo # Basic test for innodb_lru_threads (read-only) +--echo # + +SELECT @@GLOBAL.innodb_lru_threads; + +--error ER_INCORRECT_GLOBAL_LOCAL_VAR +SET GLOBAL innodb_lru_threads = OFF; + +--error ER_INCORRECT_GLOBAL_LOCAL_VAR +SET GLOBAL innodb_lru_threads = ON; + +SELECT @@GLOBAL.innodb_lru_threads; + +--echo # +--echo # Startup with innodb_lru_threads=ON +--echo # + +--let $restart_parameters=restart:--innodb-lru-threads=ON +--source include/restart_mysqld.inc +SELECT @@GLOBAL.innodb_lru_threads; + +--echo # +--echo # Startup with innodb_lru_threads=OFF +--echo # + +--let $restart_parameters=restart +--source include/restart_mysqld.inc +SELECT @@GLOBAL.innodb_lru_threads; diff --git a/storage/innobase/buf/buf0flu.cc b/storage/innobase/buf/buf0flu.cc index 7a93d2b98c8c..149bd7299b94 100644 --- a/storage/innobase/buf/buf0flu.cc +++ b/storage/innobase/buf/buf0flu.cc @@ -56,6 +56,7 @@ this program; if not, write to the Free Software Foundation, Inc., #include "ibuf0ibuf.h" #include "log0buf.h" #include "log0chkp.h" +#include "log0recv.h" #include "log0write.h" #include "my_compiler.h" #include "os0file.h" @@ -126,8 +127,8 @@ struct page_cleaner_slot_t { protected by page_cleaner_t::mutex if the worker thread got the slot and set to PAGE_CLEANER_STATE_FLUSHING, - n_flushed_list can be updated only by - the worker thread */ + lru_result and n_flushed_list can be + updated only by the worker thread */ /* This value is set during state==PAGE_CLEANER_STATE_NONE */ ulint n_pages_requested; /*!< number of requested pages @@ -135,15 +136,22 @@ struct page_cleaner_slot_t { /* These values are updated during state==PAGE_CLEANER_STATE_FLUSHING, and committed with state==PAGE_CLEANER_STATE_FINISHED. The consistency is protected by the 'state' */ + buf_flush_batch_result_t lru_result; + /*!< LRU batch result when dedicated LRU manager + threads are disabled */ ulint n_flushed_list; /*!< number of flushed pages by flush_list flushing */ bool succeeded_list; /*!< true if flush_list flushing succeeded. */ + std::chrono::milliseconds flush_lru_time; + /*!< elapsed time for LRU flushing */ std::chrono::milliseconds flush_list_time; /*!< elapsed time for flush_list flushing */ + ulint flush_lru_pass; + /*!< count to attempt LRU flushing */ ulint flush_list_pass; /*!< count to attempt flush_list flushing */ @@ -1442,9 +1450,9 @@ after a call to this function there will be 'max' blocks in the free list. The caller must hold the LRU list mutex. @param[in] buf_pool buffer pool instance @param[in] max desired number of blocks in the free_list -@return number of blocks moved to the free list. */ -static ulint buf_free_from_unzip_LRU_list_batch(buf_pool_t *buf_pool, - ulint max) { +@return batch result. This path never flushes, so n_flushed is always 0. */ +static buf_flush_batch_result_t buf_free_from_unzip_LRU_list_batch( + buf_pool_t *buf_pool, ulint max) { ulint scanned = 0; ulint count = 0; ulint free_len = UT_LIST_GET_LEN(buf_pool->free); @@ -1479,19 +1487,7 @@ static ulint buf_free_from_unzip_LRU_list_batch(buf_pool_t *buf_pool, ut_ad(mutex_own(&buf_pool->LRU_list_mutex)); - if (count) { - MONITOR_INC_VALUE_CUMULATIVE(MONITOR_LRU_BATCH_EVICT_TOTAL_PAGE, - MONITOR_LRU_BATCH_EVICT_COUNT, - MONITOR_LRU_BATCH_EVICT_PAGES, count); - } - - if (scanned) { - MONITOR_INC_VALUE_CUMULATIVE(MONITOR_LRU_BATCH_SCANNED, - MONITOR_LRU_BATCH_SCANNED_NUM_CALL, - MONITOR_LRU_BATCH_SCANNED_PER_CALL, scanned); - } - - return (count); + return {0, count, scanned}; } /** This utility flushes dirty blocks from the end of the LRU list. @@ -1501,11 +1497,9 @@ it is a best effort attempt and it is not guaranteed that after a call to this function there will be 'max' blocks in the free list. @param[in] buf_pool buffer pool instance @param[in] max desired number for blocks in the free_list -@return pair of numbers where first number is the blocks for which -flush request is queued and second is the number of blocks that were -clean and simply evicted from the LRU. */ -static std::pair buf_flush_LRU_list_batch(buf_pool_t *buf_pool, - ulint max) { +@return batch result */ +static buf_flush_batch_result_t buf_flush_LRU_list_batch(buf_pool_t *buf_pool, + ulint max) { buf_page_t *bpage; ulint scanned = 0; ulint evict_count = 0; @@ -1571,48 +1565,36 @@ static std::pair buf_flush_LRU_list_batch(buf_pool_t *buf_pool, ut_ad(mutex_own(&buf_pool->LRU_list_mutex)); - if (evict_count) { - MONITOR_INC_VALUE_CUMULATIVE(MONITOR_LRU_BATCH_EVICT_TOTAL_PAGE, - MONITOR_LRU_BATCH_EVICT_COUNT, - MONITOR_LRU_BATCH_EVICT_PAGES, evict_count); - } - - if (scanned) { - MONITOR_INC_VALUE_CUMULATIVE(MONITOR_LRU_BATCH_SCANNED, - MONITOR_LRU_BATCH_SCANNED_NUM_CALL, - MONITOR_LRU_BATCH_SCANNED_PER_CALL, scanned); - } - - return (std::make_pair(count, evict_count)); + return {count, evict_count, scanned}; } /** Flush and move pages from LRU or unzip_LRU list to the free list. Whether LRU or unzip_LRU is used depends on the state of the system. @param[in] buf_pool buffer pool instance @param[in] max desired number of blocks in the free_list -@return number of blocks for which either the write request was queued -or in case of unzip_LRU the number of blocks actually moved to the -free list */ -static std::pair buf_do_LRU_batch(buf_pool_t *buf_pool, - ulint max) { - ulint count = 0; - std::pair res; +@return batch result */ +static buf_flush_batch_result_t buf_do_LRU_batch(buf_pool_t *buf_pool, + ulint max) { + buf_flush_batch_result_t result{}; ut_ad(mutex_own(&buf_pool->LRU_list_mutex)); if (buf_LRU_evict_from_unzip_LRU(buf_pool)) { - count = buf_free_from_unzip_LRU_list_batch(buf_pool, max); + const auto unzip_result = buf_free_from_unzip_LRU_list_batch(buf_pool, max); + result.n_flushed += unzip_result.n_flushed; + result.n_evicted += unzip_result.n_evicted; + result.n_scanned += unzip_result.n_scanned; } - if (max > count) { - res = buf_flush_LRU_list_batch(buf_pool, max - count); + const ulint done = result.n_flushed + result.n_evicted; + if (max > done) { + const auto lru_result = buf_flush_LRU_list_batch(buf_pool, max - done); + result.n_flushed += lru_result.n_flushed; + result.n_evicted += lru_result.n_evicted; + result.n_scanned += lru_result.n_scanned; } - /* Add evicted pages from unzip_LRU to the evicted pages from the simple - LRU. */ - res.second += count; - - return (res); + return result; } /** This utility flushes dirty blocks from the end of the flush_list. @@ -1694,10 +1676,10 @@ not guaranteed that the actual number is that big, though) @param[in] lsn_limit in the case of BUF_FLUSH_LIST all blocks whose oldest_modification is smaller than this should be flushed (if their number does not exceed min_n), otherwise ignored -@return pair of numbers of flushed and evicted blocks */ -static std::pair buf_flush_batch(buf_pool_t *buf_pool, - buf_flush_t flush_type, - ulint min_n, lsn_t lsn_limit) { +@return batch result. For BUF_FLUSH_LIST n_evicted and n_scanned are 0. */ +static buf_flush_batch_result_t buf_flush_batch(buf_pool_t *buf_pool, + buf_flush_t flush_type, + ulint min_n, lsn_t lsn_limit) { ut_ad(flush_type == BUF_FLUSH_LRU || flush_type == BUF_FLUSH_LIST); #ifdef UNIV_DEBUG @@ -1708,29 +1690,29 @@ static std::pair buf_flush_batch(buf_pool_t *buf_pool, } #endif /* UNIV_DEBUG */ - std::pair res; + buf_flush_batch_result_t result{}; /* Note: The buffer pool mutexes is released and reacquired within the flush functions. */ switch (flush_type) { case BUF_FLUSH_LRU: mutex_enter(&buf_pool->LRU_list_mutex); - res = buf_do_LRU_batch(buf_pool, min_n); + result = buf_do_LRU_batch(buf_pool, min_n); mutex_exit(&buf_pool->LRU_list_mutex); break; case BUF_FLUSH_LIST: - res.first = buf_do_flush_list_batch(buf_pool, min_n, lsn_limit); - res.second = 0; + /* The flush list path only flushes; nothing is evicted here. */ + result.n_flushed = buf_do_flush_list_batch(buf_pool, min_n, lsn_limit); break; default: ut_error; } - DBUG_PRINT("ib_buf", - ("flush %u completed, flushed %u pages, evicted %u pages", - unsigned(flush_type), unsigned(res.first), unsigned(res.second))); + DBUG_PRINT("ib_buf", ("flush %u completed, %u flushed, %u evicted", + unsigned(flush_type), unsigned(result.n_flushed), + unsigned(result.n_evicted))); - return (res); + return result; } /** Gather the aggregated stats for both flush list and LRU list flushing. @@ -1751,6 +1733,7 @@ static void buf_flush_stats(ulint page_count_flush, ulint page_count_LRU) { @param[in] flush_type BUF_FLUSH_LRU or BUF_FLUSH_LIST */ static bool buf_flush_start(buf_pool_t *buf_pool, buf_flush_t flush_type) { ut_ad(flush_type == BUF_FLUSH_LRU || flush_type == BUF_FLUSH_LIST); + bool started = false; buf_pool->change_flush_state(flush_type, [&]() { /* Can't start a new batch of the same type as one already running - @@ -1804,24 +1787,27 @@ void buf_flush_await_no_flushing(buf_pool_t *buf_pool, buf_flush_t flush_type) { } } -bool buf_flush_do_batch(buf_pool_t *buf_pool, buf_flush_t type, ulint min_n, - lsn_t lsn_limit, ulint *n_processed) { + +bool buf_flush_do_batch(buf_pool_t *buf_pool, buf_flush_t type, + ulint min_n, lsn_t lsn_limit, + buf_flush_batch_result_t *result) { ut_ad(type == BUF_FLUSH_LRU || type == BUF_FLUSH_LIST); - if (n_processed != nullptr) { - *n_processed = 0; + if (result != nullptr) { + *result = {}; } if (!buf_flush_start(buf_pool, type)) { return (false); } - const auto res = buf_flush_batch(buf_pool, type, min_n, lsn_limit); + const buf_flush_batch_result_t batch_result = + buf_flush_batch(buf_pool, type, min_n, lsn_limit); - buf_flush_end(buf_pool, type, res.first); + buf_flush_end(buf_pool, type, batch_result.n_flushed); - if (n_processed != nullptr) { - *n_processed = res.first + res.second; + if (result != nullptr) { + *result = batch_result; } return (true); @@ -1846,12 +1832,12 @@ bool buf_flush_lists(ulint min_n, lsn_t lsn_limit, ulint *n_processed) { /* Flush to lsn_limit in all buffer pool instances */ for (ulint i = 0; i < srv_buf_pool_instances; i++) { buf_pool_t *buf_pool; - ulint page_count = 0; + buf_flush_batch_result_t result{}; buf_pool = buf_pool_from_array(i); if (!buf_flush_do_batch(buf_pool, BUF_FLUSH_LIST, min_n, lsn_limit, - &page_count)) { + &result)) { /* We have two choices here. If lsn_limit was specified then skipping an instance of buffer pool means we cannot guarantee that all pages @@ -1867,7 +1853,11 @@ bool buf_flush_lists(ulint min_n, lsn_t lsn_limit, ulint *n_processed) { continue; } - n_flushed += page_count; + /* BUF_FLUSH_LIST never evicts: buf_flush_batch() reaches the LRU + eviction code only for BUF_FLUSH_LRU. */ + ut_ad(result.n_evicted == 0); + ut_ad(result.n_scanned == 0); + n_flushed += result.n_flushed; } if (n_flushed) { @@ -1973,10 +1963,12 @@ Clears up tail of the LRU list of a given buffer pool instance: The depth to which we scan each buffer pool is controlled by dynamic config parameter innodb_LRU_scan_depth. @param buf_pool buffer pool instance -@return total pages flushed and evicted */ -static ulint buf_flush_LRU_list(buf_pool_t *buf_pool) { +@param[out] started whether an LRU batch was admitted to start +@return batch result */ +static buf_flush_batch_result_t buf_flush_LRU_list(buf_pool_t *buf_pool, + bool *started) { ulint scan_depth, withdraw_depth; - ulint n_flushed = 0; + buf_flush_batch_result_t result{}; ut_ad(buf_pool); @@ -1991,13 +1983,108 @@ static ulint buf_flush_LRU_list(buf_pool_t *buf_pool) { scan_depth = std::min(static_cast(srv_LRU_scan_depth), scan_depth); } - /* Currently one of page_cleaners is the only thread - that can trigger an LRU flush at the same time. - So, it is not possible that a batch triggered during - last iteration is still running, */ - buf_flush_do_batch(buf_pool, BUF_FLUSH_LRU, scan_depth, 0, &n_flushed); + *started = + buf_flush_do_batch(buf_pool, BUF_FLUSH_LRU, scan_depth, 0, &result); + + return result; +} + +/** Record one LRU pass in the common per-instance statistics accumulator. */ +static void buf_lru_flush_stat_record( + buf_pool_t *buf_pool, const buf_flush_batch_result_t &result, + std::chrono::steady_clock::duration elapsed) { + mutex_enter(&buf_pool->flush_state_mutex); + auto &stat = buf_pool->lru_flush_stat; + + stat.n_flushed_pages += result.n_flushed; + if (result.n_flushed > 0) { + ++stat.n_flush_batches; + stat.max_flushed_pages_per_batch = + std::max(stat.max_flushed_pages_per_batch, + static_cast(result.n_flushed)); + } - return (n_flushed); + stat.n_evicted_pages += result.n_evicted; + if (result.n_evicted > 0) { + ++stat.n_evict_batches; + stat.max_evicted_pages_per_batch = + std::max(stat.max_evicted_pages_per_batch, + static_cast(result.n_evicted)); + } + + stat.n_scanned_pages += result.n_scanned; + if (result.n_scanned > 0) { + ++stat.n_scan_batches; + stat.max_scanned_pages_per_batch = + std::max(stat.max_scanned_pages_per_batch, + static_cast(result.n_scanned)); + } + + const auto elapsed_ms = + std::chrono::duration_cast(elapsed).count(); + ++stat.n_lru_passes; + stat.lru_flush_time_ms += elapsed_ms; + mutex_exit(&buf_pool->flush_state_mutex); +} + +/** Drains per-instance LRU batch counters and updates the accumulator. */ +static void pc_publish_lru_batch_stats() { + buf_pool_t::lru_flush_stat_t lru_stat{}; + for (ulint i = 0; i < srv_buf_pool_instances; i++) { + buf_pool_t *const buf_pool = buf_pool_from_array(i); + mutex_enter(&buf_pool->flush_state_mutex); + auto &instance_stat = buf_pool->lru_flush_stat; + lru_stat.n_flushed_pages += instance_stat.n_flushed_pages; + lru_stat.n_flush_batches += instance_stat.n_flush_batches; + lru_stat.max_flushed_pages_per_batch = + std::max(lru_stat.max_flushed_pages_per_batch, + instance_stat.max_flushed_pages_per_batch); + lru_stat.n_evicted_pages += instance_stat.n_evicted_pages; + lru_stat.n_evict_batches += instance_stat.n_evict_batches; + lru_stat.max_evicted_pages_per_batch = + std::max(lru_stat.max_evicted_pages_per_batch, + instance_stat.max_evicted_pages_per_batch); + lru_stat.n_scanned_pages += instance_stat.n_scanned_pages; + lru_stat.n_scan_batches += instance_stat.n_scan_batches; + lru_stat.max_scanned_pages_per_batch = + std::max(lru_stat.max_scanned_pages_per_batch, + instance_stat.max_scanned_pages_per_batch); + instance_stat.n_flushed_pages = 0; + instance_stat.n_flush_batches = 0; + instance_stat.max_flushed_pages_per_batch = 0; + instance_stat.n_evicted_pages = 0; + instance_stat.n_evict_batches = 0; + instance_stat.max_evicted_pages_per_batch = 0; + instance_stat.n_scanned_pages = 0; + instance_stat.n_scan_batches = 0; + instance_stat.max_scanned_pages_per_batch = 0; + mutex_exit(&buf_pool->flush_state_mutex); + } + + const auto publish_lru_batch_stat = + [](monitor_id_t total_monitor, monitor_id_t count_monitor, + monitor_id_t per_call_monitor, uint64_t n_pages, uint64_t n_batches, + uint64_t max_pages_per_batch) { + if (n_batches > 0 && MONITOR_IS_ON(total_monitor)) { + monitor_inc_value_nocheck(total_monitor, n_pages); + monitor_inc_value_nocheck(count_monitor, n_batches, false); + monitor_set(per_call_monitor, max_pages_per_batch, true, false); + monitor_set(per_call_monitor, n_pages / n_batches, false, false); + } + }; + + publish_lru_batch_stat( + MONITOR_LRU_BATCH_FLUSH_TOTAL_PAGE, MONITOR_LRU_BATCH_FLUSH_COUNT, + MONITOR_LRU_BATCH_FLUSH_PAGES, lru_stat.n_flushed_pages, + lru_stat.n_flush_batches, lru_stat.max_flushed_pages_per_batch); + publish_lru_batch_stat( + MONITOR_LRU_BATCH_EVICT_TOTAL_PAGE, MONITOR_LRU_BATCH_EVICT_COUNT, + MONITOR_LRU_BATCH_EVICT_PAGES, lru_stat.n_evicted_pages, + lru_stat.n_evict_batches, lru_stat.max_evicted_pages_per_batch); + publish_lru_batch_stat( + MONITOR_LRU_BATCH_SCANNED, MONITOR_LRU_BATCH_SCANNED_NUM_CALL, + MONITOR_LRU_BATCH_SCANNED_PER_CALL, lru_stat.n_scanned_pages, + lru_stat.n_scan_batches, lru_stat.max_scanned_pages_per_batch); } namespace Adaptive_flush { @@ -2128,15 +2215,32 @@ void set_average() { slot = &page_cleaner->slots[i]; + lru_tm += slot->flush_lru_time.count(); + lru_pass += slot->flush_lru_pass; list_tm += slot->flush_list_time.count(); list_pass += slot->flush_list_pass; + slot->flush_lru_time = std::chrono::seconds::zero(); + slot->flush_lru_pass = 0; slot->flush_list_time = std::chrono::seconds::zero(); slot->flush_list_pass = 0; } mutex_exit(&page_cleaner->mutex); + /* Dedicated LRU managers do not use page-cleaner slots. Consume only their + timing fields here; batch counters are published independently each second. */ + for (ulint i = 0; i < srv_buf_pool_instances; i++) { + buf_pool_t *const buf_pool = buf_pool_from_array(i); + mutex_enter(&buf_pool->flush_state_mutex); + auto &instance_stat = buf_pool->lru_flush_stat; + lru_tm += instance_stat.lru_flush_time_ms; + lru_pass += static_cast(instance_stat.n_lru_passes); + instance_stat.n_lru_passes = 0; + instance_stat.lru_flush_time_ms = 0; + mutex_exit(&buf_pool->flush_state_mutex); + } + /* minimum values are 1, to avoid dividing by zero. */ if (lru_tm < 1) { lru_tm = 1; @@ -2585,14 +2689,16 @@ void buf_flush_page_cleaner_init() { /* Make sure page cleaner is active. */ ut_a(buf_flush_page_cleaner_is_active()); - /* One LRU manager thread per buf_pool instance. */ - for (size_t i = 0; i < srv_threads.m_lru_managers_n; ++i) { - srv_threads.m_lru_managers[i] = os_thread_create( - buf_lru_manager_thread_key, i, buf_lru_manager_thread, i); - srv_threads.m_lru_managers[i].start(); - } + /* One LRU manager thread per buf_pool instance when enabled. */ + if (srv_lru_threads_enabled) { + for (size_t i = 0; i < srv_threads.m_lru_managers_n; ++i) { + srv_threads.m_lru_managers[i] = os_thread_create( + buf_lru_manager_thread_key, i, buf_lru_manager_thread, i); + srv_threads.m_lru_managers[i].start(); + } - ut_a(buf_flush_active_lru_managers() == srv_buf_pool_instances); + ut_a(buf_flush_active_lru_managers() == srv_buf_pool_instances); + } } /** @@ -2609,8 +2715,10 @@ static void buf_flush_page_cleaner_close(void) { srv_shutdown_state > SRV_SHUTDOWN_CLEANUP and break out of their loops; the run_lru event is always set during normal shutdown so they do not block. */ - for (size_t i = 0; i < srv_threads.m_lru_managers_n; ++i) { - srv_threads.m_lru_managers[i].wait(); + if (srv_lru_threads_enabled) { + for (size_t i = 0; i < srv_threads.m_lru_managers_n; ++i) { + srv_threads.m_lru_managers[i].wait(); + } } mutex_destroy(&page_cleaner->mutex); @@ -2677,7 +2785,9 @@ static void pc_request(ulint min_n, lsn_t lsn_limit) { Do flush for one slot. @return the number of the slots which has not been treated yet. */ static ulint pc_flush_slot(void) { + std::chrono::steady_clock::duration lru_time{}; std::chrono::steady_clock::duration flush_list_time{}; + int lru_pass = 0; int list_pass = 0; mutex_enter(&page_cleaner->mutex); @@ -2709,34 +2819,69 @@ static ulint pc_flush_slot(void) { } if (!page_cleaner->is_running) { + slot->lru_result = {}; slot->n_flushed_list = 0; } else { + const auto n_pages_requested = slot->n_pages_requested; + const auto requested = page_cleaner->requested; + const auto lsn_limit = page_cleaner->lsn_limit; + mutex_exit(&page_cleaner->mutex); + /* Flush pages from LRU tail if required. */ + buf_flush_batch_result_t lru_result; + if (!srv_lru_threads_enabled) { + const auto lru_start = std::chrono::steady_clock::now(); + bool started = false; + lru_result = buf_flush_LRU_list(buf_pool, &started); + lru_pass = started ? 1 : 0; + lru_time = std::chrono::steady_clock::now() - lru_start; + buf_lru_flush_stat_record(buf_pool, lru_result, lru_time); + } else { + lru_result = {}; + } + /* Flush pages from flush_list if required. LRU-tail flushing is the - responsibility of the per-pool buf_lru_manager_thread (one per - buf_pool instance); the page cleaner only flushes the flush_list. */ - if (page_cleaner->requested) { + responsibility of the per-pool buf_lru_manager_thread when + innodb_lru_threads is on. When it is off, the page cleaner performs + LRU flushing (including during recovery via recv_writer). */ + ulint n_flushed_list; + bool succeeded_list; + if (requested) { const auto flush_list_start = std::chrono::steady_clock::now(); - slot->succeeded_list = buf_flush_do_batch( - buf_pool, BUF_FLUSH_LIST, slot->n_pages_requested, - page_cleaner->lsn_limit, &slot->n_flushed_list); + buf_flush_batch_result_t result{}; + succeeded_list = buf_flush_do_batch( + buf_pool, BUF_FLUSH_LIST, n_pages_requested, + lsn_limit, &result); + /* BUF_FLUSH_LIST never evicts and does not report its scan count + through this result yet. */ + ut_ad(result.n_evicted == 0); + ut_ad(result.n_scanned == 0); + n_flushed_list = result.n_flushed; flush_list_time = std::chrono::steady_clock::now() - flush_list_start; list_pass = 1; } else { - slot->n_flushed_list = 0; - slot->succeeded_list = true; + n_flushed_list = 0; + succeeded_list = true; } + mutex_enter(&page_cleaner->mutex); + slot->lru_result = lru_result; + slot->succeeded_list = succeeded_list; + slot->n_flushed_list = n_flushed_list; } + page_cleaner->n_slots_flushing--; page_cleaner->n_slots_finished++; slot->state = PAGE_CLEANER_STATE_FINISHED; + slot->flush_lru_time += + std::chrono::duration_cast(lru_time); slot->flush_list_time += std::chrono::duration_cast(flush_list_time); + slot->flush_lru_pass += lru_pass; slot->flush_list_pass += list_pass; if (page_cleaner->n_slots_requested == 0 && @@ -2754,12 +2899,15 @@ static ulint pc_flush_slot(void) { /** Wait until all flush requests are finished. +@param lru_result aggregate LRU result from all slots @param n_flushed_list number of pages flushed from the end of the flush_list. @return true if all flush_list flushing batch were success. */ -static bool pc_wait_finished(ulint *n_flushed_list) { +static bool pc_wait_finished(buf_flush_batch_result_t *lru_result, + ulint *n_flushed_list) { bool all_succeeded = true; + *lru_result = {}; *n_flushed_list = 0; os_event_wait(page_cleaner->is_finished); @@ -2775,6 +2923,9 @@ static bool pc_wait_finished(ulint *n_flushed_list) { ut_ad(slot->state == PAGE_CLEANER_STATE_FINISHED); + lru_result->n_flushed += slot->lru_result.n_flushed; + lru_result->n_evicted += slot->lru_result.n_evicted; + lru_result->n_scanned += slot->lru_result.n_scanned; *n_flushed_list += slot->n_flushed_list; all_succeeded &= slot->succeeded_list; @@ -2889,8 +3040,16 @@ void buf_flush_page_cleaner_disabled_debug_update(THD *, SYS_VAR *, void *, mutex_enter(&page_cleaner->mutex); + /* With LRU manager threads enabled, convergence also waits for each of + them to reach buf_flush_page_cleaner_disabled_loop(). An LRU manager + parked on os_event_wait(buf_pool->run_lru) (a buf_pool_invalidate_instance() + window) does not reach that loop until run_lru is set again, so this + wait can stall for as long as that window lasts. Both this and the + manager's own adaptive sleep (up to 1s) are transient in practice, but + worth knowing if a debug test using this variable ever hangs here. */ const auto all_flushing_threads = - srv_n_page_cleaners + srv_buf_pool_instances; + srv_n_page_cleaners + + (srv_lru_threads_enabled ? srv_buf_pool_instances : 0); ut_ad(page_cleaner->n_disabled_debug <= all_flushing_threads); if (page_cleaner->n_disabled_debug == all_flushing_threads) { @@ -2912,6 +3071,7 @@ static void buf_flush_page_coordinator_thread() { ulint n_flushed = 0; ulint last_activity = srv_get_activity_count(); ulint last_pages = 0; + ulint n_evicted = 0; THD *thd = create_internal_thd(); @@ -2942,6 +3102,7 @@ static void buf_flush_page_coordinator_thread() { srv_shutdown_state.load() < SRV_SHUTDOWN_CLEANUP && recv_sys->spaces != nullptr) { /* treat flushing requests during recovery. */ + buf_flush_batch_result_t lru_result{}; ulint n_flushed_list = 0; os_event_wait(recv_sys->flush_start); @@ -2951,13 +3112,27 @@ static void buf_flush_page_coordinator_thread() { break; } - /* Flush all pages. LRU flushing during recovery is handled by the - per-pool LRU manager threads, which run alongside this loop. */ - do { - pc_request(ULINT_MAX, LSN_MAX); - while (pc_flush_slot() > 0) { - } - } while (!pc_wait_finished(&n_flushed_list)); + switch (recv_sys->flush_type) { + case BUF_FLUSH_LRU: + /* Flush pages from end of LRU if required */ + pc_request(0, LSN_MAX); + while (pc_flush_slot() > 0) { + } + pc_wait_finished(&lru_result, &n_flushed_list); + break; + + case BUF_FLUSH_LIST: + /* Flush all pages */ + do { + pc_request(ULINT_MAX, LSN_MAX); + while (pc_flush_slot() > 0) { + } + } while (!pc_wait_finished(&lru_result, &n_flushed_list)); + break; + + default: + ut_d(ut_error); + } os_event_reset(recv_sys->flush_start); os_event_set(recv_sys->flush_end); @@ -3024,7 +3199,7 @@ static void buf_flush_page_coordinator_thread() { ib::info(ER_IB_MSG_128) << "Page cleaner took " << diff_ms.count() << "ms to flush " - << n_flushed_last << " pages"; + << n_flushed_last << " and evict " << n_evicted << " pages"; if (warn_interval > 300) { warn_interval = 600; @@ -3043,10 +3218,12 @@ static void buf_flush_page_coordinator_thread() { } loop_start_time = curr_time; - n_flushed_last = 0; + n_flushed_last = n_evicted = 0; was_server_active = srv_check_activity(last_activity); last_activity = srv_get_activity_count(); + + pc_publish_lru_batch_stats(); } lsn_t lsn_limit; @@ -3119,26 +3296,29 @@ static void buf_flush_page_coordinator_thread() { page_cleaner->flush_pass++; /* Wait for all slots to be finished */ + buf_flush_batch_result_t lru_result{}; ulint n_flushed_list = 0; - pc_wait_finished(&n_flushed_list); + pc_wait_finished(&lru_result, &n_flushed_list); + pc_publish_lru_batch_stats(); - if (n_flushed_list > 0) { - buf_flush_stats(n_flushed_list, 0); + if (n_flushed_list > 0 || lru_result.n_flushed > 0) { + buf_flush_stats(n_flushed_list, lru_result.n_flushed); } if (n_to_flush != 0) { last_pages = n_flushed_list; } + n_evicted += lru_result.n_flushed; n_flushed_last += n_flushed_list; - n_flushed = n_flushed_list; + n_flushed = lru_result.n_flushed + n_flushed_list; if (is_sync_flush) { - MONITOR_INC_VALUE_CUMULATIVE(MONITOR_FLUSH_SYNC_TOTAL_PAGE, - MONITOR_FLUSH_SYNC_COUNT, - MONITOR_FLUSH_SYNC_PAGES, n_flushed_list); + MONITOR_INC_VALUE_CUMULATIVE( + MONITOR_FLUSH_SYNC_TOTAL_PAGE, MONITOR_FLUSH_SYNC_COUNT, + MONITOR_FLUSH_SYNC_PAGES, lru_result.n_flushed + n_flushed_list); } else { if (n_flushed_list) { MONITOR_INC_VALUE_CUMULATIVE( @@ -3208,10 +3388,11 @@ static void buf_flush_page_coordinator_thread() { while (pc_flush_slot() > 0) { } + buf_flush_batch_result_t lru_result{}; ulint n_flushed_list = 0; - pc_wait_finished(&n_flushed_list); + pc_wait_finished(&lru_result, &n_flushed_list); - n_flushed = n_flushed_list; + n_flushed = n_flushed_list + lru_result.n_flushed; /* We sleep only if there are no pages to flush */ if (n_flushed == 0) { @@ -3252,6 +3433,7 @@ static void buf_flush_page_coordinator_thread() { sweep and we'll come out of the loop leaving behind dirty pages in the flush_list */ buf_flush_await_no_flushing(nullptr, BUF_FLUSH_LIST); + buf_flush_await_no_flushing(nullptr, BUF_FLUSH_LRU); ut_ad(buf_flush_active_lru_managers() == 0); bool success; @@ -3273,12 +3455,14 @@ static void buf_flush_page_coordinator_thread() { while (pc_flush_slot() > 0) { } + buf_flush_batch_result_t lru_result{}; ulint n_flushed_list = 0; - success = pc_wait_finished(&n_flushed_list); + success = pc_wait_finished(&lru_result, &n_flushed_list); - n_flushed = n_flushed_list; + n_flushed = n_flushed_list + lru_result.n_flushed; buf_flush_await_no_flushing(nullptr, BUF_FLUSH_LIST); + buf_flush_await_no_flushing(nullptr, BUF_FLUSH_LRU); } while (!success || n_flushed > 0 || are_any_read_ios_still_underway || buf_get_flush_list_len(nullptr) > 0); @@ -3477,16 +3661,31 @@ static void buf_lru_manager_thread(size_t buf_pool_instance) { next_loop_time = std::chrono::steady_clock::now() + lru_sleep_time; - lru_n_flushed = buf_flush_LRU_list(buf_pool); + /* {n_flushed, n_evicted}: evicting a clean page refills the free list + just as a flush does, so the adaptive-sleep "made progress" signal must + consider both. The flush-specific stats below count flushes only. + buf_flush_LRU_list() (via buf_flush_start()) is a no-op returning a zero + result if buf_pool_invalidate_instance() has cleared flushing_allowed in + the meantime; resetting run_lru only guarantees that the *next* + os_event_wait() above parks, so this check is what stops a thread that + already returned from the wait (or was in the sleep) from starting a + batch after invalidation began. */ + + const auto lru_start = std::chrono::steady_clock::now(); + bool started = false; + const auto result = buf_flush_LRU_list(buf_pool, &started); + const auto lru_time = std::chrono::steady_clock::now() - lru_start; + + if (!started) { + continue; + } - buf_flush_await_no_flushing(buf_pool, BUF_FLUSH_LRU); + buf_lru_flush_stat_record(buf_pool, result, lru_time); - if (lru_n_flushed) { - srv_stats.buf_pool_flushed.add(lru_n_flushed); + buf_flush_await_no_flushing(buf_pool, BUF_FLUSH_LRU); - MONITOR_INC_VALUE_CUMULATIVE( - MONITOR_LRU_BATCH_FLUSH_TOTAL_PAGE, MONITOR_LRU_BATCH_FLUSH_COUNT, - MONITOR_LRU_BATCH_FLUSH_PAGES, lru_n_flushed); + if (result.n_flushed) { + srv_stats.buf_pool_flushed.add(result.n_flushed); } } } diff --git a/storage/innobase/handler/ha_innodb.cc b/storage/innobase/handler/ha_innodb.cc index 6cd12d389d5b..a2be3f2f25ef 100644 --- a/storage/innobase/handler/ha_innodb.cc +++ b/storage/innobase/handler/ha_innodb.cc @@ -775,6 +775,7 @@ static PSI_mutex_info all_innodb_mutexes[] = { PSI_MUTEX_KEY(dblwr_mutex, 0, 0, PSI_DOCUMENT_ME), PSI_MUTEX_KEY(purge_sys_pq_mutex, 0, 0, PSI_DOCUMENT_ME), PSI_MUTEX_KEY(recv_sys_mutex, 0, 0, PSI_DOCUMENT_ME), + PSI_MUTEX_KEY(recv_writer_mutex, 0, 0, PSI_DOCUMENT_ME), PSI_MUTEX_KEY(temp_space_rseg_mutex, 0, 0, PSI_DOCUMENT_ME), PSI_MUTEX_KEY(undo_space_rseg_mutex, 0, 0, PSI_DOCUMENT_ME), #ifdef UNIV_DEBUG @@ -885,6 +886,8 @@ static PSI_thread_info all_innodb_threads[] = { PSI_FLAG_SINGLETON, 0, PSI_DOCUMENT_ME), PSI_THREAD_KEY(log_flush_notifier_thread, "ib_log_fl_notif", PSI_FLAG_SINGLETON, 0, PSI_DOCUMENT_ME), + PSI_THREAD_KEY(recv_writer_thread, "ib_recv_write", PSI_FLAG_SINGLETON, 0, + PSI_DOCUMENT_ME), PSI_THREAD_KEY(buf_lru_manager_thread, "ib_buf_lru", 0, 0, PSI_DOCUMENT_ME), PSI_THREAD_KEY(srv_error_monitor_thread, "ib_srv_err", PSI_FLAG_SINGLETON, 0, PSI_DOCUMENT_ME), @@ -23597,6 +23600,14 @@ static MYSQL_SYSVAR_ULONG(lru_scan_depth, srv_LRU_scan_depth, "How deep to scan LRU to keep it clean", nullptr, nullptr, 1024, 100, UINT32_MAX, 0); +static MYSQL_SYSVAR_BOOL( + lru_threads, srv_lru_threads_enabled, + PLUGIN_VAR_OPCMDARG | PLUGIN_VAR_READONLY, + "Enable to use LRU manager threads that flush the LRU tail and refill " + "the free list. There would be one thread for each buffer pool instance. " + "When disabled (the default), page cleaners perform the LRU flushing.", + nullptr, nullptr, false); + static MYSQL_SYSVAR_ULONG(flush_neighbors, srv_flush_neighbors, PLUGIN_VAR_OPCMDARG, "Set to 0 (don't flush neighbors from buffer pool)," @@ -24518,6 +24529,7 @@ static SYS_VAR *innobase_system_variables[] = { MYSQL_SYSVAR(buffer_pool_load_abort), MYSQL_SYSVAR(buffer_pool_load_at_startup), MYSQL_SYSVAR(lru_scan_depth), + MYSQL_SYSVAR(lru_threads), MYSQL_SYSVAR(flush_neighbors), MYSQL_SYSVAR(checksum_algorithm), MYSQL_SYSVAR(log_checksums), diff --git a/storage/innobase/include/buf0buf.h b/storage/innobase/include/buf0buf.h index 7228f72cc7a0..1a97d40c66d1 100644 --- a/storage/innobase/include/buf0buf.h +++ b/storage/innobase/include/buf0buf.h @@ -2503,6 +2503,38 @@ struct buf_pool_t { for the duration of a teardown. Protected by flush_state_mutex. */ bool flushing_allowed; + /** Per-instance LRU flush accounting. Written by either this instance's + LRU manager or its page-cleaner slot and read (summed across instances) by + the page cleaner coordinator. pc_publish_lru_batch_stats() publishes the + buffer_LRU_batch_* counters independently, while + Adaptive_flush::set_average() consumes the timing fields. Protected by + flush_state_mutex so each consumer can atomically gather and reset its + fields. */ + struct lru_flush_stat_t { + /** Pages written to disk by LRU batches. */ + uint64_t n_flushed_pages; + /** LRU batches that wrote at least one page. */ + uint64_t n_flush_batches; + /** Largest number of pages written by one LRU batch. */ + uint64_t max_flushed_pages_per_batch; + /** Clean or stale pages evicted by LRU batches. */ + uint64_t n_evicted_pages; + /** LRU batches that evicted at least one page. */ + uint64_t n_evict_batches; + /** Largest number of pages evicted by one LRU batch. */ + uint64_t max_evicted_pages_per_batch; + /** Pages examined by LRU batches. */ + uint64_t n_scanned_pages; + /** LRU batches that examined at least one page. */ + uint64_t n_scan_batches; + /** Largest number of pages examined by one LRU batch. */ + uint64_t max_scanned_pages_per_batch; + /** Number of buf_flush_LRU_list() calls in the interval. */ + uint64_t n_lru_passes; + /** LRU time accumulated in the interval, in milliseconds. */ + uint64_t lru_flush_time_ms; + } lru_flush_stat; + /** A sequence number used to count the number of buffer blocks removed from the end of the LRU list; NOTE that this counter may wrap around at 4 billion! A thread is allowed to read this for heuristic purposes without diff --git a/storage/innobase/include/buf0flu.h b/storage/innobase/include/buf0flu.h index 9d8053aa8b7a..7f356cf35041 100644 --- a/storage/innobase/include/buf0flu.h +++ b/storage/innobase/include/buf0flu.h @@ -35,6 +35,8 @@ this program; if not, write to the Free Software Foundation, Inc., #ifndef buf0flu_h #define buf0flu_h +#include + #include "buf0types.h" #include "log0types.h" #include "univ.i" @@ -110,7 +112,17 @@ buf_flush_batch() and buf_flush_page(). [[nodiscard]] bool buf_flush_page_try(buf_pool_t *buf_pool, buf_block_t *block); #endif /* UNIV_DEBUG || UNIV_IBUF_DEBUG */ -/** Do flushing batch of a given type. +/** Result of a buffer flush batch. */ +struct buf_flush_batch_result_t { + /** Pages for which a write was queued. */ + ulint n_flushed{}; + /** Clean or stale pages moved directly to the free list. */ + ulint n_evicted{}; + /** Pages examined by the batch. Currently reported for BUF_FLUSH_LRU. */ + ulint n_scanned{}; +}; + +/** Do a flush-list batch. NOTE: The calling thread is not allowed to own any latches on pages! @param[in,out] buf_pool buffer pool instance @param[in] type flush type @@ -119,12 +131,12 @@ NOTE: The calling thread is not allowed to own any latches on pages! @param[in] lsn_limit in the case BUF_FLUSH_LIST all blocks whose oldest_modification is smaller than this should be flushed (if their number does not exceed min_n), otherwise ignored -@param[out] n_processed the number of pages which were processed is -passed back to caller. Ignored if NULL +@param[out] result batch result; n_evicted and n_scanned are 0. +Ignored if NULL @retval true if a batch was queued successfully. @retval false if another batch of same type was already running. */ bool buf_flush_do_batch(buf_pool_t *buf_pool, buf_flush_t type, ulint min_n, - lsn_t lsn_limit, ulint *n_processed); + lsn_t lsn_limit, buf_flush_batch_result_t *result); /** This utility flushes dirty blocks from the end of the flush list of all buffer pool instances. @@ -189,8 +201,8 @@ bool buf_flush_ready_for_replace(const buf_page_t *bpage); #ifdef UNIV_DEBUG struct SYS_VAR; -/** Disables page cleaner threads (coordinator and workers) and LRU manager -threads. It's used by: SET GLOBAL innodb_page_cleaner_disabled_debug = 1 (0). +/** Disables page cleaner threads (coordinator and workers) and LRU threads. +It's used by: SET GLOBAL innodb_page_cleaner_disabled_debug = 1 (0). @param[in] thd thread handle @param[in] var pointer to system variable @param[out] var_ptr where the formal string goes diff --git a/storage/innobase/include/log0recv.h b/storage/innobase/include/log0recv.h index 4ba08b069c6c..2e731fb8546d 100644 --- a/storage/innobase/include/log0recv.h +++ b/storage/innobase/include/log0recv.h @@ -536,12 +536,20 @@ struct recv_sys_t { n_pages_to_recover, and the state field in each recv_addr struct */ ib_mutex_t mutex; + /** mutex coordinating flushing between recv_writer_thread and + the recovery thread. */ + ib_mutex_t writer_mutex; + /** event to activate page cleaner threads */ os_event_t flush_start; /** event to signal that the page cleaner has finished the request */ os_event_t flush_end; + /** type of the flush request. BUF_FLUSH_LRU: flush end of LRU, + keeping free blocks. BUF_FLUSH_LIST: flush all of blocks. */ + buf_flush_t flush_type; + #else /* !UNIV_HOTBACKUP */ bool apply_file_operations; #endif /* !UNIV_HOTBACKUP */ diff --git a/storage/innobase/include/srv0srv.h b/storage/innobase/include/srv0srv.h index 1b936541b628..45a08fdb6a85 100644 --- a/storage/innobase/include/srv0srv.h +++ b/storage/innobase/include/srv0srv.h @@ -231,6 +231,9 @@ struct Srv_threads { /** Thread doing rollbacks during recovery. */ IB_thread m_trx_recovery_rollback; + /** Thread writing recovered pages during recovery. */ + IB_thread m_recv_writer; + /** Purge coordinator (also being a worker) */ IB_thread m_purge_coordinator; @@ -255,8 +258,10 @@ struct Srv_threads { buf_pool instance. */ size_t m_lru_managers_n; - /** LRU manager threads — sole owners of buf_flush_LRU_list for - free-list refill. The page cleaner only flushes the flush_list. */ + /** LRU manager threads. When innodb_lru_threads is enabled, they are the + sole owners of buf_flush_LRU_list for free-list refill and the page + cleaner only flushes the flush_list; when disabled, no threads are + started and the page cleaner performs LRU flushing. */ IB_thread *m_lru_managers; /** Archiver's log archiver (used by Clone). */ @@ -630,6 +635,8 @@ extern bool srv_validate_tablespace_paths; extern bool srv_use_fdatasync; /** Scan depth for LRU flush batch i.e.: number of blocks scanned*/ extern ulong srv_LRU_scan_depth; +/** Whether per-pool LRU manager threads are enabled (after recovery). */ +extern bool srv_lru_threads_enabled; /** Whether or not to flush neighbors of a block */ extern ulong srv_flush_neighbors; /** Previously requested size. Accesses protected by memory barriers. */ @@ -913,6 +920,7 @@ extern mysql_pfs_key_t log_write_notifier_thread_key; extern mysql_pfs_key_t log_flush_notifier_thread_key; extern mysql_pfs_key_t page_flush_coordinator_thread_key; extern mysql_pfs_key_t page_flush_thread_key; +extern mysql_pfs_key_t recv_writer_thread_key; extern mysql_pfs_key_t srv_error_monitor_thread_key; extern mysql_pfs_key_t srv_lock_timeout_thread_key; extern mysql_pfs_key_t srv_master_thread_key; diff --git a/storage/innobase/include/sync0sync.h b/storage/innobase/include/sync0sync.h index 87bf49ce8438..8d75b6b772d1 100644 --- a/storage/innobase/include/sync0sync.h +++ b/storage/innobase/include/sync0sync.h @@ -135,6 +135,7 @@ extern mysql_pfs_key_t recalc_pool_mutex_key; extern mysql_pfs_key_t page_cleaner_mutex_key; extern mysql_pfs_key_t purge_sys_pq_mutex_key; extern mysql_pfs_key_t recv_sys_mutex_key; +extern mysql_pfs_key_t recv_writer_mutex_key; extern mysql_pfs_key_t rtr_active_mutex_key; extern mysql_pfs_key_t rtr_match_mutex_key; extern mysql_pfs_key_t rtr_path_mutex_key; diff --git a/storage/innobase/include/sync0types.h b/storage/innobase/include/sync0types.h index 7c564c051542..758ff4868d49 100644 --- a/storage/innobase/include/sync0types.h +++ b/storage/innobase/include/sync0types.h @@ -328,6 +328,8 @@ enum latch_level_t { SYNC_TRX_I_S_RWLOCK, + SYNC_RECV_WRITER, + /** Level is varying. Only used with buffer pool page locks, which do not have a fixed level, but instead have their level set after the page is locked; see e.g. ibuf_bitmap_get_map_page(). */ @@ -405,6 +407,7 @@ enum latch_id_t { LATCH_ID_PURGE_SYS_PQ, LATCH_ID_RECALC_POOL, LATCH_ID_RECV_SYS, + LATCH_ID_RECV_WRITER, LATCH_ID_TEMP_SPACE_RSEG, LATCH_ID_UNDO_SPACE_RSEG, LATCH_ID_RW_LOCK_DEBUG, @@ -1133,7 +1136,8 @@ struct dict_sync_check : public sync_check_functor_t { if (!m_dict_mutex_allowed || (level != SYNC_DICT && level != SYNC_UNDO_SPACES && level != SYNC_FTS_CACHE && level != SYNC_DICT_OPERATION && - level != SYNC_NO_ORDER_CHECK)) { + /* This only happens in recv_apply_hashed_log_recs. */ + level != SYNC_RECV_WRITER && level != SYNC_NO_ORDER_CHECK)) { m_result = true; #ifdef UNIV_NO_ERR_MSGS ib::error() diff --git a/storage/innobase/log/log0recv.cc b/storage/innobase/log/log0recv.cc index 286373195ab4..4adb6f847c8c 100644 --- a/storage/innobase/log/log0recv.cc +++ b/storage/innobase/log/log0recv.cc @@ -190,6 +190,17 @@ is bigger than the lsn we are able to scan up to, that is an indication that the recovery failed and the database may be corrupt. */ static lsn_t recv_max_page_lsn; +#ifndef UNIV_HOTBACKUP +#ifdef UNIV_PFS_THREAD +mysql_pfs_key_t recv_writer_thread_key; +#endif /* UNIV_PFS_THREAD */ + +static bool recv_writer_is_active() { + return srv_thread_is_active(srv_threads.m_recv_writer); +} + +#endif /* !UNIV_HOTBACKUP */ + /* prototypes */ #ifndef UNIV_HOTBACKUP @@ -320,6 +331,7 @@ void recv_sys_create() { ut::zalloc_withkey(UT_NEW_THIS_FILE_PSI_KEY, sizeof(*recv_sys))); ut_a(recv_sys->last_block_first_mtr_boundary == 0); mutex_create(LATCH_ID_RECV_SYS, &recv_sys->mutex); + mutex_create(LATCH_ID_RECV_WRITER, &recv_sys->writer_mutex); recv_sys->spaces = nullptr; } @@ -419,6 +431,11 @@ void recv_sys_close() { mutex_free(&recv_sys->mutex); +#ifndef UNIV_HOTBACKUP + ut_ad(!recv_writer_is_active()); +#endif /* !UNIV_HOTBACKUP */ + mutex_free(&recv_sys->writer_mutex); + ut::free(recv_sys); recv_sys = nullptr; } @@ -684,6 +701,58 @@ void MetadataRecover::store() { mutex_exit(&dict_persist->mutex); } +/** recv_writer thread tasked with flushing dirty pages from the buffer +pools. */ +static void recv_writer_thread() { + ut_ad(!srv_read_only_mode); + + /* The code flow is as follows: + Step 1: In recv_recovery_from_checkpoint_start(). + Step 2: This recv_writer thread is started. + Step 3: In recv_recovery_from_checkpoint_finish(). + Step 4: Wait for recv_writer thread to complete. + Step 5: Assert that recv_writer thread is not active anymore. + + It is possible that the thread that is started in step 2, + becomes active only after step 4 and hence the assert in + step 5 fails. So mark this thread active only if necessary. */ + mutex_enter(&recv_sys->writer_mutex); + + if (!recv_recovery_on) { + mutex_exit(&recv_sys->writer_mutex); + return; + } + mutex_exit(&recv_sys->writer_mutex); + + while (srv_shutdown_state.load() == SRV_SHUTDOWN_NONE) { + ut_a(srv_shutdown_state_matches([](auto state) { + return state == SRV_SHUTDOWN_NONE || state == SRV_SHUTDOWN_EXIT_THREADS; + })); + + std::this_thread::sleep_for(std::chrono::milliseconds(100)); + + mutex_enter(&recv_sys->writer_mutex); + + if (!recv_recovery_on) { + mutex_exit(&recv_sys->writer_mutex); + break; + } + + if (log_test != nullptr) { + mutex_exit(&recv_sys->writer_mutex); + continue; + } + + /* Flush pages from end of LRU if required */ + os_event_reset(recv_sys->flush_end); + recv_sys->flush_type = BUF_FLUSH_LRU; + os_event_set(recv_sys->flush_start); + os_event_wait(recv_sys->flush_end); + + mutex_exit(&recv_sys->writer_mutex); + } +} + #endif /* !UNIV_HOTBACKUP */ /** Frees the recovery system. */ @@ -698,6 +767,7 @@ void recv_sys_free() { /* wake page cleaner up to progress */ if (!srv_read_only_mode) { ut_ad(!recv_recovery_on); + ut_ad(!recv_writer_is_active()); if (buf_flush_event != nullptr) { os_event_reset(buf_flush_event); } @@ -1191,6 +1261,16 @@ void recv_apply_hashed_log_recs(log_t &log) { mutex_exit(&recv_sys->mutex); + /* Stop the recv_writer thread from issuing any LRU + flush batches. */ + mutex_enter(&recv_sys->writer_mutex); + + /* Wait for any currently run batch to end. Note that BUF_FLUSH_LIST could + only be initiated by us in earlier call, but buf_pool_invalidate() waits for + all batches to finish, so only BUF_FLUSH_LRU can be running. + TBD: why is it important to wait for BUF_FLUSH_LRU to finish here? */ + buf_flush_await_no_flushing(nullptr, BUF_FLUSH_LRU); + os_event_reset(recv_sys->flush_end); /* We are about to request BUF_FLUSH_LIST, in hope to write all dirty pages @@ -1206,12 +1286,17 @@ void recv_apply_hashed_log_recs(log_t &log) { which has the same issue: skips over io-fixed pages. */ buf_pool_wait_for_no_pending_io(); + recv_sys->flush_type = BUF_FLUSH_LIST; + os_event_set(recv_sys->flush_start); os_event_wait(recv_sys->flush_end); buf_pool_invalidate(); + /* Allow batches from recv_writer thread. */ + mutex_exit(&recv_sys->writer_mutex); + ut_d(log.disable_redo_writes = false); mutex_enter(&recv_sys->mutex); @@ -3668,6 +3753,16 @@ static void recv_init_crash_recovery() { ib::info(ER_IB_MSG_727); recv_sys->dblwr->recover(); + + if (srv_force_recovery < SRV_FORCE_NO_LOG_REDO) { + /* Spawn the background thread to flush dirty pages + from the buffer pools. */ + + srv_threads.m_recv_writer = + os_thread_create(recv_writer_thread_key, 0, recv_writer_thread); + + srv_threads.m_recv_writer.start(); + } } dberr_t recv_recovery_from_checkpoint_start(log_t &log, lsn_t flush_lsn) { @@ -3855,17 +3950,39 @@ static void verify_page_type(page_id_t page_id, page_type_t type) { } MetadataRecover *recv_recovery_from_checkpoint_finish(bool aborting) { + /* Make sure that the recv_writer thread is done. This is + required because it grabs various mutexes and we want to + ensure that when we enable sync_order_checks there is no + mutex currently held by any thread. */ + mutex_enter(&recv_sys->writer_mutex); + /* Restore state. */ if (recv_sys->is_meb_db) dblwr::g_mode = recv_sys->dblwr_state; /* Free the resources of the recovery system */ recv_recovery_on = false; - /* Wait for LRU batches currently in progress (dispatched by the LRU - manager threads) to finish. Note that BUF_FLUSH_LIST batches are awaited - to finish before we get here. */ + /* By acquiring the mutex we ensure that the recv_writer thread won't trigger + any more LRU batches. Now wait for currently in progress batches to finish. + Note that BUF_FLUSH_LIST batches are awaited to finish before we get here. + TBD: Why is it important to wait for BUF_FLUSH_LRU to finish here? */ buf_flush_await_no_flushing(nullptr, BUF_FLUSH_LRU); + mutex_exit(&recv_sys->writer_mutex); + + uint32_t count = 0; + + while (recv_writer_is_active()) { + ++count; + + std::this_thread::sleep_for(std::chrono::milliseconds(100)); + + if (count >= 600) { + ib::info(ER_IB_MSG_738); + count = 0; + } + } + MetadataRecover *metadata{}; if (!aborting) { diff --git a/storage/innobase/srv/srv0mon.cc b/storage/innobase/srv/srv0mon.cc index c4a6dc7822d6..c282f90db181 100644 --- a/storage/innobase/srv/srv0mon.cc +++ b/storage/innobase/srv/srv0mon.cc @@ -404,7 +404,7 @@ static monitor_info_t innodb_counter_info[] = { "Avg time (ms) spent for adaptive flushing recently per slot.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_FLUSH_ADAPTIVE_AVG_TIME_SLOT}, - {"buffer_LRU_batch_flush_avg_time_slot", "buffer", // TODO: always zero + {"buffer_LRU_batch_flush_avg_time_slot", "buffer", "Avg time (ms) spent for LRU batch flushing recently per slot.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_LRU_BATCH_FLUSH_AVG_TIME_SLOT}, @@ -414,7 +414,7 @@ static monitor_info_t innodb_counter_info[] = { MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_FLUSH_ADAPTIVE_AVG_TIME_THREAD}, - {"buffer_LRU_batch_flush_avg_time_thread", "buffer", // TODO: always zero + {"buffer_LRU_batch_flush_avg_time_thread", "buffer", "Avg time (ms) spent for LRU batch flushing recently per thread.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_LRU_BATCH_FLUSH_AVG_TIME_THREAD}, @@ -423,7 +423,7 @@ static monitor_info_t innodb_counter_info[] = { "Estimated time (ms) spent for adaptive flushing recently.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_FLUSH_ADAPTIVE_AVG_TIME_EST}, - {"buffer_LRU_batch_flush_avg_time_est", "buffer", // TODO: always zero + {"buffer_LRU_batch_flush_avg_time_est", "buffer", "Estimated time (ms) spent for LRU batch flushing recently.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_LRU_BATCH_FLUSH_AVG_TIME_EST}, @@ -435,7 +435,7 @@ static monitor_info_t innodb_counter_info[] = { "Number of adaptive flushes passed during the recent Avg period.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_FLUSH_ADAPTIVE_AVG_PASS}, - {"buffer_LRU_batch_flush_avg_pass", "buffer", // TODO: always zero + {"buffer_LRU_batch_flush_avg_pass", "buffer", "Number of LRU batch flushes passed during the recent Avg period.", MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_LRU_BATCH_FLUSH_AVG_PASS}, diff --git a/storage/innobase/srv/srv0srv.cc b/storage/innobase/srv/srv0srv.cc index 98458b9f6eda..56199ac81a36 100644 --- a/storage/innobase/srv/srv0srv.cc +++ b/storage/innobase/srv/srv0srv.cc @@ -460,6 +460,8 @@ bool srv_validate_tablespace_paths = true; bool srv_use_fdatasync = false; /** Scan depth for LRU flush batch i.e.: number of blocks scanned*/ ulong srv_LRU_scan_depth = 1024; +/** Whether per-pool LRU manager threads are enabled (after recovery). */ +bool srv_lru_threads_enabled = false; /** Whether or not to flush neighbors of a block */ ulong srv_flush_neighbors = 1; /** Previously requested size. Accesses protected by memory barriers. */ diff --git a/storage/innobase/srv/srv0start.cc b/storage/innobase/srv/srv0start.cc index ab55cee0b4ee..4fecf9947f16 100644 --- a/storage/innobase/srv/srv0start.cc +++ b/storage/innobase/srv/srv0start.cc @@ -2489,6 +2489,8 @@ void srv_pre_dd_shutdown() { /* Crash if some query threads are still alive. */ ut_a(srv_conc_get_active_threads() == 0); + ut_a(!srv_thread_is_active(srv_threads.m_recv_writer)); + /* Avoid fast shutdown, if redo logging is disabled. Otherwise, we won't be able to recover. */ if (mtr_t::s_logging.is_disabled() && srv_fast_shutdown == 2) { @@ -2844,6 +2846,7 @@ void srv_shutdown() { std::cref(srv_threads.m_purge_coordinator), std::cref(srv_threads.m_ts_alter_encrypt), std::cref(srv_threads.m_fts_optimize), + std::cref(srv_threads.m_recv_writer), std::cref(srv_threads.m_dict_stats)}; for (const auto &thread : threads_stopped_before_shutdown) { diff --git a/storage/innobase/sync/sync0debug.cc b/storage/innobase/sync/sync0debug.cc index 81593c84bf38..151ec5edca7e 100644 --- a/storage/innobase/sync/sync0debug.cc +++ b/storage/innobase/sync/sync0debug.cc @@ -473,6 +473,7 @@ LatchDebug::LatchDebug() { LEVEL_MAP_INSERT(SYNC_FTS_BG_THREADS); LEVEL_MAP_INSERT(SYNC_FTS_CACHE_INIT); LEVEL_MAP_INSERT(SYNC_RECV); + LEVEL_MAP_INSERT(SYNC_RECV_WRITER); LEVEL_MAP_INSERT(SYNC_LOG_ONLINE); LEVEL_MAP_INSERT(SYNC_LOG_SN); LEVEL_MAP_INSERT(SYNC_LOG_SN_MUTEX); @@ -723,6 +724,7 @@ Latches *LatchDebug::check_order(const latch_t *latch, case SYNC_LOCK_FREE_HASH: case SYNC_MONITOR_MUTEX: case SYNC_RECV: + case SYNC_RECV_WRITER: case SYNC_FTS_BG_THREADS: case SYNC_WORK_QUEUE: case SYNC_FTS_TOKENIZE: @@ -1328,6 +1330,8 @@ static void sync_latch_meta_init() UNIV_NOTHROW { LATCH_ADD_MUTEX(RECV_SYS, SYNC_RECV, recv_sys_mutex_key); + LATCH_ADD_MUTEX(RECV_WRITER, SYNC_RECV_WRITER, recv_writer_mutex_key); + LATCH_ADD_MUTEX(TEMP_SPACE_RSEG, SYNC_TEMP_SPACE_RSEG, temp_space_rseg_mutex_key); diff --git a/storage/innobase/sync/sync0sync.cc b/storage/innobase/sync/sync0sync.cc index b30e0560976e..428731720270 100644 --- a/storage/innobase/sync/sync0sync.cc +++ b/storage/innobase/sync/sync0sync.cc @@ -102,6 +102,7 @@ mysql_pfs_key_t recalc_pool_mutex_key; mysql_pfs_key_t page_cleaner_mutex_key; mysql_pfs_key_t purge_sys_pq_mutex_key; mysql_pfs_key_t recv_sys_mutex_key; +mysql_pfs_key_t recv_writer_mutex_key; mysql_pfs_key_t temp_space_rseg_mutex_key; mysql_pfs_key_t undo_space_rseg_mutex_key; mysql_pfs_key_t page_zip_stat_per_index_mutex_key; From 4d7fa5ddd20cf41d5cf2787bfdce4f913173bf56 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Pawe=C5=82=20Olchawa?= Date: Tue, 21 Jul 2026 19:42:34 +0200 Subject: [PATCH 27/32] PS-11445 [trunk]: [4] LRU: Count evictions as adaptive-sleep progress https://perconadev.atlassian.net/browse/PS-11445 The per-pool sleep-time adaptation only considered pages the LRU manager thread had flushed as "made progress" (lru_n_flushed). A batch that only evicted clean pages - which refills the free list exactly as a flush does - was treated the same as a batch that made no progress at all, causing the manager to back off its sleep time even while it was successfully keeping the free list topped up. Renames the counter to lru_n_processed and feeds it n_flushed + n_evicted, so eviction-only batches correctly count as progress for the free-list-fullness heuristic in buf_lru_manager_adapt_sleep_time(). The flush-specific counters (srv_stats.buf_pool_flushed, lru_manager_stat) are unaffected and continue to count flushes only. --- storage/innobase/buf/buf0flu.cc | 35 +++++++++++++++++---------------- 1 file changed, 18 insertions(+), 17 deletions(-) diff --git a/storage/innobase/buf/buf0flu.cc b/storage/innobase/buf/buf0flu.cc index 149bd7299b94..612bed7538f5 100644 --- a/storage/innobase/buf/buf0flu.cc +++ b/storage/innobase/buf/buf0flu.cc @@ -3587,36 +3587,32 @@ static void buf_lru_manager_sleep_if_needed( } } -/** Adjust the LRU manager thread sleep time based on the free list length and -the last flush result -@param[in] buf_pool buffer pool whom we are flushing -@param[in] lru_n_flushed last LRU flush page count -@param[in,out] lru_sleep_time LRU manager thread sleep time */ +/** Adjust the LRU manager thread's per-pool sleep time based on free-list +fullness and the work the last iteration accomplished. Aggressive shrinking +when the free list is near-empty, gradual growth when it's healthy. */ static void buf_lru_manager_adapt_sleep_time( - const buf_pool_t *buf_pool, ulint lru_n_flushed, + const buf_pool_t *buf_pool, size_t lru_n_processed, std::chrono::milliseconds &lru_sleep_time) { const auto free_len = UT_LIST_GET_LEN(buf_pool->free); const auto max_free_len = std::min(UT_LIST_GET_LEN(buf_pool->LRU), srv_LRU_scan_depth); - if (free_len < max_free_len / 100 && lru_n_flushed) { - /* Free list filled less than 1% and the last iteration was - able to flush, no sleep */ + if (free_len < max_free_len / 100 && lru_n_processed) { + /* Free list < 1% and we made progress last time: don't sleep. */ lru_sleep_time = std::chrono::milliseconds::zero(); } else if (free_len > max_free_len / 5 || - (free_len < max_free_len / 100 && lru_n_flushed == 0)) { - /* Free list filled more than 20% or no pages flushed in the - previous batch, sleep a bit more */ + (free_len < max_free_len / 100 && lru_n_processed == 0)) { + /* Free list > 20%, or near-empty but we made no progress (unusual): + back off a bit. */ lru_sleep_time += std::chrono::milliseconds{1}; if (lru_sleep_time > std::chrono::milliseconds{1000}) lru_sleep_time = std::chrono::milliseconds{1000}; } else if (free_len < max_free_len / 20 && lru_sleep_time >= std::chrono::milliseconds{50}) { - /* Free list filled less than 5%, sleep a bit less */ + /* Free list < 5%: shrink the sleep. */ lru_sleep_time -= std::chrono::milliseconds{50}; - } else { - /* Free lists filled between 5% and 20%, no change */ } + /* Otherwise (5%–20%): no change. */ } /** LRU manager thread. One per buf_pool instance. Periodically calls @@ -3648,7 +3644,9 @@ static void buf_lru_manager_thread(size_t buf_pool_instance) { std::chrono::milliseconds lru_sleep_time{1000}; auto next_loop_time = std::chrono::steady_clock::now() + lru_sleep_time; - ulint lru_n_flushed = 1; + /* Seed nonzero so the first adapt iteration treats us as "made progress" + and does not immediately back off. */ + size_t lru_n_processed = 1; while (srv_shutdown_state.load() < SRV_SHUTDOWN_FLUSH_PHASE) { ut_d(buf_flush_page_cleaner_disabled_loop()); @@ -3657,7 +3655,7 @@ static void buf_lru_manager_thread(size_t buf_pool_instance) { buf_lru_manager_sleep_if_needed(next_loop_time); - buf_lru_manager_adapt_sleep_time(buf_pool, lru_n_flushed, lru_sleep_time); + buf_lru_manager_adapt_sleep_time(buf_pool, lru_n_processed, lru_sleep_time); next_loop_time = std::chrono::steady_clock::now() + lru_sleep_time; @@ -3675,6 +3673,7 @@ static void buf_lru_manager_thread(size_t buf_pool_instance) { bool started = false; const auto result = buf_flush_LRU_list(buf_pool, &started); const auto lru_time = std::chrono::steady_clock::now() - lru_start; + lru_n_processed = result.n_flushed + result.n_evicted; if (!started) { continue; @@ -3682,6 +3681,8 @@ static void buf_lru_manager_thread(size_t buf_pool_instance) { buf_lru_flush_stat_record(buf_pool, result, lru_time); + /* Wait for the batch this iteration kicked off (if any) to finish so + the next iteration sees free pages on the free list. */ buf_flush_await_no_flushing(buf_pool, BUF_FLUSH_LRU); if (result.n_flushed) { From 306040186470bb20dcdf329d9fedaac62fd15b42 Mon Sep 17 00:00:00 2001 From: =?UTF-8?q?Pawe=C5=82=20Olchawa?= Date: Wed, 22 Jul 2026 14:42:08 +0200 Subject: [PATCH 28/32] PS-11445 [trunk]: [5] LRU: Single-page-flush concurrency cap https://perconadev.atlassian.net/browse/PS-11447 Previously, whenever an LRU batch flush was already in progress on a buffer pool instance and the doublewrite buffer was enabled, a user thread looking for a free block always waited for that in-progress LRU flush to finish instead of ever attempting its own single-page flush, or even a scan for pages that can be freed immediately. This was decreasing TPS for low concurrency when user thread had to wait for the whole batch as opposed to quickly solving the issue itself. This waiting mechanism is now enabled if and only if lru threads are enabled (innodb_lru_threads = ON), so disabled by default. Also, when this mechanism is enabled, a thread will wait only when there is already an in-flight single-page flush, i.e. the concurrent single-page flush count is capped at 1 per buffer pool instance. Also, before awaiting the LRU batch, user thread is scanning LRU list, searching for a page that could be freed immediately (without flushing). Additional, "not await twice" fix ensures a given buf_LRU_get_free_block() call waits for an in-progress flush at most once, so it does not keep waiting indefinitely on repeated loop iterations (to avoid starvation). New buffer_LRU_% monitor counters (disabled by default): - buffer_LRU_single_page_flush_count - buffer_LRU_flush_await_count --- mysql-test/suite/innodb/r/monitor.result | 2 ++ .../r/innodb_monitor_disable_basic.result | 2 ++ .../r/innodb_monitor_enable_basic.result | 2 ++ .../r/innodb_monitor_reset_all_basic.result | 2 ++ .../r/innodb_monitor_reset_basic.result | 2 ++ storage/innobase/buf/buf0flu.cc | 9 ++---- storage/innobase/buf/buf0lru.cc | 29 ++++++++++--------- storage/innobase/include/srv0mon.h | 2 ++ storage/innobase/srv/srv0mon.cc | 10 +++++++ 9 files changed, 41 insertions(+), 19 deletions(-) diff --git a/mysql-test/suite/innodb/r/monitor.result b/mysql-test/suite/innodb/r/monitor.result index 79e7114989f6..3650df0ca786 100644 --- a/mysql-test/suite/innodb/r/monitor.result +++ b/mysql-test/suite/innodb/r/monitor.result @@ -107,6 +107,8 @@ buffer_LRU_search_scanned_per_call disabled buffer_LRU_unzip_search_scanned disabled buffer_LRU_unzip_search_num_scan disabled buffer_LRU_unzip_search_scanned_per_call disabled +buffer_LRU_single_page_flush_count disabled +buffer_LRU_flush_await_count disabled buffer_page_read_index_leaf disabled buffer_page_read_index_non_leaf disabled buffer_page_read_index_ibuf_leaf disabled diff --git a/mysql-test/suite/sys_vars/r/innodb_monitor_disable_basic.result b/mysql-test/suite/sys_vars/r/innodb_monitor_disable_basic.result index adbb506eb3a8..77b303242dfa 100644 --- a/mysql-test/suite/sys_vars/r/innodb_monitor_disable_basic.result +++ b/mysql-test/suite/sys_vars/r/innodb_monitor_disable_basic.result @@ -107,6 +107,8 @@ buffer_LRU_search_scanned_per_call disabled buffer_LRU_unzip_search_scanned disabled buffer_LRU_unzip_search_num_scan disabled buffer_LRU_unzip_search_scanned_per_call disabled +buffer_LRU_single_page_flush_count disabled +buffer_LRU_flush_await_count disabled buffer_page_read_index_leaf disabled buffer_page_read_index_non_leaf disabled buffer_page_read_index_ibuf_leaf disabled diff --git a/mysql-test/suite/sys_vars/r/innodb_monitor_enable_basic.result b/mysql-test/suite/sys_vars/r/innodb_monitor_enable_basic.result index adbb506eb3a8..77b303242dfa 100644 --- a/mysql-test/suite/sys_vars/r/innodb_monitor_enable_basic.result +++ b/mysql-test/suite/sys_vars/r/innodb_monitor_enable_basic.result @@ -107,6 +107,8 @@ buffer_LRU_search_scanned_per_call disabled buffer_LRU_unzip_search_scanned disabled buffer_LRU_unzip_search_num_scan disabled buffer_LRU_unzip_search_scanned_per_call disabled +buffer_LRU_single_page_flush_count disabled +buffer_LRU_flush_await_count disabled buffer_page_read_index_leaf disabled buffer_page_read_index_non_leaf disabled buffer_page_read_index_ibuf_leaf disabled diff --git a/mysql-test/suite/sys_vars/r/innodb_monitor_reset_all_basic.result b/mysql-test/suite/sys_vars/r/innodb_monitor_reset_all_basic.result index adbb506eb3a8..77b303242dfa 100644 --- a/mysql-test/suite/sys_vars/r/innodb_monitor_reset_all_basic.result +++ b/mysql-test/suite/sys_vars/r/innodb_monitor_reset_all_basic.result @@ -107,6 +107,8 @@ buffer_LRU_search_scanned_per_call disabled buffer_LRU_unzip_search_scanned disabled buffer_LRU_unzip_search_num_scan disabled buffer_LRU_unzip_search_scanned_per_call disabled +buffer_LRU_single_page_flush_count disabled +buffer_LRU_flush_await_count disabled buffer_page_read_index_leaf disabled buffer_page_read_index_non_leaf disabled buffer_page_read_index_ibuf_leaf disabled diff --git a/mysql-test/suite/sys_vars/r/innodb_monitor_reset_basic.result b/mysql-test/suite/sys_vars/r/innodb_monitor_reset_basic.result index 88eeaaf46dca..eda84d8e40a1 100644 --- a/mysql-test/suite/sys_vars/r/innodb_monitor_reset_basic.result +++ b/mysql-test/suite/sys_vars/r/innodb_monitor_reset_basic.result @@ -107,6 +107,8 @@ buffer_LRU_search_scanned_per_call disabled buffer_LRU_unzip_search_scanned disabled buffer_LRU_unzip_search_num_scan disabled buffer_LRU_unzip_search_scanned_per_call disabled +buffer_LRU_single_page_flush_count disabled +buffer_LRU_flush_await_count disabled buffer_page_read_index_leaf disabled buffer_page_read_index_non_leaf disabled buffer_page_read_index_ibuf_leaf disabled diff --git a/storage/innobase/buf/buf0flu.cc b/storage/innobase/buf/buf0flu.cc index 612bed7538f5..7cbe7da1e2ca 100644 --- a/storage/innobase/buf/buf0flu.cc +++ b/storage/innobase/buf/buf0flu.cc @@ -1787,10 +1787,8 @@ void buf_flush_await_no_flushing(buf_pool_t *buf_pool, buf_flush_t flush_type) { } } - -bool buf_flush_do_batch(buf_pool_t *buf_pool, buf_flush_t type, - ulint min_n, lsn_t lsn_limit, - buf_flush_batch_result_t *result) { +bool buf_flush_do_batch(buf_pool_t *buf_pool, buf_flush_t type, ulint min_n, + lsn_t lsn_limit, buf_flush_batch_result_t *result) { ut_ad(type == BUF_FLUSH_LRU || type == BUF_FLUSH_LIST); if (result != nullptr) { @@ -2852,8 +2850,7 @@ static ulint pc_flush_slot(void) { buf_flush_batch_result_t result{}; succeeded_list = buf_flush_do_batch( - buf_pool, BUF_FLUSH_LIST, n_pages_requested, - lsn_limit, &result); + buf_pool, BUF_FLUSH_LIST, n_pages_requested, lsn_limit, &result); /* BUF_FLUSH_LIST never evicts and does not report its scan count through this result yet. */ ut_ad(result.n_evicted == 0); diff --git a/storage/innobase/buf/buf0lru.cc b/storage/innobase/buf/buf0lru.cc index 40ccd19a8fd6..41835dc81ccb 100644 --- a/storage/innobase/buf/buf0lru.cc +++ b/storage/innobase/buf/buf0lru.cc @@ -1393,7 +1393,7 @@ we put it to free list to be used. @return the free control block, in state BUF_BLOCK_READY_FOR_USE */ buf_block_t *buf_LRU_get_free_block(buf_pool_t *buf_pool) { buf_block_t *block = nullptr; - bool freed = false; + bool freed = false, no_flush_waited = false; ulint n_iterations = 0; ulint flush_failures = 0; bool started_monitor = false; @@ -1504,15 +1504,6 @@ buf_block_t *buf_LRU_get_free_block(buf_pool_t *buf_pool) { (srv_shutdown_state.load() != SRV_SHUTDOWN_NONE && srv_shutdown_state.load() != SRV_SHUTDOWN_CLEANUP)); } - if (buf_pool->init_flush[BUF_FLUSH_LRU] && dblwr::is_enabled()) { - /* If there is an LRU flush happening in the background then we - wait for it to end instead of trying a single page flush. If, - however, we are not using doublewrite buffer then it is better to - do our own single page flush instead of waiting for LRU flush to - end. */ - buf_flush_await_no_flushing(buf_pool, BUF_FLUSH_LRU); - goto loop; - } os_rmb; @@ -1591,9 +1582,21 @@ buf_block_t *buf_LRU_get_free_block(buf_pool_t *buf_pool) { involved (particularly in case of compressed pages). We can do that in a separate patch sometime in future. */ - if (!buf_flush_single_page_from_LRU(buf_pool)) { - MONITOR_INC(MONITOR_LRU_SINGLE_FLUSH_FAILURE_COUNT); - ++flush_failures; + if (srv_lru_threads_enabled && buf_pool->init_flush[BUF_FLUSH_LRU] && + dblwr::is_enabled() && buf_pool->n_flush[BUF_FLUSH_SINGLE_PAGE] >= 1 && + !no_flush_waited) { + /* Cap reached: wait for an in-progress LRU flush instead of issuing + our own single-page flush. */ + MONITOR_INC(MONITOR_LRU_FLUSH_AWAIT_COUNT); + buf_flush_await_no_flushing(buf_pool, BUF_FLUSH_LRU); + no_flush_waited = true; + } else { + /* Below the cap: issue our own single-page flush. */ + MONITOR_INC(MONITOR_LRU_SINGLE_PAGE_FLUSH_COUNT); + if (!buf_flush_single_page_from_LRU(buf_pool)) { + MONITOR_INC(MONITOR_LRU_SINGLE_FLUSH_FAILURE_COUNT); + ++flush_failures; + } } srv_stats.buf_pool_wait_free.add(n_iterations, 1); diff --git a/storage/innobase/include/srv0mon.h b/storage/innobase/include/srv0mon.h index c0d1cc2d7b89..b677faca1479 100644 --- a/storage/innobase/include/srv0mon.h +++ b/storage/innobase/include/srv0mon.h @@ -253,6 +253,8 @@ enum monitor_id_t { MONITOR_LRU_UNZIP_SEARCH_SCANNED, MONITOR_LRU_UNZIP_SEARCH_SCANNED_NUM_CALL, MONITOR_LRU_UNZIP_SEARCH_SCANNED_PER_CALL, + MONITOR_LRU_SINGLE_PAGE_FLUSH_COUNT, + MONITOR_LRU_FLUSH_AWAIT_COUNT, /* Buffer Page I/O specific counters. */ MONITOR_MODULE_BUF_PAGE, diff --git a/storage/innobase/srv/srv0mon.cc b/storage/innobase/srv/srv0mon.cc index c282f90db181..7aae2ed4c45b 100644 --- a/storage/innobase/srv/srv0mon.cc +++ b/storage/innobase/srv/srv0mon.cc @@ -601,6 +601,16 @@ static monitor_info_t innodb_counter_info[] = { MONITOR_LRU_UNZIP_SEARCH_SCANNED, MONITOR_LRU_UNZIP_SEARCH_SCANNED_PER_CALL}, + {"buffer_LRU_single_page_flush_count", "buffer", + "Times a user thread issued its own single-page flush while" + " searching for a free block", + MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_LRU_SINGLE_PAGE_FLUSH_COUNT}, + + {"buffer_LRU_flush_await_count", "buffer", + "Times a user thread waited for an in-progress LRU flush instead of" + " issuing single-page flush", + MONITOR_NONE, MONITOR_DEFAULT_START, MONITOR_LRU_FLUSH_AWAIT_COUNT}, + /* ========== Counters for Buffer Page I/O ========== */ {"module_buffer_page", "buffer_page_io", "Buffer Page I/O Module", static_cast(MONITOR_MODULE | MONITOR_GROUP_MODULE), From f52e5f3781b4f8138bcb302ead6bcd6de2aa2755 Mon Sep 17 00:00:00 2001 From: Przemyslaw Skibinski Date: Fri, 31 Jul 2026 10:38:05 +0200 Subject: [PATCH 29/32] PS-11473 [trunk] azure-pipelines: build five configs by default Cut Azure Pipelines usage on pull requests, and make the full compiler matrix something CI can ask for explicitly. - Drop the CI trigger (`trigger: none`); pushes no longer queue builds. The nightly 1:00 AM UTC schedule for 8.4 and PR validation are unaffected. - Skip PR validation while a pull request is a draft (`pr: drafts: false`); builds start when it is marked ready for review. - Replace the `clang-22 Debug INVERTED` matrix leg with `gcc-16 Debug INVERTED`, dropping the now-inert `UBUNTU_CODE_NAME` (it only feeds the apt.llvm.org repository line, which is guarded on clang). - Build only clang-22 RelWithDebInfo, clang-22 Debug, gcc-16 RelWithDebInfo, gcc-16 Debug and gcc-16 Debug INVERTED by default. The other 31 legs keep a per-leg compile-time guard, now on a queue-time parameter: ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }} This replaces the "fullci" branch-name escape hatch, which could not work. A guard is evaluated during template expansion, and the branch is not knowable then. Measured on a pull request from branch PS-11473-8.4-fullci, the compile-time values are: Build.Reason 'PullRequest' Build.SourceBranch 'refs/pull/6116/merge' Build.SourceBranchName 'merge' System.PullRequest.SourceBranch '' (runtime: the branch name) So Build.SourceBranchName is always "merge" on a pull request and the old expression never fired there; it only ever worked for non-PR runs, which its Build.Reason term already covered. A parameter is resolved when the run is created and therefore is visible to ${{ }}. It makes the pipeline callable from GitHub Actions in either mode: POST https://dev.azure.com/it0639/percona-server/_apis/pipelines//runs?api-version=7.1 { "resources": { "repositories": { "self": { "refName": "refs/pull//merge" } } }, "templateParameters": { "fullCI": true } } Queued without templateParameters it builds the five default configs; with fullCI true, all 36. The guard tests the parameter rather than "not a pull request" because an API-created run reports Build.Reason=Manual, which would otherwise make every queued run build everything. Schedule is kept so the nightly still covers all 36; a manual queue from the Azure UI now gets five unless the checkbox is ticked. Guards stay per-leg rather than grouped under a single ${{ if }}: that is the shape this pipeline used before and is known to expand. Verified by expanding both branches of the guard: the default run yields the five configs and a fullCI run yields the same 36 legs as before, with the job's steps, variables, pool and timeout unchanged. --- azure-pipelines.yml | 259 +++++++++++++++++++++++++------------------- 1 file changed, 149 insertions(+), 110 deletions(-) diff --git a/azure-pipelines.yml b/azure-pipelines.yml index b824dad2a423..b24df6d08417 100644 --- a/azure-pipelines.yml +++ b/azure-pipelines.yml @@ -1,28 +1,34 @@ +# Queue-time parameter. Resolved when the run is created, so it is one of the +# few things a ${{ }} expression can see -- which is what makes it usable as the +# full-CI gate (see the matrix below). +# +# GitHub Actions queues a run with: +# +# POST https://dev.azure.com/it0639/percona-server/_apis/pipelines//runs?api-version=7.1 +# { +# "resources": { "repositories": { "self": { "refName": "refs/pull//merge" } } }, +# "templateParameters": { "fullCI": true } +# } +# +# Omit templateParameters (or pass false) for the five default configs; pass +# true for all 30. It also appears as a checkbox in the Run pipeline dialog. +parameters: +- name: fullCI + displayName: 'Build all configs, not just the default five' + type: boolean + default: false + schedules: -- cron: "0 2 * * *" - displayName: Daily 2:00 AM UTC build +- cron: "0 3 * * *" + displayName: Daily 3:00 AM UTC build branches: include: - trunk -trigger: - branches: - include: - - '*' - exclude: - - trunk - paths: - exclude: - - doc - - build-ps - - man - - mysql-test - - packaging - - policy - - scripts - - support-files +trigger: none pr: + drafts: false branches: include: - '*' @@ -73,15 +79,75 @@ jobs: PARENT_BRANCH: trunk BUILD_PARAMS_TYPE: normal + # Two tiers of coverage: + # + # default the five configs listed first -- built on every pull request. + # extended the remaining 25, each guarded so they are built only when the + # fullCI parameter is set (how GitHub Actions asks for a full run) + # or on the nightly schedule. + # + # The guards are compile-time insertions, so on a default run the extra legs + # are never generated: no agents, and no skipped entries in the run summary. + # + # The gate keys on the parameter rather than on Build.Reason because a run + # created through the REST API reports Build.Reason=Manual -- so testing + # "not a pull request" would make every GitHub-Actions-queued run build all 30. + # Schedule is kept so the nightly still covers everything. + # + # It cannot key on the branch, which is what an earlier attempt did. Measured + # on a pull request from a branch whose name contained "fullci", the + # compile-time values are Build.SourceBranch='refs/pull//merge', + # Build.SourceBranchName='merge' and System.PullRequest.SourceBranch='' -- its + # runtime value is the branch, but expansion happens before that is set. + # + # A runtime condition cannot substitute either: + # job conditions are evaluated before matrix variables are in scope, so a + # per-leg condition lets every leg through. strategy: matrix: - macOS 14 RelWithDebInfo: - imageName: 'macOS-14' + clang-22 RelWithDebInfo [Ubuntu 24.04 Noble]: + imageName: 'ubuntu-24.04' + UBUNTU_CODE_NAME: noble + Compiler: clang + CompilerVer: 22 + BuildType: RelWithDebInfo + + clang-22 Debug [Ubuntu 24.04 Noble]: + imageName: 'ubuntu-24.04' + UBUNTU_CODE_NAME: noble Compiler: clang + CompilerVer: 22 + BuildType: Debug + + gcc-16 RelWithDebInfo [Ubuntu 24.04 Noble]: + imageName: 'ubuntu-24.04' + Compiler: gcc + CompilerVer: 16 BuildType: RelWithDebInfo - # skip for a pull request if branch name doesn't contain "fullci" - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + gcc-16 Debug [Ubuntu 24.04 Noble]: + imageName: 'ubuntu-24.04' + Compiler: gcc + CompilerVer: 16 + BuildType: Debug + + gcc-16 Debug INVERTED [Ubuntu 24.04 Noble]: + imageName: 'ubuntu-24.04' + Compiler: gcc + CompilerVer: 16 + BuildType: Debug + BUILD_PARAMS_TYPE: inverted + + # Extended coverage below: built only with fullCI, or on the nightly. + + # macOS 14 + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: + macOS 14 RelWithDebInfo: + imageName: 'macOS-14' + Compiler: clang + BuildType: RelWithDebInfo + + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: macOS 14 Debug: imageName: 'macOS-14' Compiler: clang @@ -89,14 +155,7 @@ jobs: # clang-16 and newer compilers - clang-22 RelWithDebInfo [Ubuntu 24.04 Noble]: - imageName: 'ubuntu-24.04' - UBUNTU_CODE_NAME: noble - Compiler: clang - CompilerVer: 22 - BuildType: RelWithDebInfo - - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: clang-22 RelWithDebInfo INVERTED [Ubuntu 24.04 Noble]: imageName: 'ubuntu-24.04' UBUNTU_CODE_NAME: noble @@ -105,23 +164,7 @@ jobs: BuildType: RelWithDebInfo BUILD_PARAMS_TYPE: inverted - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: - clang-22 Debug [Ubuntu 24.04 Noble]: - imageName: 'ubuntu-24.04' - UBUNTU_CODE_NAME: noble - Compiler: clang - CompilerVer: 22 - BuildType: Debug - - clang-22 Debug INVERTED [Ubuntu 24.04 Noble]: - imageName: 'ubuntu-24.04' - UBUNTU_CODE_NAME: noble - Compiler: clang - CompilerVer: 22 - BuildType: Debug - BUILD_PARAMS_TYPE: inverted - - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: clang-21 RelWithDebInfo [Ubuntu 24.04 Noble]: imageName: 'ubuntu-24.04' UBUNTU_CODE_NAME: noble @@ -129,7 +172,7 @@ jobs: CompilerVer: 21 BuildType: RelWithDebInfo - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: clang-21 Debug [Ubuntu 24.04 Noble]: imageName: 'ubuntu-24.04' UBUNTU_CODE_NAME: noble @@ -137,7 +180,7 @@ jobs: CompilerVer: 21 BuildType: Debug - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: clang-20 RelWithDebInfo [Ubuntu 24.04 Noble]: imageName: 'ubuntu-24.04' UBUNTU_CODE_NAME: noble @@ -145,14 +188,15 @@ jobs: CompilerVer: 20 BuildType: RelWithDebInfo - clang-20 Debug [Ubuntu 24.04 Noble]: - imageName: 'ubuntu-24.04' - UBUNTU_CODE_NAME: noble - Compiler: clang - CompilerVer: 20 - BuildType: Debug + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: + clang-20 Debug [Ubuntu 24.04 Noble]: + imageName: 'ubuntu-24.04' + UBUNTU_CODE_NAME: noble + Compiler: clang + CompilerVer: 20 + BuildType: Debug - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: clang-19 RelWithDebInfo [Ubuntu 24.04 Noble]: imageName: 'ubuntu-24.04' UBUNTU_CODE_NAME: noble @@ -160,7 +204,7 @@ jobs: CompilerVer: 19 BuildType: RelWithDebInfo - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: clang-19 Debug [Ubuntu 24.04 Noble]: imageName: 'ubuntu-24.04' UBUNTU_CODE_NAME: noble @@ -168,7 +212,7 @@ jobs: CompilerVer: 19 BuildType: Debug - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: clang-18 RelWithDebInfo [Ubuntu 24.04 Noble]: imageName: 'ubuntu-24.04' UBUNTU_CODE_NAME: noble @@ -176,114 +220,109 @@ jobs: CompilerVer: 18 BuildType: RelWithDebInfo - clang-18 Debug [Ubuntu 24.04 Noble]: - imageName: 'ubuntu-24.04' - UBUNTU_CODE_NAME: noble - Compiler: clang - CompilerVer: 18 - BuildType: Debug + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: + clang-18 Debug [Ubuntu 24.04 Noble]: + imageName: 'ubuntu-24.04' + UBUNTU_CODE_NAME: noble + Compiler: clang + CompilerVer: 18 + BuildType: Debug - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: clang-17 RelWithDebInfo [Ubuntu 22.04 Jammy]: imageName: 'ubuntu-22.04' Compiler: clang CompilerVer: 17 BuildType: RelWithDebInfo - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: clang-17 Debug [Ubuntu 22.04 Jammy]: imageName: 'ubuntu-22.04' Compiler: clang CompilerVer: 17 BuildType: Debug - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: clang-16 RelWithDebInfo [Ubuntu 22.04 Jammy]: imageName: 'ubuntu-22.04' Compiler: clang CompilerVer: 16 BuildType: RelWithDebInfo - clang-16 Debug [Ubuntu 22.04 Jammy]: - imageName: 'ubuntu-22.04' - Compiler: clang - CompilerVer: 16 - BuildType: Debug + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: + clang-16 Debug [Ubuntu 22.04 Jammy]: + imageName: 'ubuntu-22.04' + Compiler: clang + CompilerVer: 16 + BuildType: Debug # gcc-11 and newer compilers - gcc-16 RelWithDebInfo [Ubuntu 24.04 Noble]: - imageName: 'ubuntu-24.04' - Compiler: gcc - CompilerVer: 16 - BuildType: RelWithDebInfo - - gcc-16 Debug [Ubuntu 24.04 Noble]: - imageName: 'ubuntu-24.04' - Compiler: gcc - CompilerVer: 16 - BuildType: Debug - - gcc-15 RelWithDebInfo [Ubuntu 24.04 Noble]: - imageName: 'ubuntu-24.04' - Compiler: gcc - CompilerVer: 15 - BuildType: RelWithDebInfo + # (gcc-16 RelWithDebInfo and gcc-16 Debug are default configs, listed above) + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: + gcc-15 RelWithDebInfo [Ubuntu 24.04 Noble]: + imageName: 'ubuntu-24.04' + Compiler: gcc + CompilerVer: 15 + BuildType: RelWithDebInfo - gcc-15 Debug [Ubuntu 24.04 Noble]: - imageName: 'ubuntu-24.04' - Compiler: gcc - CompilerVer: 15 - BuildType: Debug + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: + gcc-15 Debug [Ubuntu 24.04 Noble]: + imageName: 'ubuntu-24.04' + Compiler: gcc + CompilerVer: 15 + BuildType: Debug - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: gcc-14 RelWithDebInfo [Ubuntu 22.04 Jammy]: imageName: 'ubuntu-22.04' Compiler: gcc CompilerVer: 14 BuildType: RelWithDebInfo - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: gcc-14 Debug [Ubuntu 22.04 Jammy]: imageName: 'ubuntu-22.04' Compiler: gcc CompilerVer: 14 BuildType: Debug - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: gcc-13 RelWithDebInfo [Ubuntu 22.04 Jammy]: imageName: 'ubuntu-22.04' Compiler: gcc CompilerVer: 13 BuildType: RelWithDebInfo - gcc-13 Debug [Ubuntu 22.04 Jammy]: - imageName: 'ubuntu-22.04' - Compiler: gcc - CompilerVer: 13 - BuildType: Debug + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: + gcc-13 Debug [Ubuntu 22.04 Jammy]: + imageName: 'ubuntu-22.04' + Compiler: gcc + CompilerVer: 13 + BuildType: Debug - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: gcc-12 RelWithDebInfo [Ubuntu 22.04 Jammy]: imageName: 'ubuntu-22.04' Compiler: gcc CompilerVer: 12 BuildType: RelWithDebInfo - gcc-12 Debug [Ubuntu 22.04 Jammy]: - imageName: 'ubuntu-22.04' - Compiler: gcc - CompilerVer: 12 - BuildType: Debug + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: + gcc-12 Debug [Ubuntu 22.04 Jammy]: + imageName: 'ubuntu-22.04' + Compiler: gcc + CompilerVer: 12 + BuildType: Debug - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: gcc-11 RelWithDebInfo [Ubuntu 22.04 Jammy]: imageName: 'ubuntu-22.04' Compiler: gcc CompilerVer: 11 BuildType: RelWithDebInfo - ${{ if or(ne(variables['Build.Reason'], 'PullRequest'), contains(variables['Build.SourceBranchName'], 'fullci')) }}: + ${{ if or(parameters.fullCI, eq(variables['Build.Reason'], 'Schedule')) }}: gcc-11 Debug [Ubuntu 22.04 Jammy]: imageName: 'ubuntu-22.04' Compiler: gcc From a35903462e5d49dc3401a00e5d4b4a2b1b16af8b Mon Sep 17 00:00:00 2001 From: Jaideep Karande Date: Fri, 26 Jun 2026 15:33:16 +0530 Subject: [PATCH 30/32] PXC-5220 [trunk]: Fix mysql_reconnect leaking options extension on charset error MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit == Root Cause == mysql_reconnect() shallow-copies the caller's MYSQL options struct into a stack-allocated tmp_mysql before calling mysql_real_connect(): tmp_mysql.options = mysql->options; If mysql->options.extension was NULL at that point (e.g. a connection created without any extended options), the shallow copy left tmp_mysql.options.extension = NULL as well. mysql_real_connect() calls ENSURE_EXTENSIONS_PRESENT(&mysql->options) near its start (client.cc:6607). When extension is NULL the macro allocates a new st_mysql_options_extention (232 bytes via calloc / my_raw_malloc) and stores it in tmp_mysql.options.extension. If mysql_set_character_set() subsequently fails (lines 7595-7605), the error path executes: memset(&tmp_mysql.options, 0, sizeof(tmp_mysql.options)); mysql_close(&tmp_mysql); The memset zeroes tmp_mysql.options.extension to NULL *without freeing it*. mysql_close() then skips the extension cleanup because the pointer is already zero. The 232-byte allocation is permanently lost, along with the connect_attributes map (My_hash) that may have been allocated inside it (an additional 744 bytes across 7 blocks in the observed case). The same code path is reached during the inner mysql_reconnect() call that occurs when mysql_close(&tmp_mysql) sends COM_QUIT and cli_advanced_command() triggers a second reconnect attempt because the socket is already in an error state (client.cc:1434). == Valgrind Report (Build 816, PXC 9.7.1-1 RelWithDebInfo Ubuntu Noble) == 976 (232 direct, 744 indirect) bytes in 1 blocks are definitely lost at calloc (vgpreload_memcheck) by my_raw_malloc (my_malloc.cc:321) by my_internal_malloc (my_malloc.cc:371) by mysql_real_connect (client.cc:6607) <- ENSURE_EXTENSIONS_PRESENT by mysql_reconnect (client.cc:7583) by cli_advanced_command (client.cc:1434) by mysql_close (client.cc:8006) <- COM_QUIT path by mysql_reconnect (client.cc:7601) <- charset-error cleanup by connect_to_master_via_namespace (rpl_replica.cc:8792) by try_to_reconnect (rpl_replica.cc:5578) by handle_slave_io (rpl_replica.cc:5828) PID 116239. 232 bytes directly lost + 744 bytes indirectly lost (7 connect-attribute strings inside the extension). == Fix == Call ENSURE_EXTENSIONS_PRESENT(&mysql->options) before the shallow copy so that mysql->options.extension is guaranteed to be non-NULL: ENSURE_EXTENSIONS_PRESENT(&mysql->options); /* <-- new */ tmp_mysql.options = mysql->options; After this change: • tmp_mysql.options.extension always aliases mysql->options.extension (same pointer, not a new allocation). • ENSURE_EXTENSIONS_PRESENT inside mysql_real_connect is a no-op. • The memset in the charset-error path still zeroes the pointer in tmp_mysql.options, but the actual extension struct remains alive and reachable through mysql->options.extension — nothing is leaked. • mysql_close_free_options() called during normal mysql_close(mysql) will free the extension via the mysql pointer, as before. ENSURE_EXTENSIONS_PRESENT is already defined and used throughout client.cc; no new headers are required. == Failing Test (Build 816) == rpl.rpl_tlsv13 (PID 116239) == Developer Notes == * The leak only triggers when mysql->options.extension is NULL at the time of the reconnect. Connections that have had any extended option set (SSL, compression, connect-attrs, etc.) already have a non-NULL extension and are unaffected. * The inner reconnect via cli_advanced_command (client.cc:1434) bypasses the mysql->reconnect guard that mysql_close sets to false (client.cc: 8004) because line 1434 calls mysql_reconnect() unconditionally on any net_write_command failure that is not ER_NET_PACKET_TOO_LARGE. * The memset-before-mysql_close pattern at lines 7589 and 7600 is intentional (it prevents mysql_close_free_options from double-freeing strings that were shallow-copied from the original mysql handle). The fix does not change that pattern; it only ensures there is nothing new to leak. --- sql-common/client.cc | 5 +++++ 1 file changed, 5 insertions(+) diff --git a/sql-common/client.cc b/sql-common/client.cc index 29af4ffc11fa..86c521aa1c5c 100644 --- a/sql-common/client.cc +++ b/sql-common/client.cc @@ -7573,6 +7573,11 @@ bool mysql_reconnect(MYSQL *mysql) { } mysql_init(&tmp_mysql); mysql_close_free_options(&tmp_mysql); + /* Guarantee the extension struct exists before the shallow copy so that + mysql_real_connect's ENSURE_EXTENSIONS_PRESENT is a no-op and does not + allocate a new extension that would be leaked if mysql_set_character_set + fails (the error path does memset before mysql_close). */ + ENSURE_EXTENSIONS_PRESENT(&mysql->options); tmp_mysql.options = mysql->options; tmp_mysql.options.my_cnf_file = tmp_mysql.options.my_cnf_group = nullptr; #ifdef MYSQL_SERVER From d7cd68fd15d96368e507bef9568c905cda608c11 Mon Sep 17 00:00:00 2001 From: Jaideep Karande Date: Fri, 26 Jun 2026 15:32:27 +0530 Subject: [PATCH 31/32] PXC-5220 [trunk]: Fix SSL/socket resource leak in GCS XCom join retry path MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit == Root Cause == In Gcs_xcom_control::try_send_add_node_request_to_seeds(), a connection established by connect_to_peer() was only closed inside the if (!finalized && connected) block. If m_view_control->is_finalized() became true between the connect_to_peer() return and the condition check (a TOCTOU race during retry_do_join), the block was skipped entirely. free_connection() only calls free() on the connection_descriptor struct itself (node_connection.h:90) — it does NOT call SSL_free() or close the socket. The SSL object allocated by SSL_new() in timed_connect_ssl_msec() (xcom_network_provider_ssl_native_lib.cc:694), plus all memory allocated internally by SSL_connect() (session state, cipher context, etc.), was permanently lost. == Valgrind Report (Build 816, PXC 9.7.1-1 RelWithDebInfo Ubuntu Noble) == 96,664 (7,640 direct, 89,024 indirect) bytes in 1 blocks are definitely lost at malloc (vgpreload_memcheck) by CRYPTO_zalloc by SSL_new (libssl.so.3) by timed_connect_ssl_msec (xcom_network_provider_ssl_native_lib.cc:694) by Xcom_network_provider::open_connection (xcom_network_provider.cc:332) by Network_provider_manager::open_xcom_connection (network_provider_manager.cc:241) by Gcs_xcom_control::connect_to_peer (gcs_xcom_control_interface.cc:623) by Gcs_xcom_control::try_send_add_node_request_to_seeds(gcs_xcom_control_interface.cc:563) by Gcs_xcom_control::send_add_node_request (gcs_xcom_control_interface.cc:543) by Gcs_xcom_control::retry_do_join (gcs_xcom_control_interface.cc:477) by Gcs_xcom_control::do_join (gcs_xcom_control_interface.cc:291) Observed in 9 of 11 affected mysqld PIDs (PIDs 16098, 164923, 61642 …). Each leak is 7,640 bytes direct + ~89–98 KB indirect per occurrence. == Fix == Add an else-if branch that calls xcom_client_close_connection(con) whenever connected == true but finalized == true. This ensures the SSL object and socket are always released exactly once: • !finalized && connected → existing path: add_node + close • finalized && connected → new path: close only • !connected → close already done inside connect_to_peer (disable_nagle failure path) or ssl_fd is null (connect failure); nothing to do == Failing Tests (Build 816) == group_replication.gr_ssl_options group_replication.gr_ssl_tls13_runtime_valid_configuration group_replication.gr_recovery_tlsv13_* group_replication.gr_rejoin_bootstrap group_replication.gr_rejoin_no_bootstrap group_replication.gr_clone_integration_* group_replication.gr_acf_receiver_* group_replication.gr_flush_logs group_replication.gr_primary_mode_group_operations_22_1 group_replication.gr_reset_slave_channel == Developer Notes == * free_connection() (node_connection.h:88) is intentionally a bare free() — it does not own the SSL or socket lifetime. Only xcom_client_close_connection() / close_xcom_connection() drives the provider's close_connection() which calls ssl_free_con() -> SSL_free(). * The race is narrow but reproducible under Valgrind (slow process) or high-load retry scenarios where is_finalized() transitions while the SSL handshake is in progress. * No behaviour change for the normal path (not finalized): the existing xcom_client_close_connection() call inside the if-block is untouched. --- .../src/bindings/xcom/gcs_xcom_control_interface.cc | 4 ++++ 1 file changed, 4 insertions(+) diff --git a/plugin/group_replication/libmysqlgcs/src/bindings/xcom/gcs_xcom_control_interface.cc b/plugin/group_replication/libmysqlgcs/src/bindings/xcom/gcs_xcom_control_interface.cc index 1149c61d0c33..ad709b694e72 100644 --- a/plugin/group_replication/libmysqlgcs/src/bindings/xcom/gcs_xcom_control_interface.cc +++ b/plugin/group_replication/libmysqlgcs/src/bindings/xcom/gcs_xcom_control_interface.cc @@ -593,6 +593,10 @@ bool Gcs_xcom_control::try_send_add_node_request_to_seeds( In this case, we continue the loop and try again using the next peer. */ if (xcom_will_process) add_node_accepted = true; + } else if (connected) { + /* GCS was finalized while we were connecting; close the connection to + free the SSL object and socket so they are not leaked. */ + m_xcom_proxy->xcom_client_close_connection(con); } free_connection(con); From 192b2a10fa785d9cd881e50f2d90b5f73d46dfaa Mon Sep 17 00:00:00 2001 From: Martin Hansson Date: Wed, 27 May 2026 08:47:00 +0000 Subject: [PATCH 32/32] PS-11264: Vector index support in Data Dictionary Vector indexes have the type (algorithm) SE_SPECIFIC, and we add a column option in the data dictionary saying `vector_index=1;` which gets picked up by dedicated code in the data dictionary and the handler part of InnoDB. In the SQL layer, the vector index is very much a thing; there is an `HA_KEY_ALG_VECTOR`, an `HA_VECTOR` and a `KEYTYPE_VECTOR`. Extra SQL is added to display the type of a vector index as VECTOR rather than SE_SPECIFIC. --- include/my_base.h | 14 +++-- share/messages_to_clients.txt | 9 +++ sql/create_field.cc | 15 +++-- sql/dd/dd_table.cc | 9 +++ sql/dd/impl/types/column_impl.cc | 16 ++++-- sql/dd/info_schema/show.cc | 46 ++++++++++++++- sql/dd_table_share.cc | 36 ++++++++++++ sql/field.cc | 4 +- sql/handler.h | 27 ++++----- sql/key_spec.h | 3 +- sql/sql_show.cc | 23 +++++++- sql/sql_table.cc | 66 ++++++++++++++++++++-- storage/innobase/btr/btr0btr.cc | 3 +- storage/innobase/dict/dict0crea.cc | 4 +- storage/innobase/dict/dict0dd.cc | 17 ++++-- storage/innobase/dict/dict0dict.cc | 42 +++++++++++++- storage/innobase/handler/ha_innodb.cc | 80 +++++++++++++++++++++------ storage/innobase/include/dict0dict.ic | 16 ++++++ storage/innobase/include/dict0mem.h | 14 +++++ storage/innobase/include/dict0mem.ic | 1 + storage/temptable/src/handler.cc | 1 + storage/temptable/src/table.cc | 1 + 22 files changed, 380 insertions(+), 67 deletions(-) diff --git a/include/my_base.h b/include/my_base.h index 66a4b1e556b7..df03e746f7c6 100644 --- a/include/my_base.h +++ b/include/my_base.h @@ -105,10 +105,11 @@ enum ha_key_alg { SEs default algorithm for keys in mysql_prepare_create_table(). */ HA_KEY_ALG_SE_SPECIFIC = 0, - HA_KEY_ALG_BTREE = 1, /* B-tree. */ - HA_KEY_ALG_RTREE = 2, /* R-tree, for spatial searches */ - HA_KEY_ALG_HASH = 3, /* HASH keys (HEAP, NDB). */ - HA_KEY_ALG_FULLTEXT = 4 /* FULLTEXT. */ + HA_KEY_ALG_BTREE = 1, /* B-tree. */ + HA_KEY_ALG_RTREE = 2, /* R-tree, for spatial searches */ + HA_KEY_ALG_HASH = 3, /* HASH keys (HEAP, NDB). */ + HA_KEY_ALG_FULLTEXT = 4, /* FULLTEXT. */ + HA_KEY_ALG_VECTOR = 5, /* VECTOR. */ }; /* Storage media types */ @@ -521,11 +522,14 @@ enum ha_base_keytype { #define HA_USES_COMMENT (1 << 12) /** Key was automatically created to support Foreign Key constraint. */ #define HA_GENERATED_KEY (1 << 13) +/** Vector key (Percona). */ +#define HA_VECTOR (1 << 30) /* The combination of the above can be used for key type comparison. */ #define HA_KEYFLAG_MASK \ (HA_NOSAME | HA_PACK_KEY | HA_AUTO_KEY | HA_BINARY_PACK_KEY | HA_FULLTEXT | \ - HA_UNIQUE_CHECK | HA_SPATIAL | HA_NULL_ARE_EQUAL | HA_GENERATED_KEY) + HA_UNIQUE_CHECK | HA_SPATIAL | HA_NULL_ARE_EQUAL | HA_GENERATED_KEY | \ + HA_VECTOR) /** Fulltext index uses [pre]parser */ #define HA_USES_PARSER (1 << 14) diff --git a/share/messages_to_clients.txt b/share/messages_to_clients.txt index e8a5df1a936d..4f759bf41239 100644 --- a/share/messages_to_clients.txt +++ b/share/messages_to_clients.txt @@ -11112,6 +11112,15 @@ ER_LOG_NAME_NOT_MATCHING_SEC_LOG_PATH_CLIENT # Start of Percona Server 8.4/9.7 error messages to be sent to client # +ER_VECTOR_INDEX_NEEDS_PK + eng "Vector index can only be created in tables with a BIGINT UNSIGNED primary key." + +ER_ONLY_SINGLE_VECTOR_INDEX_ALLOWED + eng "A table can have at most one vector index." + +ER_TABLE_CANT_HANDLE_INDEX + eng "The used table type doesn't support %.20s indexes" + start-error-number 7100 # diff --git a/sql/create_field.cc b/sql/create_field.cc index 168d5293cd94..ac61abbaeeca 100644 --- a/sql/create_field.cc +++ b/sql/create_field.cc @@ -23,6 +23,7 @@ #include "sql/create_field.h" +#include "field_types.h" #include "m_string.h" #include "mysql/strings/dtoa.h" #include "sql-common/my_decimal.h" @@ -200,9 +201,10 @@ bool Create_field::init( const LEX_CSTRING *fld_comment, const char *fld_change, List *fld_interval_list, const CHARSET_INFO *fld_charset, bool has_explicit_collation, uint fld_geom_type, - const LEX_CSTRING *fld_zip_dict_name, Value_generator *fld_gcol_info, Value_generator *fld_default_val_expr, - LEX_CSTRING fld_masking_policy, std::optional srid, - dd::Column::enum_hidden_type hidden, bool is_array_arg) { + const LEX_CSTRING *fld_zip_dict_name, Value_generator *fld_gcol_info, + Value_generator *fld_default_val_expr, LEX_CSTRING fld_masking_policy, + std::optional srid, dd::Column::enum_hidden_type hidden, + bool is_array_arg) { uint sign_len, allowed_type_modifier = 0; ulong max_field_charlength = MAX_FIELD_CHARLENGTH; @@ -780,7 +782,8 @@ size_t Create_field::key_length() const { case MYSQL_TYPE_JSON: case MYSQL_TYPE_VAR_STRING: case MYSQL_TYPE_STRING: - case MYSQL_TYPE_VARCHAR: { + case MYSQL_TYPE_VARCHAR: + case MYSQL_TYPE_VECTOR: { return std::min(max_display_width_in_bytes(), static_cast(MAX_FIELD_BLOBLENGTH)); } @@ -794,10 +797,6 @@ size_t Create_field::key_length() const { } return pack_length() + (max_display_width_in_bytes() & 7 ? 1 : 0); } - /* LCOV_EXCL_START */ - case MYSQL_TYPE_VECTOR: - assert(false); // Key on VECTOR type column is not supported. - /* LCOV_EXCL_STOP */ default: { return pack_length(is_array); } diff --git a/sql/dd/dd_table.cc b/sql/dd/dd_table.cc index 8a14c56e8599..e3fc670c6512 100644 --- a/sql/dd/dd_table.cc +++ b/sql/dd/dd_table.cc @@ -735,6 +735,10 @@ bool fill_dd_columns_from_create_fields(THD *thd, dd::Abstract_table *tab_obj, col_options->set("is_array", true); } + if (field.sql_type == MYSQL_TYPE_VECTOR) { + col_options->set("vector_index", true); + } + // // Write intervals // @@ -825,6 +829,9 @@ static dd::Index::enum_index_algorithm dd_get_new_index_algorithm_type( case HA_KEY_ALG_FULLTEXT: return dd::Index::IA_FULLTEXT; + + case HA_KEY_ALG_VECTOR: + return dd::Index::IA_SE_SPECIFIC; } /* purecov: begin deadcode */ @@ -836,6 +843,8 @@ static dd::Index::enum_index_algorithm dd_get_new_index_algorithm_type( } static dd::Index::enum_index_type dd_get_new_index_type(const KEY *key) { + if (key->flags & HA_VECTOR) return dd::Index::IT_MULTIPLE; + if (key->flags & HA_FULLTEXT) return dd::Index::IT_FULLTEXT; if (key->flags & HA_SPATIAL) return dd::Index::IT_SPATIAL; diff --git a/sql/dd/impl/types/column_impl.cc b/sql/dd/impl/types/column_impl.cc index 34cb323def7b..1b6669ed6bc4 100644 --- a/sql/dd/impl/types/column_impl.cc +++ b/sql/dd/impl/types/column_impl.cc @@ -64,11 +64,17 @@ class Sdi_rcontext; class Sdi_wcontext; static const std::set default_valid_option_keys = { - "column_format", "geom_type", - "interval_count", "not_secondary", - "storage", "treat_bit_as_char", "zip_dict_id", - "is_array", "gipk" /* generated implicit primary key column */, - "masking_policy"}; + "column_format", + "geom_type", + "interval_count", + "not_secondary", + "storage", + "treat_bit_as_char", + "zip_dict_id", + "is_array", + "gipk" /* generated implicit primary key column */, + "masking_policy", + "vector_index"}; /////////////////////////////////////////////////////////////////////////// // Column_impl implementation. diff --git a/sql/dd/info_schema/show.cc b/sql/dd/info_schema/show.cc index f37078e4fce8..94cc320d3552 100644 --- a/sql/dd/info_schema/show.cc +++ b/sql/dd/info_schema/show.cc @@ -868,6 +868,50 @@ Query_block *build_show_keys_query(const POS &pos, THD *thd, Select_lex_builder top_query(&pos, thd); + Item *index_type_item = + new (thd->mem_root) Item_field(pos, NullS, NullS, alias_type.str); + if (index_type_item == nullptr) return nullptr; + + Item *se_specific_item = new (thd->mem_root) + Item_string(STRING_WITH_LEN("SE_SPECIFIC"), system_charset_info); + if (se_specific_item == nullptr) return nullptr; + + Item *is_se_specific = + new (thd->mem_root) Item_func_eq(pos, index_type_item, se_specific_item); + if (is_se_specific == nullptr) return nullptr; + + Item *sub_part_item = + new (thd->mem_root) Item_field(pos, NullS, NullS, alias_sub_part.str); + if (sub_part_item == nullptr) return nullptr; + + Item *one_item = new (thd->mem_root) + Item_string(STRING_WITH_LEN("1"), system_charset_info); + if (one_item == nullptr) return nullptr; + + Item *is_single_sub_part = + new (thd->mem_root) Item_func_eq(pos, sub_part_item, one_item); + if (is_single_sub_part == nullptr) return nullptr; + + Item *vector_item = new (thd->mem_root) + Item_string(STRING_WITH_LEN("VECTOR"), system_charset_info); + if (vector_item == nullptr) return nullptr; + + Item *index_type_else_item = + new (thd->mem_root) Item_field(pos, NullS, NullS, alias_type.str); + if (index_type_else_item == nullptr) return nullptr; + + Item *index_type_if = new (thd->mem_root) + Item_func_if(pos, is_single_sub_part, vector_item, index_type_else_item); + if (index_type_if == nullptr) return nullptr; + + Item *index_type_default_item = + new (thd->mem_root) Item_field(pos, NullS, NullS, alias_type.str); + if (index_type_default_item == nullptr) return nullptr; + + Item *index_type_expr = new (thd->mem_root) + Item_func_if(pos, is_se_specific, index_type_if, index_type_default_item); + if (index_type_expr == nullptr) return nullptr; + // SELECT * FROM ... if (top_query.add_select_item(alias_table, alias_table) || top_query.add_select_item(alias_non_unique, alias_non_unique) || @@ -879,7 +923,7 @@ Query_block *build_show_keys_query(const POS &pos, THD *thd, top_query.add_select_item(alias_sub_part, alias_sub_part) || top_query.add_select_item(alias_packed, alias_packed) || top_query.add_select_item(alias_null, alias_null) || - top_query.add_select_item(alias_type, alias_type) || + top_query.add_select_expr(index_type_expr, alias_type) || top_query.add_select_item(alias_comment, alias_comment) || top_query.add_select_item(alias_index_comment, alias_index_comment) || top_query.add_select_item(alias_visible, alias_visible) || diff --git a/sql/dd_table_share.cc b/sql/dd_table_share.cc index c5504726c0a9..56bd874e29f6 100644 --- a/sql/dd_table_share.cc +++ b/sql/dd_table_share.cc @@ -235,6 +235,29 @@ static enum ha_key_alg dd_get_old_index_algorithm_type( return HA_KEY_ALG_SE_SPECIFIC; } +/** + Check whether any visible index element is marked as vector. + + @param[in] idx_obj Index metadata object. + + @return Whether any visible element belongs to a vector column. +*/ +static bool dd_index_has_vector_column(const dd::Index &idx_obj) { + for (const dd::Index_element *idx_elem : idx_obj.elements()) { + if (idx_elem->is_hidden()) continue; + + const dd::Properties &col_options = idx_elem->column().options(); + bool is_vector_column = false; + if (col_options.exists("vector_index") && + !col_options.get("vector_index", &is_vector_column) && + is_vector_column) { + return true; + } + } + + return false; +} + /* Check if the given key_part is suitable to be promoted as part of primary key. @@ -347,6 +370,8 @@ static bool prepare_share(THD *thd, TABLE_SHARE *share, share->key_info[key].algorithm == HA_KEY_ALG_FULLTEXT); assert(!(share->key_info[key].flags & HA_SPATIAL) || share->key_info[key].algorithm == HA_KEY_ALG_RTREE); + assert(!(share->key_info[key].flags & HA_VECTOR) || + share->key_info[key].algorithm == HA_KEY_ALG_VECTOR); if (primary_key >= MAX_KEY && (keyinfo->flags & HA_NOSAME)) { /* @@ -1385,6 +1410,8 @@ static bool fill_index_from_dd(THD *thd, TABLE_SHARE *share, keyinfo->algorithm = dd_get_old_index_algorithm_type(idx_obj->algorithm()); keyinfo->is_algorithm_explicit = idx_obj->is_algorithm_explicit(); + const bool has_vector_column = dd_index_has_vector_column(*idx_obj); + // Visibility keyinfo->is_visible = idx_obj->is_visible(); @@ -1395,6 +1422,11 @@ static bool fill_index_from_dd(THD *thd, TABLE_SHARE *share, if (!idx_ele->is_hidden()) keyinfo->user_defined_key_parts++; } + if (has_vector_column && keyinfo->user_defined_key_parts == 1) { + keyinfo->algorithm = HA_KEY_ALG_VECTOR; + keyinfo->is_algorithm_explicit = false; + } + // flags switch (idx_obj->type()) { case dd::Index::IT_MULTIPLE: @@ -1416,6 +1448,10 @@ static bool fill_index_from_dd(THD *thd, TABLE_SHARE *share, break; } + if (has_vector_column && keyinfo->user_defined_key_parts == 1) { + keyinfo->flags |= HA_VECTOR; + } + if (idx_obj->is_generated()) keyinfo->flags |= HA_GENERATED_KEY; /* diff --git a/sql/field.cc b/sql/field.cc index abc98b6f449c..6f969c1c5ddf 100644 --- a/sql/field.cc +++ b/sql/field.cc @@ -34,6 +34,7 @@ #include #include "decimal.h" +#include "field_types.h" #include "my_alloc.h" #include "my_byteorder.h" #include "my_compare.h" @@ -75,7 +76,7 @@ #include "sql/mysqld_cs.h" #include "sql/protocol.h" #include "sql/psi_memory_key.h" -#include "sql/spatial.h" // Geometry +#include "sql/spatial.h" // Geometry #include "sql/sql_base.h" #include "sql/sql_class.h" // THD #include "sql/sql_exception_handler.h" // handle_std_exception @@ -1656,6 +1657,7 @@ bool Field::type_can_have_key_part(enum enum_field_types type) { case MYSQL_TYPE_VAR_STRING: case MYSQL_TYPE_STRING: case MYSQL_TYPE_GEOMETRY: + case MYSQL_TYPE_VECTOR: return true; default: return false; diff --git a/sql/handler.h b/sql/handler.h index cdb1ac488eb4..9a7028e1e1db 100644 --- a/sql/handler.h +++ b/sql/handler.h @@ -535,6 +535,7 @@ enum class SelectExecutedIn : bool { kPrimaryEngine, kSecondaryEngine }; ANALYZE TABLE on it */ #define HA_ONLINE_ANALYZE (1LL << 56) +#define HA_CAN_VECTOR (1LL << 57) /* Bits in index_flags(index_number) for what you can do with index. @@ -7055,16 +7056,16 @@ class handler { for details. */ [[nodiscard]] int ha_fast_update(THD *thd, - mem_root_deque &update_fields, - mem_root_deque &update_values, - Item *conds); + mem_root_deque &update_fields, + mem_root_deque &update_values, + Item *conds); /** @brief Offload an upsert to the storage engine. See handler::upsert() for details. */ [[nodiscard]] int ha_upsert(THD *thd, mem_root_deque &update_fields, - mem_root_deque &update_values); + mem_root_deque &update_values); private: /** @@ -7087,11 +7088,11 @@ class handler { handler::ha_update_row(...) does not accept conditions. */ [[nodiscard]] virtual int fast_update(THD *thd [[maybe_unused]], - mem_root_deque &update_fields - [[maybe_unused]], - mem_root_deque &update_values - [[maybe_unused]], - Item *conds [[maybe_unused]]) { + mem_root_deque &update_fields + [[maybe_unused]], + mem_root_deque &update_values + [[maybe_unused]], + Item *conds [[maybe_unused]]) { return ENOTSUP; } @@ -7112,10 +7113,10 @@ class handler { @return an error if the insert should be terminated. */ [[nodiscard]] virtual int upsert(THD *thd [[maybe_unused]], - mem_root_deque &update_fields - [[maybe_unused]], - mem_root_deque &update_values - [[maybe_unused]]) { + mem_root_deque &update_fields + [[maybe_unused]], + mem_root_deque &update_values + [[maybe_unused]]) { return ENOTSUP; } diff --git a/sql/key_spec.h b/sql/key_spec.h index b58e0651b759..47abcff67da2 100644 --- a/sql/key_spec.h +++ b/sql/key_spec.h @@ -43,7 +43,8 @@ enum keytype { KEYTYPE_MULTIPLE = 2, KEYTYPE_FULLTEXT = 4, KEYTYPE_SPATIAL = 8, - KEYTYPE_FOREIGN = 16 + KEYTYPE_FOREIGN = 16, + KEYTYPE_VECTOR = 32, }; enum fk_option { diff --git a/sql/sql_show.cc b/sql/sql_show.cc index 52b03e9f785e..8f699cd6ca77 100644 --- a/sql/sql_show.cc +++ b/sql/sql_show.cc @@ -2791,6 +2791,8 @@ bool store_create_info(THD *thd, Table_ref *table_list, String *packet, packet->append(STRING_WITH_LEN("FULLTEXT KEY ")); else if (key_info->flags & HA_SPATIAL) packet->append(STRING_WITH_LEN("SPATIAL KEY ")); + else if (key_info->flags & HA_VECTOR) + packet->append(STRING_WITH_LEN("VECTOR KEY ")); else packet->append(STRING_WITH_LEN("KEY ")); @@ -2823,7 +2825,7 @@ bool store_create_info(THD *thd, Table_ref *table_list, String *packet, if (key_part->field && (key_part->length != table->field[key_part->fieldnr - 1]->key_length() && - !(key_info->flags & (HA_FULLTEXT | HA_SPATIAL)))) { + !(key_info->flags & (HA_FULLTEXT | HA_SPATIAL | HA_VECTOR)))) { packet->append_parenthesized((long)key_part->length / key_part->field->charset()->mbmaxlen); } @@ -5482,6 +5484,23 @@ static int fill_schema_engines(THD *thd, Table_ref *tables, Item *) { #define TMP_TABLE_KEYS_IS_VISIBLE 14 #define TMP_TABLE_KEYS_EXPRESSION 15 +/** + Detect vector indexes for SHOW INDEX output. + + @param[in] key_info index metadata from TABLE_SHARE + + @return Whether this is a vector index. +*/ +static bool is_show_index_vector_type(const KEY *key_info) { + if (key_info->flags & HA_VECTOR) return true; + + if (key_info->user_defined_key_parts != 1) return false; + + const KEY_PART_INFO *key_part = key_info->key_part; + return key_part != nullptr && key_part->field != nullptr && + key_part->field->real_type() == MYSQL_TYPE_VECTOR; +} + static int get_schema_tmp_table_keys_record(THD *thd, Table_ref *tables, TABLE *table, bool res, LEX_CSTRING, LEX_CSTRING table_name) { @@ -5570,6 +5589,8 @@ static int get_schema_tmp_table_keys_record(THD *thd, Table_ref *tables, // INDEX_TYPE if (key_info->flags & HA_SPATIAL) str = "SPATIAL"; + else if (is_show_index_vector_type(key_info)) + str = "VECTOR"; else { const ha_key_alg key_alg = key_info->algorithm; /* If index algorithm is implicit get SE default. */ diff --git a/sql/sql_table.cc b/sql/sql_table.cc index faaf57304faf..766e11c707d9 100644 --- a/sql/sql_table.cc +++ b/sql/sql_table.cc @@ -5175,8 +5175,8 @@ static bool prepare_key_column(THD *thd, HA_CREATE_INFO *create_info, return true; } - // VECTOR columns cannot be used as keys - if (sql_field->sql_type == MYSQL_TYPE_VECTOR) { + if (sql_field->sql_type == MYSQL_TYPE_VECTOR && + ((key_info->flags & HA_VECTOR) == 0)) { my_error(ER_NON_SCALAR_USED_AS_KEY, MYF(0), column->get_field_name()); return true; } @@ -5227,6 +5227,13 @@ static bool prepare_key_column(THD *thd, HA_CREATE_INFO *create_info, data prefix, ignoring column->length). */ column_length = is_blob(sql_field->sql_type); + } else if (key->type == KEYTYPE_VECTOR) { + // VECTOR indexes are only allowed on VECTOR columns. + if (sql_field->sql_type != MYSQL_TYPE_VECTOR) { + my_error(ER_UNKNOWN_ERROR, MYF(0)); + return true; + } + column_length = 1; // Dummy value. } else { switch (sql_field->sql_type) { case MYSQL_TYPE_GEOMETRY: @@ -5833,7 +5840,7 @@ static bool prepare_self_ref_fk_parent_key( for (const KEY *key = key_info_buffer; key < key_info_buffer + key_count; key++) { // We can't use FULLTEXT or SPATIAL indexes. - if (key->flags & (HA_FULLTEXT | HA_SPATIAL)) continue; + if (key->flags & (HA_FULLTEXT | HA_SPATIAL | HA_VECTOR)) continue; if (hton->foreign_keys_flags & HTON_FKS_NEED_DIFFERENT_PARENT_AND_SUPPORTING_KEYS) { @@ -6036,7 +6043,7 @@ static const KEY *find_fk_supporting_key(handlerton *hton, for (const KEY *key = key_info_buffer; key < key_info_buffer + key_count; key++) { // We can't use FULLTEXT or SPATIAL indexes. - if (key->flags & (HA_FULLTEXT | HA_SPATIAL)) continue; + if (key->flags & (HA_FULLTEXT | HA_SPATIAL | HA_VECTOR)) continue; if (key->algorithm == HA_KEY_ALG_HASH) { if (hton->foreign_keys_flags & HTON_FKS_WITH_SUPPORTING_HASH_KEYS) { @@ -7686,6 +7693,17 @@ static bool prepare_key( switch (static_cast(key->type)) { case KEYTYPE_MULTIPLE: break; + case KEYTYPE_VECTOR: + if (!(file->ha_table_flags() & HA_CAN_VECTOR)) { + my_error(ER_TABLE_CANT_HANDLE_INDEX, MYF(0), "vector"); + return true; + } + if (key->columns.size() != 1) { + my_error(ER_TOO_MANY_KEY_PARTS, MYF(0), 1); + return true; + } + key_info->flags |= HA_VECTOR; + break; case KEYTYPE_FULLTEXT: if (!(file->ha_table_flags() & HA_CAN_FULLTEXT)) { my_error(ER_TABLE_CANT_HANDLE_FT, MYF(0)); @@ -7743,6 +7761,9 @@ static bool prepare_key( } else if (key_info->flags & HA_FULLTEXT) { assert(!key->key_create_info.is_algorithm_explicit); key_info->algorithm = HA_KEY_ALG_FULLTEXT; + } else if (key_info->flags & HA_VECTOR) { + assert(!key->key_create_info.is_algorithm_explicit); + key_info->algorithm = HA_KEY_ALG_VECTOR; } else { if (key->key_create_info.is_algorithm_explicit) { if (key->key_create_info.algorithm != HA_KEY_ALG_RTREE) { @@ -8644,6 +8665,7 @@ bool mysql_prepare_create_table( uint key_number = 0; bool primary_key = false; + uint vector_key_number = 0; // First prepare non-foreign keys so that they are ready when // we prepare foreign keys. @@ -8660,6 +8682,14 @@ bool mysql_prepare_create_table( primary_key = true; } + if (key->type == KEYTYPE_VECTOR) { + if (vector_key_number) { + my_error(ER_ONLY_SINGLE_VECTOR_INDEX_ALLOWED, MYF(0)); + return true; + } + ++vector_key_number; + } + if (key->type != KEYTYPE_FOREIGN) { if (prepare_key(thd, error_schema_name, error_table_name, create_info, &alter_info->create_list, key, key_info_buffer, key_info, @@ -8678,6 +8708,12 @@ bool mysql_prepare_create_table( } } + // We allow VECTOR keys only with tables with PK + if (!primary_key && vector_key_number) { + my_error(ER_VECTOR_INDEX_NEEDS_PK, MYF(0)); + return true; + } + /* At this point all KEY objects are for indexes are fully constructed. So we can check for duplicate indexes for keys for which it was requested. @@ -8724,6 +8760,26 @@ bool mysql_prepare_create_table( /* Sort keys in optimized order */ std::sort(*key_info_buffer, *key_info_buffer + *key_count, sort_keys()); + // We allow VECTOR indexes only on tables with BIGINT UNSIGNED PKs. + if (vector_key_number) { + assert(primary_key); + const KEY &primary_info = *key_info_buffer[0]; + + if (primary_info.actual_key_parts > 1) { + my_error(ER_VECTOR_INDEX_NEEDS_PK, MYF(0)); + return true; + } + + for (it.rewind(), field_no = 0; (sql_field = it++); field_no++) { + if (field_no >= primary_info.key_part[0].fieldnr) break; + } + assert(sql_field); + if (sql_field->sql_type != MYSQL_TYPE_LONGLONG || !sql_field->is_unsigned) { + my_error(ER_VECTOR_INDEX_NEEDS_PK, MYF(0)); + return true; + } + } + /* Normal keys are done, now prepare foreign keys. @@ -16319,6 +16375,8 @@ bool prepare_fields_and_keys(THD *thd, const dd::Table *src_table, TABLE *table, key_type = KEYTYPE_UNIQUE; } else if (key_info->flags & HA_FULLTEXT) key_type = KEYTYPE_FULLTEXT; + else if (key_info->flags & HA_VECTOR) + key_type = KEYTYPE_VECTOR; else key_type = KEYTYPE_MULTIPLE; diff --git a/storage/innobase/btr/btr0btr.cc b/storage/innobase/btr/btr0btr.cc index 3c06ee5adfb1..7a9e0fdf1bd2 100644 --- a/storage/innobase/btr/btr0btr.cc +++ b/storage/innobase/btr/btr0btr.cc @@ -4655,7 +4655,8 @@ bool btr_validate_index( /* Full Text index are implemented by auxiliary tables, not the B-tree */ - if (dict_index_is_online_ddl(index) || (index->type & DICT_FTS)) { + if (dict_index_is_online_ddl(index) || + ((index->type & DICT_FTS) || dict_index_is_vector(index))) { return (true); } diff --git a/storage/innobase/dict/dict0crea.cc b/storage/innobase/dict/dict0crea.cc index 182bf0051a85..d2c85793e473 100644 --- a/storage/innobase/dict/dict0crea.cc +++ b/storage/innobase/dict/dict0crea.cc @@ -438,8 +438,8 @@ dberr_t dict_create_index_tree_in_mem(dict_index_t *index, trx_t *trx) { DBUG_EXECUTE_IF("ib_dict_create_index_tree_fail", return (DB_OUT_OF_MEMORY);); - if (index->type == DICT_FTS) { - /* FTS index does not need an index tree */ + if ((index->type & DICT_FTS) || dict_index_is_vector(index)) { + /* FTS nor vector index do not need an index tree */ return (DB_SUCCESS); } diff --git a/storage/innobase/dict/dict0dd.cc b/storage/innobase/dict/dict0dd.cc index f77448568965..63798ec908d4 100644 --- a/storage/innobase/dict/dict0dd.cc +++ b/storage/innobase/dict/dict0dd.cc @@ -1979,7 +1979,7 @@ void dd_visit_keys_with_too_long_parts( std::function visitor) { for (uint key_num = 0; key_num < table->s->keys; key_num++) { const KEY &key = table->key_info[key_num]; - if (!(key.flags & (HA_SPATIAL | HA_FULLTEXT))) { + if (!(key.flags & (HA_SPATIAL | HA_FULLTEXT | HA_VECTOR))) { for (unsigned i = 0; i < key.user_defined_key_parts; i++) { const KEY_PART_INFO *key_part = &key.key_part[i]; if (max_part_len < key_part->length) { @@ -2909,7 +2909,7 @@ MY_COMPILER_DIAGNOSTIC_POP() */ static inline uint16_t get_index_prefix_len(const KEY &key, const KEY_PART_INFO *key_part) { - if (key.flags & (HA_SPATIAL | HA_FULLTEXT)) { + if (key.flags & (HA_SPATIAL | HA_FULLTEXT | HA_VECTOR)) { return 0; } @@ -2949,6 +2949,7 @@ template const dict_index_t *dd_find_index( uint key_num) { const KEY &key = form->key_info[key_num]; ulint type = 0; + bool is_vector = false; unsigned n_fields = key.user_defined_key_parts; unsigned n_uniq = n_fields; @@ -2969,6 +2970,10 @@ template const dict_index_t *dd_find_index( ut_ad(!table->is_intrinsic()); type = DICT_FTS; n_uniq = 0; + } else if (key.flags & HA_VECTOR) { + ut_ad(!table->is_intrinsic()); + is_vector = true; + n_uniq = 0; } else if (key_num == form->primary_key) { ut_ad(key.flags & HA_NOSAME); ut_ad(n_uniq > 0); @@ -2977,11 +2982,13 @@ template const dict_index_t *dd_find_index( type = (key.flags & HA_NOSAME) ? DICT_UNIQUE : 0; } - ut_ad(!!(type & DICT_FTS) == (n_uniq == 0)); + ut_ad((!!(type & DICT_FTS) || is_vector) == (n_uniq == 0)); dict_index_t *index = dict_mem_index_create(table->name.m_name, key.name, 0, type, n_fields); + index->is_vector_index = is_vector; + index->n_uniq = n_uniq; DBUG_EXECUTE_IF("ib_create_table_fail_at_create_index", @@ -5206,8 +5213,8 @@ dict_table_t *dd_open_table_one(dd::cache::Dictionary_client *client, } ut_ad(root > 1); - ut_ad(index->type & DICT_FTS || root != FIL_NULL || - dict_table_is_discarded(m_table)); + ut_ad((index->type & DICT_FTS) || dict_index_is_vector(index) || + root != FIL_NULL || dict_table_is_discarded(m_table)); ut_ad(id != 0); index->page = root; index->space = sid; diff --git a/storage/innobase/dict/dict0dict.cc b/storage/innobase/dict/dict0dict.cc index 6710d5abff14..56344f30c2ee 100644 --- a/storage/innobase/dict/dict0dict.cc +++ b/storage/innobase/dict/dict0dict.cc @@ -218,6 +218,9 @@ static dict_index_t *dict_index_build_internal_fts( dict_table_t *table, /*!< in: table */ dict_index_t *index); /*!< in: user representation of an FTS index */ +static dict_index_t *dict_index_build_internal_vec(dict_table_t *table, + dict_index_t *index); + /** Removes an index from the dictionary cache. */ static void dict_index_remove_from_cache_low( dict_table_t *table, /*!< in/out: table */ @@ -2217,7 +2220,7 @@ static bool dict_index_too_big_for_tree(const dict_table_t *table, const dict_index_t *new_index) { /* FTS index consists of auxiliary tables, they shall be excluded from index row size check */ - if (new_index->type & DICT_FTS) { + if ((new_index->type & DICT_FTS) || dict_index_is_vector(new_index)) { return (false); } @@ -2446,6 +2449,8 @@ dberr_t dict_index_add_to_cache_w_vcol(dict_table_t *table, dict_index_t *index, if (index->type == DICT_FTS) { new_index = dict_index_build_internal_fts(table, index); + } else if (dict_index_is_vector(index)) { + new_index = dict_index_build_internal_vec(table, index); } else if (index->is_clustered()) { new_index = dict_index_build_internal_clust(table, index); } else { @@ -3284,6 +3289,37 @@ static dict_index_t *dict_index_build_internal_fts( return (new_index); } + +static dict_index_t *dict_index_build_internal_vec( + dict_table_t *table, /*!< in: table */ + dict_index_t *index) /*!< in: user representation of a vector index */ +{ + ut_ad(table && index); + ut_ad(dict_index_is_vector(index)); + ut_ad(!dict_sys_mutex_own()); + ut_ad(table->magic_n == DICT_TABLE_MAGIC_N); + + /* Create a new index */ + auto new_index = + dict_mem_index_create(table->name.m_name, index->name, index->space, + index->type, index->n_fields); + new_index->is_vector_index = index->is_vector_index; + + /* Copy other relevant data from the old index struct to the new + struct: it inherits the values */ + + new_index->n_user_defined_cols = index->n_fields; + + new_index->id = index->id; + + /* Copy fields from index to new_index */ + dict_index_copy(new_index, index, table, 0, index->n_fields); + + new_index->n_uniq = 0; + new_index->cached = true; + + return (new_index); +} /*====================== FOREIGN KEY PROCESSING ========================*/ /** Checks if a table is referenced by foreign keys. @@ -3371,7 +3407,8 @@ NOT NULL */ while (index != nullptr) { if (types_idx != index && !(index->type & DICT_FTS) && - !dict_index_is_spatial(index) && !index->to_be_dropped && + !dict_index_is_vector(index) && !dict_index_is_spatial(index) && + !index->to_be_dropped && (!(index->uncommitted && ((index->online_status == ONLINE_INDEX_ABORTED_DROPPED) || (index->online_status == ONLINE_INDEX_ABORTED)))) && @@ -3628,6 +3665,7 @@ bool dict_index_check_search_tuple( ut_ad(index->page >= FSP_FIRST_INODE_PAGE_NO); ut_ad(dtuple_check_typed(tuple)); ut_ad(!(index->type & DICT_FTS)); + ut_ad(!dict_index_is_vector(index)); return true; } #endif /* UNIV_DEBUG */ diff --git a/storage/innobase/handler/ha_innodb.cc b/storage/innobase/handler/ha_innodb.cc index a2be3f2f25ef..5b972ff07f1c 100644 --- a/storage/innobase/handler/ha_innodb.cc +++ b/storage/innobase/handler/ha_innodb.cc @@ -3232,7 +3232,7 @@ ha_innobase::ha_innobase(handlerton *hton, TABLE_SHARE *table_arg) HA_ATTACHABLE_TRX_COMPATIBLE | HA_CAN_INDEX_VIRTUAL_GENERATED_COLUMN | HA_DESCENDING_INDEX | HA_MULTI_VALUED_KEY_SUPPORT | HA_BLOB_PARTIAL_UPDATE | HA_SUPPORTS_GEOGRAPHIC_GEOMETRY_COLUMN | - HA_SUPPORTS_DEFAULT_EXPRESSION | HA_ONLINE_ANALYZE), + HA_SUPPORTS_DEFAULT_EXPRESSION | HA_ONLINE_ANALYZE | HA_CAN_VECTOR), m_start_of_scan(), m_stored_select_lock_type(LOCK_NONE_UNSET), m_mysql_has_locked() {} @@ -5751,13 +5751,13 @@ static int innodb_init(void *p) { innobase_hton->lock_hton_log = innobase_lock_hton_log; innobase_hton->unlock_hton_log = innobase_unlock_hton_log; innobase_hton->collect_hton_log_info = innobase_collect_hton_log_info; - innobase_hton->flags = HTON_SUPPORTS_EXTENDED_KEYS | - HTON_SUPPORTS_FOREIGN_KEYS | HTON_SUPPORTS_ATOMIC_DDL | - HTON_CAN_RECREATE | HTON_SUPPORTS_SECONDARY_ENGINE | - HTON_SUPPORTS_TABLE_ENCRYPTION | - HTON_SUPPORTS_GENERATED_INVISIBLE_PK | - HTON_SUPPORTS_BULK_LOAD | HTON_SUPPORTS_SQL_FK | - HTON_SUPPORTS_ONLINE_BACKUPS | HTON_SUPPORTS_COMPRESSED_COLUMNS; + innobase_hton->flags = + HTON_SUPPORTS_EXTENDED_KEYS | HTON_SUPPORTS_FOREIGN_KEYS | + HTON_SUPPORTS_ATOMIC_DDL | HTON_CAN_RECREATE | + HTON_SUPPORTS_SECONDARY_ENGINE | HTON_SUPPORTS_TABLE_ENCRYPTION | + HTON_SUPPORTS_GENERATED_INVISIBLE_PK | HTON_SUPPORTS_BULK_LOAD | + HTON_SUPPORTS_SQL_FK | HTON_SUPPORTS_ONLINE_BACKUPS | + HTON_SUPPORTS_COMPRESSED_COLUMNS; // TODO(WL9440): to be enabled when distance scan is implemented in innodb. //| HTON_SUPPORTS_DISTANCE_SCAN; @@ -8304,7 +8304,7 @@ int ha_innobase::open(const char *name, int, uint open_flags, dict_table_autoinc_unlock(ib_table); } - /* Set plugin parser for fulltext index */ + /* Set plugin parser for fulltext index / handle vector index. */ for (uint i = 0; i < table->s->keys; i++) { if (table->key_info[i].flags & HA_USES_PARSER) { dict_index_t *index = innobase_get_index(i); @@ -11020,7 +11020,7 @@ int ha_innobase::index_read( : HA_ERR_TABLE_DEF_CHANGED; } - if (index->type & DICT_FTS) { + if ((index->type & DICT_FTS) || dict_index_is_vector(index)) { return HA_ERR_KEY_NOT_FOUND; } @@ -12872,6 +12872,7 @@ inline int create_index( index = dict_mem_index_create(table_name, key->name, 0, ind_type, key->user_defined_key_parts); + index->is_vector_index = (key->flags & HA_VECTOR); innodb_session_t *&priv = thd_to_innodb_session(trx->mysql_thd); dict_table_t *handler = priv->lookup_table_handler(table_name); @@ -12965,6 +12966,7 @@ inline int create_index( } ut_ad(key->flags & HA_FULLTEXT || !(index->type & DICT_FTS)); + ut_ad((key->flags & HA_VECTOR) || !dict_index_is_vector(index)); multi_val_idx = ((index->type & DICT_MULTI_VALUE) == DICT_MULTI_VALUE); @@ -14060,7 +14062,7 @@ bool create_table_info_t::innobase_table_flags() { if (fts_doc_id_index_bad) { goto index_bad; } - } else if (key->flags & HA_SPATIAL) { + } else if (key->flags & (HA_SPATIAL | HA_VECTOR)) { assert(~m_create_info->options & (HA_LEX_CREATE_TMP_TABLE | HA_LEX_CREATE_INTERNAL_TMP_TABLE)); } @@ -15657,16 +15659,51 @@ int innobase_truncate::exec() { template int innobase_truncate::exec(); template int innobase_truncate::exec(); -/** Check if a column is the only column in an index. -@param[in] index data dictionary index -@param[in] column the column to look for -@return whether the column is the only column in the index */ +/** + Check if a column is the only column in an index. + + @param[in] index data dictionary index + @param[in] column the column to look for + + @return Whether the column is the only column in the index. +*/ static bool dd_is_only_column(const dd::Index *index, const dd::Column *column) { return (index->elements().size() == 1 && &(*index->elements().begin())->column() == column); } +/** + Check if an index uses marker-based vector metadata. + + @param[in] index data dictionary index + + @return Whether the index is a vector index. +*/ +static bool dd_is_vector_index(const dd::Index *index) { + if (index->algorithm() != dd::Index::IA_SE_SPECIFIC || + index->type() != dd::Index::IT_MULTIPLE) { + return false; + } + + uint visible_elements = 0; + for (const dd::Index_element *elem : index->elements()) { + if (elem->is_hidden()) continue; + + visible_elements++; + + const dd::Properties &col_options = elem->column().options(); + bool is_vector_column = false; + if (col_options.exists("vector_index") && + !col_options.get("vector_index", &is_vector_column) && + is_vector_column) { + return visible_elements == 1; + } + } + + return false; +} + /** Add hidden columns and indexes to an InnoDB table definition. @param[in,out] dd_table data dictionary cache object @return error number @@ -15694,6 +15731,10 @@ int ha_innobase::get_extra_columns_and_keys(const HA_CREATE_INFO *, fts_doc_id_index = i; } + if (dd_is_vector_index(i)) { + continue; + } + switch (i->algorithm()) { case dd::Index::IA_SE_SPECIFIC: ut_d(ut_error); @@ -18335,7 +18376,8 @@ void ha_innobase::info_low_key(uint flag, const dict_table_t *ib_table) { /* We do not maintain stats for fulltext or spatial indexes. Thus, we can't calculate pct_cached below because we need dict_index_t::stat_n_leaf_pages for that. See dict_stats_should_ignore_index(). */ - if ((key->flags & HA_FULLTEXT) || (key->flags & HA_SPATIAL)) { + if ((key->flags & HA_FULLTEXT) || (key->flags & HA_SPATIAL) || + (key->flags & HA_VECTOR)) { pct_cached = IN_MEMORY_ESTIMATE_UNKNOWN; } else { pct_cached = index_pct_cached(index); @@ -18353,7 +18395,8 @@ void ha_innobase::info_low_key(uint flag, const dict_table_t *ib_table) { } for (ulong j = 0; j < key->actual_key_parts; j++) { - if ((key->flags & HA_FULLTEXT) || (key->flags & HA_SPATIAL)) { + if ((key->flags & HA_FULLTEXT) || (key->flags & HA_SPATIAL) || + (key->flags & HA_VECTOR)) { /* The record per key does not apply to FTS or Spatial indexes. */ key->set_records_per_key(j, 1.0f); continue; @@ -18679,7 +18722,8 @@ static bool innobase_get_index_column_cardinality( } DEBUG_SYNC(thd, "innodb.after_init_check"); - if (index->type & (DICT_FTS | DICT_SPATIAL)) { + if ((index->type & (DICT_FTS | DICT_SPATIAL)) || + dict_index_is_vector(index)) { /* For these indexes innodb_rec_per_key is fixed as 1.0 */ *cardinality = ib_table->stat_n_rows; diff --git a/storage/innobase/include/dict0dict.ic b/storage/innobase/include/dict0dict.ic index abb3b0a8d35d..b7e415c43c31 100644 --- a/storage/innobase/include/dict0dict.ic +++ b/storage/innobase/include/dict0dict.ic @@ -134,6 +134,22 @@ static inline ulint dict_index_is_spatial( return (index->type & DICT_SPATIAL); } +/** + Check whether the index is a vector index. + + @param[in] index Index. + + @return Nonzero for vector index, zero for other indexes +*/ +static inline ulint dict_index_is_vector( + const dict_index_t *index) /*!< in: index */ +{ + ut_ad(index); + ut_ad(index->magic_n == DICT_INDEX_MAGIC_N); + + return (index->is_vector()); +} + /** Check whether the index contains a virtual column @param[in] index index @return nonzero for the index has virtual column, zero for other indexes */ diff --git a/storage/innobase/include/dict0mem.h b/storage/innobase/include/dict0mem.h index 9a083c0c1c74..0eb6f9ba2ad1 100644 --- a/storage/innobase/include/dict0mem.h +++ b/storage/innobase/include/dict0mem.h @@ -1208,6 +1208,9 @@ struct dict_index_t { bool hidden; #endif /* !UNIV_HOTBACKUP */ + /** true if this is a vector index according to DD metadata */ + bool is_vector_index; + /** list of indexes of the table */ UT_LIST_NODE_T(dict_index_t) indexes; @@ -1337,6 +1340,17 @@ struct dict_index_t { return (type & DICT_MULTI_VALUE); } + /** + Check whether the index is a vector index. + + @return true if the index is a vector index, false otherwise + */ + [[nodiscard]] bool is_vector() const { + ut_ad(magic_n == DICT_INDEX_MAGIC_N); + + return (is_vector_index); + } + /** Returns the minimum data size of an index record. @return minimum data size in bytes */ ulint get_min_size() const { diff --git a/storage/innobase/include/dict0mem.ic b/storage/innobase/include/dict0mem.ic index 65993ac857a0..7b3bb7dbf788 100644 --- a/storage/innobase/include/dict0mem.ic +++ b/storage/innobase/include/dict0mem.ic @@ -76,6 +76,7 @@ static inline void dict_mem_fill_index_struct( index->allow_duplicates = false; index->nulls_equal = false; index->disable_ahi = false; + index->is_vector_index = false; index->last_ins_cur = nullptr; index->last_sel_cur = nullptr; #ifndef UNIV_HOTBACKUP diff --git a/storage/temptable/src/handler.cc b/storage/temptable/src/handler.cc index 7c3e8e0058bd..929e811a6972 100644 --- a/storage/temptable/src/handler.cc +++ b/storage/temptable/src/handler.cc @@ -877,6 +877,7 @@ ulong Handler::index_flags(uint index_no, uint, bool) const { case HA_KEY_ALG_SE_SPECIFIC: case HA_KEY_ALG_RTREE: case HA_KEY_ALG_FULLTEXT: + case HA_KEY_ALG_VECTOR: flags = 0; break; } diff --git a/storage/temptable/src/table.cc b/storage/temptable/src/table.cc index 3c87e1e7728c..ef3504aee96f 100644 --- a/storage/temptable/src/table.cc +++ b/storage/temptable/src/table.cc @@ -290,6 +290,7 @@ void Table::indexes_create() { case HA_KEY_ALG_SE_SPECIFIC: case HA_KEY_ALG_RTREE: case HA_KEY_ALG_FULLTEXT: + case HA_KEY_ALG_VECTOR: DBUG_ABORT(); } }