2020-07-14 14:47:23 +00:00
#!/bin/bash
2021-06-28 11:28:49 +00:00
# shellcheck disable=SC2094
2021-06-28 13:21:17 +00:00
# shellcheck disable=SC2086
2021-12-09 21:12:45 +00:00
# shellcheck disable=SC2024
2020-07-14 14:47:23 +00:00
2020-07-15 09:23:50 +00:00
set -x
2021-08-10 20:49:05 +00:00
# Thread Fuzzer allows to check more permutations of possible thread scheduling
# and find more potential issues.
2022-07-22 09:58:15 +00:00
export THREAD_FUZZER_CPU_TIME_PERIOD_US = 1000
export THREAD_FUZZER_SLEEP_PROBABILITY = 0.1
export THREAD_FUZZER_SLEEP_TIME_US = 100000
export THREAD_FUZZER_pthread_mutex_lock_BEFORE_MIGRATE_PROBABILITY = 1
export THREAD_FUZZER_pthread_mutex_lock_AFTER_MIGRATE_PROBABILITY = 1
export THREAD_FUZZER_pthread_mutex_unlock_BEFORE_MIGRATE_PROBABILITY = 1
export THREAD_FUZZER_pthread_mutex_unlock_AFTER_MIGRATE_PROBABILITY = 1
export THREAD_FUZZER_pthread_mutex_lock_BEFORE_SLEEP_PROBABILITY = 0.001
export THREAD_FUZZER_pthread_mutex_lock_AFTER_SLEEP_PROBABILITY = 0.001
export THREAD_FUZZER_pthread_mutex_unlock_BEFORE_SLEEP_PROBABILITY = 0.001
export THREAD_FUZZER_pthread_mutex_unlock_AFTER_SLEEP_PROBABILITY = 0.001
export THREAD_FUZZER_pthread_mutex_lock_BEFORE_SLEEP_TIME_US = 10000
export THREAD_FUZZER_pthread_mutex_lock_AFTER_SLEEP_TIME_US = 10000
export THREAD_FUZZER_pthread_mutex_unlock_BEFORE_SLEEP_TIME_US = 10000
export THREAD_FUZZER_pthread_mutex_unlock_AFTER_SLEEP_TIME_US = 10000
2021-08-10 20:49:05 +00:00
2021-08-20 12:17:51 +00:00
function install_packages( )
{
dpkg -i $1 /clickhouse-common-static_*.deb
dpkg -i $1 /clickhouse-common-static-dbg_*.deb
dpkg -i $1 /clickhouse-server_*.deb
dpkg -i $1 /clickhouse-client_*.deb
}
2020-07-14 14:47:23 +00:00
2021-02-15 18:02:21 +00:00
function configure( )
2020-08-24 00:14:24 +00:00
{
2021-02-15 18:02:21 +00:00
# install test configs
2022-06-29 12:46:40 +00:00
export USE_DATABASE_ORDINARY = 1
2021-02-15 18:02:21 +00:00
/usr/share/clickhouse-test/config/install.sh
2020-08-24 00:14:24 +00:00
2022-02-15 12:03:51 +00:00
# we mount tests folder from repo to /usr/share
ln -s /usr/share/clickhouse-test/clickhouse-test /usr/bin/clickhouse-test
2022-07-04 13:02:22 +00:00
ln -s /usr/share/clickhouse-test/ci/download_release_packets.py /usr/bin/download_release_packets
ln -s /usr/share/clickhouse-test/ci/get_previous_release_tag.py /usr/bin/get_previous_release_tag
2022-02-15 12:03:51 +00:00
2021-11-16 14:45:37 +00:00
# avoid too slow startup
2021-11-16 17:03:50 +00:00
sudo cat /etc/clickhouse-server/config.d/keeper_port.xml | sed "s|<snapshot_distance>100000</snapshot_distance>|<snapshot_distance>10000</snapshot_distance>|" > /etc/clickhouse-server/config.d/keeper_port.xml.tmp
sudo mv /etc/clickhouse-server/config.d/keeper_port.xml.tmp /etc/clickhouse-server/config.d/keeper_port.xml
sudo chown clickhouse /etc/clickhouse-server/config.d/keeper_port.xml
sudo chgrp clickhouse /etc/clickhouse-server/config.d/keeper_port.xml
2020-08-24 00:14:24 +00:00
2021-02-15 18:02:21 +00:00
# for clickhouse-server (via service)
echo "ASAN_OPTIONS='malloc_context_size=10 verbosity=1 allocator_release_to_os_interval_ms=10000'" >> /etc/environment
# for clickhouse-client
export ASAN_OPTIONS = 'malloc_context_size=10 allocator_release_to_os_interval_ms=10000'
# since we run clickhouse from root
sudo chown root: /var/lib/clickhouse
2021-04-24 00:27:23 +00:00
# Set more frequent update period of asynchronous metrics to more frequently update information about real memory usage (less chance of OOM).
2021-10-25 18:15:42 +00:00
echo "<clickhouse><asynchronous_metrics_update_period_s>1</asynchronous_metrics_update_period_s></clickhouse>" \
2021-04-24 00:27:23 +00:00
> /etc/clickhouse-server/config.d/asynchronous_metrics_update_period_s.xml
2021-12-06 06:05:34 +00:00
local total_mem
total_mem = $( awk '/MemTotal/ { print $(NF-1) }' /proc/meminfo) # KiB
total_mem = $(( total_mem*1024 )) # bytes
2021-04-24 00:27:23 +00:00
# Set maximum memory usage as half of total memory (less chance of OOM).
2021-12-06 06:05:34 +00:00
#
# But not via max_server_memory_usage but via max_memory_usage_for_user,
# so that we can override this setting and execute service queries, like:
# - hung check
# - show/drop database
# - ...
#
# So max_memory_usage_for_user will be a soft limit, and
# max_server_memory_usage will be hard limit, and queries that should be
# executed regardless memory limits will use max_memory_usage_for_user=0,
# instead of relying on max_untracked_memory
local max_server_mem
max_server_mem = $(( total_mem*75/100)) # 75%
echo " Setting max_server_memory_usage= $max_server_mem "
cat > /etc/clickhouse-server/config.d/max_server_memory_usage.xml <<EOL
<clickhouse>
<max_server_memory_usage>${ max_server_mem } </max_server_memory_usage>
</clickhouse>
EOL
local max_users_mem
max_users_mem = $(( total_mem*50/100)) # 50%
echo " Setting max_memory_usage_for_user= $max_users_mem "
cat > /etc/clickhouse-server/users.d/max_memory_usage_for_user.xml <<EOL
<clickhouse>
<profiles>
<default>
<max_memory_usage_for_user>${ max_users_mem } </max_memory_usage_for_user>
</default>
</profiles>
</clickhouse>
EOL
2021-02-15 18:02:21 +00:00
}
2020-08-24 00:14:24 +00:00
function stop( )
{
2022-08-26 15:47:29 +00:00
local pid
# Preserve the pid, since the server can hung after the PID will be deleted.
pid = " $( cat /var/run/clickhouse-server/clickhouse-server.pid) "
2022-05-09 17:43:51 +00:00
clickhouse stop --do-not-kill && return
# We failed to stop the server with SIGTERM. Maybe it hang, let's collect stacktraces.
2022-05-10 09:48:55 +00:00
kill -TERM " $( pidof gdb) " || :
sleep 5
2022-07-03 11:45:56 +00:00
echo "thread apply all backtrace (on stop)" >> /test_output/gdb.log
2022-08-26 15:47:29 +00:00
gdb -batch -ex 'thread apply all backtrace' -p " $pid " | ts '%Y-%m-%d %H:%M:%S' >> /test_output/gdb.log
2022-05-09 17:43:51 +00:00
clickhouse stop --force
2020-08-24 00:14:24 +00:00
}
function start( )
2020-07-14 14:47:23 +00:00
{
counter = 0
until clickhouse-client --query "SELECT 1"
do
2022-07-01 11:47:49 +00:00
if [ " $counter " -gt ${ 1 :- 120 } ]
2020-07-14 14:47:23 +00:00
then
2020-08-18 09:43:02 +00:00
echo "Cannot start clickhouse-server"
2022-07-01 11:47:49 +00:00
echo -e "Cannot start clickhouse-server\tFAIL" >> /test_output/test_results.tsv
2020-08-18 09:43:02 +00:00
cat /var/log/clickhouse-server/stdout.log
2020-08-23 20:48:27 +00:00
tail -n1000 /var/log/clickhouse-server/stderr.log
2021-08-28 16:19:21 +00:00
tail -n100000 /var/log/clickhouse-server/clickhouse-server.log | grep -F -v -e '<Warning> RaftInstance:' -e '<Information> RaftInstance' | tail -n1000
2020-07-14 14:47:23 +00:00
break
fi
2021-02-14 20:31:58 +00:00
# use root to match with current uid
2021-07-16 07:46:22 +00:00
clickhouse start --user root >/var/log/clickhouse-server/stdout.log 2>>/var/log/clickhouse-server/stderr.log
2020-07-14 14:47:23 +00:00
sleep 0.5
2020-09-30 17:06:14 +00:00
counter = $(( counter + 1 ))
2020-07-14 14:47:23 +00:00
done
2021-12-10 15:03:57 +00:00
# Set follow-fork-mode to parent, because we attach to clickhouse-server, not to watchdog
# and clickhouse-server can do fork-exec, for example, to run some bridge.
# Do not set nostop noprint for all signals, because some it may cause gdb to hang,
# explicitly ignore non-fatal signals that are used by server.
# Number of SIGRTMIN can be determined only in runtime.
2021-12-10 17:58:09 +00:00
RTMIN = $( kill -l SIGRTMIN)
2021-02-13 08:41:00 +00:00
echo "
2021-12-10 15:03:57 +00:00
set follow-fork-mode parent
handle SIGHUP nostop noprint pass
handle SIGINT nostop noprint pass
handle SIGQUIT nostop noprint pass
handle SIGPIPE nostop noprint pass
handle SIGTERM nostop noprint pass
handle SIGUSR1 nostop noprint pass
handle SIGUSR2 nostop noprint pass
handle SIG$RTMIN nostop noprint pass
info signals
2021-02-13 08:41:00 +00:00
continue
2022-01-04 11:03:40 +00:00
gcore
2021-12-10 15:03:57 +00:00
backtrace full
2022-02-13 12:02:15 +00:00
thread apply all backtrace full
2021-12-15 10:21:21 +00:00
info registers
disassemble /s
up
disassemble /s
up
disassemble /s
p \" done \"
2021-02-20 16:27:04 +00:00
detach
quit
2021-02-13 08:41:00 +00:00
" > script.gdb
2021-02-22 13:53:43 +00:00
# FIXME Hung check may work incorrectly because of attached gdb
# 1. False positives are possible
# 2. We cannot attach another gdb to get stacktraces if some queries hung
2021-12-10 17:10:49 +00:00
gdb -batch -command script.gdb -p " $( cat /var/run/clickhouse-server/clickhouse-server.pid) " | ts '%Y-%m-%d %H:%M:%S' >> /test_output/gdb.log &
2021-12-10 15:03:57 +00:00
sleep 5
# gdb will send SIGSTOP, spend some time loading debug info and then send SIGCONT, wait for it (up to send_timeout, 300s)
time clickhouse-client --query "SELECT 'Connected to clickhouse-server after attaching gdb'" || :
2020-07-14 14:47:23 +00:00
}
2021-08-20 12:17:51 +00:00
install_packages package_folder
2021-02-15 18:02:21 +00:00
configure
2020-07-14 14:47:23 +00:00
2022-08-02 12:27:45 +00:00
azurite-blob --blobHost 0.0.0.0 --blobPort 10000 --debug /azurite_log &
2022-06-10 12:51:18 +00:00
./setup_minio.sh stateful # to have a proper environment
2022-02-23 13:58:44 +00:00
2020-08-23 21:13:21 +00:00
start
2020-07-14 14:47:23 +00:00
2020-10-01 09:27:05 +00:00
# shellcheck disable=SC2086 # No quotes because I want to split it into words.
2021-11-01 10:32:56 +00:00
/s3downloader --url-prefix " $S3_URL " --dataset-names $DATASETS
2020-07-14 14:47:23 +00:00
chmod 777 -R /var/lib/clickhouse
clickhouse-client --query "ATTACH DATABASE IF NOT EXISTS datasets ENGINE = Ordinary"
clickhouse-client --query "CREATE DATABASE IF NOT EXISTS test"
2020-08-04 08:48:47 +00:00
2020-08-23 21:13:21 +00:00
stop
2022-04-06 06:52:45 +00:00
mv /var/log/clickhouse-server/clickhouse-server.log /var/log/clickhouse-server/clickhouse-server.initial.log
2020-08-23 21:13:21 +00:00
start
2020-07-14 14:47:23 +00:00
clickhouse-client --query "SHOW TABLES FROM datasets"
clickhouse-client --query "SHOW TABLES FROM test"
clickhouse-client --query "RENAME TABLE datasets.hits_v1 TO test.hits"
clickhouse-client --query "RENAME TABLE datasets.visits_v1 TO test.visits"
2022-02-23 13:58:44 +00:00
clickhouse-client --query "CREATE TABLE test.hits_s3 (WatchID UInt64, JavaEnable UInt8, Title String, GoodEvent Int16, EventTime DateTime, EventDate Date, CounterID UInt32, ClientIP UInt32, ClientIP6 FixedString(16), RegionID UInt32, UserID UInt64, CounterClass Int8, OS UInt8, UserAgent UInt8, URL String, Referer String, URLDomain String, RefererDomain String, Refresh UInt8, IsRobot UInt8, RefererCategories Array(UInt16), URLCategories Array(UInt16), URLRegions Array(UInt32), RefererRegions Array(UInt32), ResolutionWidth UInt16, ResolutionHeight UInt16, ResolutionDepth UInt8, FlashMajor UInt8, FlashMinor UInt8, FlashMinor2 String, NetMajor UInt8, NetMinor UInt8, UserAgentMajor UInt16, UserAgentMinor FixedString(2), CookieEnable UInt8, JavascriptEnable UInt8, IsMobile UInt8, MobilePhone UInt8, MobilePhoneModel String, Params String, IPNetworkID UInt32, TraficSourceID Int8, SearchEngineID UInt16, SearchPhrase String, AdvEngineID UInt8, IsArtifical UInt8, WindowClientWidth UInt16, WindowClientHeight UInt16, ClientTimeZone Int16, ClientEventTime DateTime, SilverlightVersion1 UInt8, SilverlightVersion2 UInt8, SilverlightVersion3 UInt32, SilverlightVersion4 UInt16, PageCharset String, CodeVersion UInt32, IsLink UInt8, IsDownload UInt8, IsNotBounce UInt8, FUniqID UInt64, HID UInt32, IsOldCounter UInt8, IsEvent UInt8, IsParameter UInt8, DontCountHits UInt8, WithHash UInt8, HitColor FixedString(1), UTCEventTime DateTime, Age UInt8, Sex UInt8, Income UInt8, Interests UInt16, Robotness UInt8, GeneralInterests Array(UInt16), RemoteIP UInt32, RemoteIP6 FixedString(16), WindowName Int32, OpenerName Int32, HistoryLength Int16, BrowserLanguage FixedString(2), BrowserCountry FixedString(2), SocialNetwork String, SocialAction String, HTTPError UInt16, SendTiming Int32, DNSTiming Int32, ConnectTiming Int32, ResponseStartTiming Int32, ResponseEndTiming Int32, FetchTiming Int32, RedirectTiming Int32, DOMInteractiveTiming Int32, DOMContentLoadedTiming Int32, DOMCompleteTiming Int32, LoadEventStartTiming Int32, LoadEventEndTiming Int32, NSToDOMContentLoadedTiming Int32, FirstPaintTiming Int32, RedirectCount Int8, SocialSourceNetworkID UInt8, SocialSourcePage String, ParamPrice Int64, ParamOrderID String, ParamCurrency FixedString(3), ParamCurrencyID UInt16, GoalsReached Array(UInt32), OpenstatServiceName String, OpenstatCampaignID String, OpenstatAdID String, OpenstatSourceID String, UTMSource String, UTMMedium String, UTMCampaign String, UTMContent String, UTMTerm String, FromTag String, HasGCLID UInt8, RefererHash UInt64, URLHash UInt64, CLID UInt32, YCLID UInt64, ShareService String, ShareURL String, ShareTitle String, ParsedParams Nested(Key1 String, Key2 String, Key3 String, Key4 String, Key5 String, ValueDouble Float64), IslandID FixedString(16), RequestNum UInt32, RequestTry UInt8) ENGINE = MergeTree() PARTITION BY toYYYYMM(EventDate) ORDER BY (CounterID, EventDate, intHash32(UserID)) SAMPLE BY intHash32(UserID) SETTINGS index_granularity = 8192, storage_policy='s3_cache'"
clickhouse-client --query "INSERT INTO test.hits_s3 SELECT * FROM test.hits"
2020-07-14 14:47:23 +00:00
clickhouse-client --query "SHOW TABLES FROM test"
2021-06-03 15:16:12 +00:00
./stress --hung-check --drop-databases --output-folder test_output --skip-func-tests " $SKIP_TESTS_OPTION " \
2021-02-18 22:08:44 +00:00
&& echo -e 'Test script exit code\tOK' >> /test_output/test_results.tsv \
|| echo -e 'Test script failed\tFAIL' >> /test_output/test_results.tsv
2020-07-14 14:47:23 +00:00
2020-08-23 21:13:21 +00:00
stop
2022-04-06 06:52:45 +00:00
mv /var/log/clickhouse-server/clickhouse-server.log /var/log/clickhouse-server/clickhouse-server.stress.log
2022-05-10 13:16:21 +00:00
# NOTE Disable thread fuzzer before server start with data after stress test.
# In debug build it can take a lot of time.
unset " ${ !THREAD_@ } "
2020-08-23 21:13:21 +00:00
start
2020-07-14 14:47:23 +00:00
2021-02-18 22:08:44 +00:00
clickhouse-client --query "SELECT 'Server successfully started', 'OK'" >> /test_output/test_results.tsv \
2022-05-06 21:01:03 +00:00
|| ( echo -e 'Server failed to start (see application_errors.txt and clickhouse-server.clean.log)\tFAIL' >> /test_output/test_results.tsv \
2022-06-28 11:12:21 +00:00
&& grep -a "<Error>.*Application" /var/log/clickhouse-server/clickhouse-server.log > /test_output/application_errors.txt)
2021-02-18 22:08:44 +00:00
2022-08-02 09:10:53 +00:00
echo "Get previous release tag"
previous_release_tag = $( clickhouse-client --query= "SELECT version()" | get_previous_release_tag)
echo $previous_release_tag
2022-08-01 17:47:14 +00:00
stop
2021-02-18 22:08:44 +00:00
[ -f /var/log/clickhouse-server/clickhouse-server.log ] || echo -e "Server log does not exist\tFAIL"
[ -f /var/log/clickhouse-server/stderr.log ] || echo -e "Stderr log does not exist\tFAIL"
# Grep logs for sanitizer asserts, crashes and other critical errors
# Sanitizer asserts
2021-11-30 10:24:04 +00:00
grep -Fa "==================" /var/log/clickhouse-server/stderr.log | grep -v "in query:" >> /test_output/tmp
2021-11-30 10:22:50 +00:00
grep -Fa "WARNING" /var/log/clickhouse-server/stderr.log >> /test_output/tmp
2022-07-06 12:32:17 +00:00
zgrep -Fav -e "ASan doesn't fully support makecontext/swapcontext functions" -e "DB::Exception" /test_output/tmp > /dev/null \
2021-02-18 22:08:44 +00:00
&& echo -e 'Sanitizer assert (in stderr.log)\tFAIL' >> /test_output/test_results.tsv \
|| echo -e 'No sanitizer asserts\tOK' >> /test_output/test_results.tsv
rm -f /test_output/tmp
2021-04-09 06:39:25 +00:00
# OOM
2022-04-20 11:53:16 +00:00
zgrep -Fa " <Fatal> Application: Child process was terminated by signal 9" /var/log/clickhouse-server/clickhouse-server*.log > /dev/null \
2021-04-09 06:39:25 +00:00
&& echo -e 'OOM killer (or signal 9) in clickhouse-server.log\tFAIL' >> /test_output/test_results.tsv \
|| echo -e 'No OOM messages in clickhouse-server.log\tOK' >> /test_output/test_results.tsv
2021-02-18 22:08:44 +00:00
# Logical errors
2022-04-20 11:53:16 +00:00
zgrep -Fa "Code: 49, e.displayText() = DB::Exception:" /var/log/clickhouse-server/clickhouse-server*.log > /test_output/logical_errors.txt \
2022-03-22 12:00:20 +00:00
&& echo -e 'Logical error thrown (see clickhouse-server.log or logical_errors.txt)\tFAIL' >> /test_output/test_results.tsv \
2021-02-18 22:08:44 +00:00
|| echo -e 'No logical errors\tOK' >> /test_output/test_results.tsv
2022-03-22 12:00:20 +00:00
# Remove file logical_errors.txt if it's empty
2022-03-22 17:06:35 +00:00
[ -s /test_output/logical_errors.txt ] || rm /test_output/logical_errors.txt
2022-03-22 12:00:20 +00:00
2021-02-18 22:08:44 +00:00
# Crash
2022-04-20 11:53:16 +00:00
zgrep -Fa "########################################" /var/log/clickhouse-server/clickhouse-server*.log > /dev/null \
2021-02-18 22:08:44 +00:00
&& echo -e 'Killed by signal (in clickhouse-server.log)\tFAIL' >> /test_output/test_results.tsv \
|| echo -e 'Not crashed\tOK' >> /test_output/test_results.tsv
2021-04-09 06:39:25 +00:00
# It also checks for crash without stacktrace (printed by watchdog)
2022-04-20 11:53:16 +00:00
zgrep -Fa " <Fatal> " /var/log/clickhouse-server/clickhouse-server*.log > /test_output/fatal_messages.txt \
2022-03-22 12:00:20 +00:00
&& echo -e 'Fatal message in clickhouse-server.log (see fatal_messages.txt)\tFAIL' >> /test_output/test_results.tsv \
2021-02-18 22:08:44 +00:00
|| echo -e 'No fatal messages in clickhouse-server.log\tOK' >> /test_output/test_results.tsv
2022-03-22 12:00:20 +00:00
# Remove file fatal_messages.txt if it's empty
2022-03-23 10:26:31 +00:00
[ -s /test_output/fatal_messages.txt ] || rm /test_output/fatal_messages.txt
2022-03-22 12:00:20 +00:00
2021-02-18 22:08:44 +00:00
zgrep -Fa "########################################" /test_output/* > /dev/null \
&& echo -e 'Killed by signal (output files)\tFAIL' >> /test_output/test_results.tsv
2021-12-10 15:03:57 +00:00
zgrep -Fa " received signal " /test_output/gdb.log > /dev/null \
&& echo -e 'Found signal in gdb.log\tFAIL' >> /test_output/test_results.tsv
2021-08-20 12:17:51 +00:00
echo -e "Backward compatibility check\n"
2022-07-04 13:02:22 +00:00
echo "Clone previous release repository"
2022-07-04 15:30:22 +00:00
git clone https://github.com/ClickHouse/ClickHouse.git --no-tags --progress --branch= $previous_release_tag --no-recurse-submodules --depth= 1 previous_release_repository
2022-07-04 13:02:22 +00:00
2021-08-20 12:17:51 +00:00
echo "Download previous release server"
2022-01-27 15:15:03 +00:00
mkdir previous_release_package_folder
2022-07-04 13:02:22 +00:00
echo $previous_release_tag | download_release_packets && echo -e 'Download script exit code\tOK' >> /test_output/test_results.tsv \
2022-03-22 12:00:20 +00:00
|| echo -e 'Download script failed\tFAIL' >> /test_output/test_results.tsv
2021-08-20 12:17:51 +00:00
2022-04-06 06:52:45 +00:00
mv /var/log/clickhouse-server/clickhouse-server.log /var/log/clickhouse-server/clickhouse-server.clean.log
2022-07-04 13:02:22 +00:00
# Check if we cloned previous release repository successfully
2022-07-04 18:38:21 +00:00
if ! [ " $( ls -A previous_release_repository/tests/queries) " ]
2022-07-04 13:02:22 +00:00
then
echo -e "Backward compatibility check: Failed to clone previous release tests\tFAIL" >> /test_output/test_results.tsv
2022-07-04 18:38:21 +00:00
elif ! [ " $( ls -A previous_release_package_folder/clickhouse-common-static_*.deb && ls -A previous_release_package_folder/clickhouse-server_*.deb) " ]
2021-08-20 12:17:51 +00:00
then
2022-07-04 13:02:22 +00:00
echo -e "Backward compatibility check: Failed to download previous release packets\tFAIL" >> /test_output/test_results.tsv
else
echo -e "Successfully cloned previous release tests\tOK" >> /test_output/test_results.tsv
2022-03-22 12:00:20 +00:00
echo -e "Successfully downloaded previous release packets\tOK" >> /test_output/test_results.tsv
2021-08-20 12:17:51 +00:00
# Uninstall current packages
dpkg --remove clickhouse-client
dpkg --remove clickhouse-server
dpkg --remove clickhouse-common-static-dbg
dpkg --remove clickhouse-common-static
2021-09-28 11:09:14 +00:00
rm -rf /var/lib/clickhouse/*
2022-06-23 19:38:43 +00:00
# Make BC check more funny by forcing Ordinary engine for system database
mkdir /var/lib/clickhouse/metadata
echo "ATTACH DATABASE system ENGINE=Ordinary" > /var/lib/clickhouse/metadata/system.sql
2021-08-20 12:17:51 +00:00
# Install previous release packages
install_packages previous_release_package_folder
# Start server from previous release
configure
2022-07-01 11:47:49 +00:00
2022-08-24 12:43:02 +00:00
# Avoid "Setting s3_check_objects_after_upload is neither a builtin setting..."
rm -f /etc/clickhouse-server/users.d/enable_blobs_check.xml || :
2022-08-15 22:56:27 +00:00
2022-07-14 20:04:39 +00:00
# Remove s3 related configs to avoid "there is no disk type `cache`"
rm -f /etc/clickhouse-server/config.d/storage_conf.xml || :
2022-08-09 11:02:52 +00:00
rm -f /etc/clickhouse-server/config.d/azure_storage_conf.xml || :
2022-07-01 11:47:49 +00:00
2021-08-20 12:17:51 +00:00
start
clickhouse-client --query= "SELECT 'Server version: ', version()"
# Install new package before running stress test because we should use new clickhouse-client and new clickhouse-test
2022-07-15 10:23:28 +00:00
# But we should leave old binary in /usr/bin/ for gdb (so it will print sane stacktarces)
mv /usr/bin/clickhouse previous_release_package_folder/
2021-08-20 12:17:51 +00:00
install_packages package_folder
2022-07-15 10:23:28 +00:00
mv /usr/bin/clickhouse package_folder/
mv previous_release_package_folder/clickhouse /usr/bin/
2021-08-20 12:17:51 +00:00
mkdir tmp_stress_output
2022-04-06 06:48:18 +00:00
2022-07-04 13:02:22 +00:00
./stress --test-cmd= "/usr/bin/clickhouse-test --queries=\"previous_release_repository/tests/queries\"" --backward-compatibility-check --output-folder tmp_stress_output --global-time-limit= 1200 \
2022-03-22 12:00:20 +00:00
&& echo -e 'Backward compatibility check: Test script exit code\tOK' >> /test_output/test_results.tsv \
|| echo -e 'Backward compatibility check: Test script failed\tFAIL' >> /test_output/test_results.tsv
2021-08-20 12:17:51 +00:00
rm -rf tmp_stress_output
clickhouse-client --query= "SELECT 'Tables count:', count() FROM system.tables"
2021-09-28 11:17:50 +00:00
2022-04-06 06:48:18 +00:00
stop
mv /var/log/clickhouse-server/clickhouse-server.log /var/log/clickhouse-server/clickhouse-server.backward.stress.log
2021-08-20 12:17:51 +00:00
# Start new server
2022-07-15 10:23:28 +00:00
mv package_folder/clickhouse /usr/bin/
2021-08-20 12:17:51 +00:00
configure
2022-03-10 11:38:39 +00:00
start 500
2022-03-22 12:00:20 +00:00
clickhouse-client --query "SELECT 'Backward compatibility check: Server successfully started', 'OK'" >> /test_output/test_results.tsv \
2022-03-23 10:28:40 +00:00
|| ( echo -e 'Backward compatibility check: Server failed to start\tFAIL' >> /test_output/test_results.tsv \
2022-06-28 11:12:21 +00:00
&& grep -a "<Error>.*Application" /var/log/clickhouse-server/clickhouse-server.log >> /test_output/bc_check_application_errors.txt)
2021-08-20 12:17:51 +00:00
clickhouse-client --query= "SELECT 'Server version: ', version()"
# Let the server run for a while before checking log.
sleep 60
2022-04-06 06:48:18 +00:00
2021-08-20 12:17:51 +00:00
stop
2022-04-06 06:48:18 +00:00
mv /var/log/clickhouse-server/clickhouse-server.log /var/log/clickhouse-server/clickhouse-server.backward.clean.log
2021-08-20 12:17:51 +00:00
2021-09-28 11:09:14 +00:00
# Error messages (we should ignore some errors)
2022-07-08 13:47:54 +00:00
# FIXME https://github.com/ClickHouse/ClickHouse/issues/38643 ("Unknown index: idx.")
2022-07-13 12:22:36 +00:00
# FIXME https://github.com/ClickHouse/ClickHouse/issues/39174 ("Cannot parse string 'Hello' as UInt64")
2022-07-14 14:29:08 +00:00
# FIXME Not sure if it's expected, but some tests from BC check may not be finished yet when we restarting server.
# Let's just ignore all errors from queries ("} <Error> TCPHandler: Code:", "} <Error> executeQuery: Code:")
2022-07-15 17:03:00 +00:00
# FIXME https://github.com/ClickHouse/ClickHouse/issues/39197 ("Missing columns: 'v3' while processing query: 'v3, k, v1, v2, p'")
2022-07-20 10:19:53 +00:00
# NOTE Incompatibility was introduced in https://github.com/ClickHouse/ClickHouse/pull/39263, it's expected
# ("This engine is deprecated and is not supported in transactions", "[Queue = DB::MergeMutateRuntimeQueue]: Code: 235. DB::Exception: Part")
2022-03-22 12:00:20 +00:00
echo "Check for Error messages in server log:"
2022-02-18 13:36:48 +00:00
zgrep -Fav -e "Code: 236. DB::Exception: Cancelled merging parts" \
2022-03-10 14:55:45 +00:00
-e "Code: 236. DB::Exception: Cancelled mutating parts" \
2022-02-18 13:36:48 +00:00
-e "REPLICA_IS_ALREADY_ACTIVE" \
2022-03-14 16:40:17 +00:00
-e "REPLICA_IS_ALREADY_EXIST" \
2022-03-22 12:00:20 +00:00
-e "ALL_REPLICAS_LOST" \
2022-02-18 13:36:48 +00:00
-e "DDLWorker: Cannot parse DDL task query" \
-e "RaftInstance: failed to accept a rpc connection due to error 125" \
-e "UNKNOWN_DATABASE" \
-e "NETWORK_ERROR" \
-e "UNKNOWN_TABLE" \
-e "ZooKeeperClient" \
-e "KEEPER_EXCEPTION" \
-e "DirectoryMonitor" \
2022-02-28 10:33:10 +00:00
-e "TABLE_IS_READ_ONLY" \
2022-02-18 13:36:48 +00:00
-e "Code: 1000, e.code() = 111, Connection refused" \
2022-03-16 11:30:45 +00:00
-e "UNFINISHED" \
2022-08-19 11:31:57 +00:00
-e "NETLINK_ERROR" \
2022-03-16 11:30:45 +00:00
-e "Renaming unexpected part" \
2022-06-02 11:33:27 +00:00
-e "PART_IS_TEMPORARILY_LOCKED" \
2022-06-28 11:28:06 +00:00
-e "and a merge is impossible: we didn't find" \
2022-06-16 10:24:21 +00:00
-e "found in queue and some source parts for it was lost" \
-e "is lost forever." \
2022-07-08 13:47:54 +00:00
-e "Unknown index: idx." \
2022-07-13 12:22:36 +00:00
-e "Cannot parse string 'Hello' as UInt64" \
2022-07-14 14:29:08 +00:00
-e "} <Error> TCPHandler: Code:" \
-e "} <Error> executeQuery: Code:" \
2022-07-15 17:03:00 +00:00
-e "Missing columns: 'v3' while processing query: 'v3, k, v1, v2, p'" \
2022-07-20 10:19:53 +00:00
-e "This engine is deprecated and is not supported in transactions" \
-e "[Queue = DB::MergeMutateRuntimeQueue]: Code: 235. DB::Exception: Part" \
2022-08-15 11:53:14 +00:00
-e "The set of parts restored in place of" \
2022-04-08 11:55:10 +00:00
/var/log/clickhouse-server/clickhouse-server.backward.clean.log | zgrep -Fa "<Error>" > /test_output/bc_check_error_messages.txt \
&& echo -e 'Backward compatibility check: Error message in clickhouse-server.log (see bc_check_error_messages.txt)\tFAIL' >> /test_output/test_results.tsv \
2022-03-22 12:00:20 +00:00
|| echo -e 'Backward compatibility check: No Error messages in clickhouse-server.log\tOK' >> /test_output/test_results.tsv
# Remove file bc_check_error_messages.txt if it's empty
2022-03-22 17:06:35 +00:00
[ -s /test_output/bc_check_error_messages.txt ] || rm /test_output/bc_check_error_messages.txt
2021-08-20 12:17:51 +00:00
# Sanitizer asserts
zgrep -Fa "==================" /var/log/clickhouse-server/stderr.log >> /test_output/tmp
zgrep -Fa "WARNING" /var/log/clickhouse-server/stderr.log >> /test_output/tmp
2022-07-06 12:32:17 +00:00
zgrep -Fav -e "ASan doesn't fully support makecontext/swapcontext functions" -e "DB::Exception" /test_output/tmp > /dev/null \
2022-03-22 12:00:20 +00:00
&& echo -e 'Backward compatibility check: Sanitizer assert (in stderr.log)\tFAIL' >> /test_output/test_results.tsv \
|| echo -e 'Backward compatibility check: No sanitizer asserts\tOK' >> /test_output/test_results.tsv
2021-08-20 12:17:51 +00:00
rm -f /test_output/tmp
# OOM
2022-04-06 06:48:18 +00:00
zgrep -Fa " <Fatal> Application: Child process was terminated by signal 9" /var/log/clickhouse-server/clickhouse-server.backward.*.log > /dev/null \
2022-04-08 11:55:10 +00:00
&& echo -e 'Backward compatibility check: OOM killer (or signal 9) in clickhouse-server.log\tFAIL' >> /test_output/test_results.tsv \
2022-03-22 12:00:20 +00:00
|| echo -e 'Backward compatibility check: No OOM messages in clickhouse-server.log\tOK' >> /test_output/test_results.tsv
2021-08-20 12:17:51 +00:00
# Logical errors
2022-03-22 12:00:20 +00:00
echo "Check for Logical errors in server log:"
2022-04-06 06:48:18 +00:00
zgrep -Fa -A20 "Code: 49, e.displayText() = DB::Exception:" /var/log/clickhouse-server/clickhouse-server.backward.*.log > /test_output/bc_check_logical_errors.txt \
2022-03-22 12:00:20 +00:00
&& echo -e 'Backward compatibility check: Logical error thrown (see clickhouse-server.log or bc_check_logical_errors.txt)\tFAIL' >> /test_output/test_results.tsv \
|| echo -e 'Backward compatibility check: No logical errors\tOK' >> /test_output/test_results.tsv
# Remove file bc_check_logical_errors.txt if it's empty
2022-03-22 17:06:35 +00:00
[ -s /test_output/bc_check_logical_errors.txt ] || rm /test_output/bc_check_logical_errors.txt
2021-08-20 12:17:51 +00:00
# Crash
2022-04-06 06:48:18 +00:00
zgrep -Fa "########################################" /var/log/clickhouse-server/clickhouse-server.backward.*.log > /dev/null \
2022-03-22 12:00:20 +00:00
&& echo -e 'Backward compatibility check: Killed by signal (in clickhouse-server.log)\tFAIL' >> /test_output/test_results.tsv \
|| echo -e 'Backward compatibility check: Not crashed\tOK' >> /test_output/test_results.tsv
2021-08-20 12:17:51 +00:00
# It also checks for crash without stacktrace (printed by watchdog)
2022-03-22 12:00:20 +00:00
echo "Check for Fatal message in server log:"
2022-04-06 06:48:18 +00:00
zgrep -Fa " <Fatal> " /var/log/clickhouse-server/clickhouse-server.backward.*.log > /test_output/bc_check_fatal_messages.txt \
2022-04-08 11:55:10 +00:00
&& echo -e 'Backward compatibility check: Fatal message in clickhouse-server.log (see bc_check_fatal_messages.txt)\tFAIL' >> /test_output/test_results.tsv \
2022-03-22 12:00:20 +00:00
|| echo -e 'Backward compatibility check: No fatal messages in clickhouse-server.log\tOK' >> /test_output/test_results.tsv
# Remove file bc_check_fatal_messages.txt if it's empty
2022-03-22 17:06:35 +00:00
[ -s /test_output/bc_check_fatal_messages.txt ] || rm /test_output/bc_check_fatal_messages.txt
2021-08-20 12:17:51 +00:00
fi
2022-08-27 10:46:16 +00:00
dmesg -T > /test_output/dmesg.log
# OOM in dmesg -- those are real
grep -q -F -e 'Out of memory: Killed process' -e 'oom_reaper: reaped process' -e 'oom-kill:constraint=CONSTRAINT_NONE' /test_output/dmesg.log \
&& echo -e 'OOM in dmesg\tFAIL' >> /test_output/test_results.tsv \
|| echo -e 'No OOM in dmesg\tOK' >> /test_output/test_results.tsv
2021-03-07 14:44:30 +00:00
tar -chf /test_output/coordination.tar /var/lib/clickhouse/coordination || :
2021-02-19 09:57:09 +00:00
mv /var/log/clickhouse-server/stderr.log /test_output/
2021-10-23 16:58:10 +00:00
# Replace the engine with Ordinary to avoid extra symlinks stuff in artifacts.
# (so that clickhouse-local --path can read it w/o extra care).
sed -i -e "s/ATTACH DATABASE _ UUID '[^']*'/ATTACH DATABASE system/" -e "s/Atomic/Ordinary/" /var/lib/clickhouse/metadata/system.sql
for table in query_log trace_log; do
sed -i " s/ATTACH TABLE _ UUID '[^']*'/ATTACH TABLE $table / " /var/lib/clickhouse/metadata/system/${ table } .sql
tar -chf /test_output/${ table } _dump.tar /var/lib/clickhouse/metadata/system.sql /var/lib/clickhouse/metadata/system/${ table } .sql /var/lib/clickhouse/data/system/${ table } || :
done
2021-02-19 09:57:09 +00:00
2021-02-18 22:08:44 +00:00
# Write check result into check_status.tsv
2022-05-09 14:48:10 +00:00
clickhouse-local --structure "test String, res String" -q "SELECT 'failure', test FROM table WHERE res != 'OK' order by (lower(test) like '%hung%'), rowNumberInAllBlocks() LIMIT 1" < /test_output/test_results.tsv > /test_output/check_status.tsv
2021-02-19 19:39:42 +00:00
[ -s /test_output/check_status.tsv ] || echo -e "success\tNo errors found" > /test_output/check_status.tsv
2022-01-04 11:03:40 +00:00
# Core dumps (see gcore)
# Default filename is 'core.PROCESS_ID'
for core in core.*; do
pigz $core
2022-02-13 12:02:15 +00:00
mv $core .gz /test_output/
2022-01-04 11:03:40 +00:00
done