mirror of
https://github.com/ClickHouse/ClickHouse.git
synced 2024-12-04 21:42:39 +00:00
60f9f6855d
This commit moves the catboost model evaluation out of the server process into the library-bridge binary. This serves two goals: On the one hand, crashes / memory corruptions of the catboost library no longer affect the server. On the other hand, we can forbid loading dynamic libraries in the server (catboost was the last consumer of this functionality), thus improving security. SQL syntax: SELECT catboostEvaluate('/path/to/model.bin', FEAT_1, ..., FEAT_N) > 0 AS prediction, ACTION AS target FROM amazon_train LIMIT 10 Required configuration: <catboost_lib_path>/path/to/libcatboostmodel.so</catboost_lib_path> *** Implementation Details *** The internal protocol between the server and the library-bridge is simple: - HTTP GET on path "/extdict_ping": A ping, used during the handshake to check if the library-bridge runs. - HTTP POST on path "extdict_request" (1) Send a "catboost_GetTreeCount" request from the server to the bridge, containing a library path (e.g /home/user/libcatboost.so) and a model path (e.g. /home/user/model.bin). Rirst, this unloads the catboost library handler associated to the model path (if it was loaded), then loads the catboost library handler associated to the model path, then executes GetTreeCount() on the library handler and finally sends the result back to the server. Step (1) is called once by the server from FunctionCatBoostEvaluate::getReturnTypeImpl(). The library path handler is unloaded in the beginning because it contains state which may no longer be valid if the user runs catboost("/path/to/model.bin", ...) more than once and if "model.bin" was updated in between. (2) Send "catboost_Evaluate" from the server to the bridge, containing the model path and the features to run the interference on. Step (2) is called multiple times (once per chunk) by the server from function FunctionCatBoostEvaluate::executeImpl(). The library handler for the given model path is expected to be already loaded by Step (1). Fixes #27870
66 lines
2.0 KiB
C++
66 lines
2.0 KiB
C++
#include "ExternalDictionaryLibraryHandlerFactory.h"
|
|
|
|
#include <Common/logger_useful.h>
|
|
|
|
namespace DB
|
|
{
|
|
|
|
ExternalDictionaryLibraryHandlerPtr ExternalDictionaryLibraryHandlerFactory::get(const String & dictionary_id)
|
|
{
|
|
std::lock_guard lock(mutex);
|
|
|
|
if (auto handler = library_handlers.find(dictionary_id); handler != library_handlers.end())
|
|
return handler->second;
|
|
return nullptr;
|
|
}
|
|
|
|
|
|
void ExternalDictionaryLibraryHandlerFactory::create(
|
|
const String & dictionary_id,
|
|
const String & library_path,
|
|
const std::vector<String> & library_settings,
|
|
const Block & sample_block,
|
|
const std::vector<String> & attributes_names)
|
|
{
|
|
std::lock_guard lock(mutex);
|
|
|
|
if (library_handlers.contains(dictionary_id))
|
|
{
|
|
LOG_WARNING(&Poco::Logger::get("ExternalDictionaryLibraryHandlerFactory"), "Library handler with dictionary id {} already exists", dictionary_id);
|
|
return;
|
|
}
|
|
|
|
library_handlers.emplace(std::make_pair(dictionary_id, std::make_shared<ExternalDictionaryLibraryHandler>(library_path, library_settings, sample_block, attributes_names)));
|
|
}
|
|
|
|
|
|
bool ExternalDictionaryLibraryHandlerFactory::clone(const String & from_dictionary_id, const String & to_dictionary_id)
|
|
{
|
|
std::lock_guard lock(mutex);
|
|
auto from_library_handler = library_handlers.find(from_dictionary_id);
|
|
|
|
if (from_library_handler == library_handlers.end())
|
|
return false;
|
|
|
|
/// extDict_libClone method will be called in copy constructor
|
|
library_handlers[to_dictionary_id] = std::make_shared<ExternalDictionaryLibraryHandler>(*from_library_handler->second);
|
|
return true;
|
|
}
|
|
|
|
|
|
bool ExternalDictionaryLibraryHandlerFactory::remove(const String & dictionary_id)
|
|
{
|
|
std::lock_guard lock(mutex);
|
|
/// extDict_libDelete is called in destructor.
|
|
return library_handlers.erase(dictionary_id);
|
|
}
|
|
|
|
|
|
ExternalDictionaryLibraryHandlerFactory & ExternalDictionaryLibraryHandlerFactory::instance()
|
|
{
|
|
static ExternalDictionaryLibraryHandlerFactory instance;
|
|
return instance;
|
|
}
|
|
|
|
}
|