{"id":299869,"date":"2020-03-09T09:00:15","date_gmt":"2020-03-09T09:00:15","guid":{"rendered":"http:\/\/savepearlharbor.com\/?p=299869"},"modified":"-0001-11-30T00:00:00","modified_gmt":"-0001-11-29T21:00:00","slug":"","status":"publish","type":"post","link":"https:\/\/savepearlharbor.com\/?p=299869","title":{"rendered":"\u0414\u0435\u0442\u0435\u043a\u0442\u0438\u0440\u043e\u0432\u0430\u043d\u0438\u0435 \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0439 \u0441 \u043f\u043e\u043c\u043e\u0449\u044c\u044e \u0430\u0432\u0442\u043e\u0435\u043d\u043a\u043e\u0434\u0435\u0440\u043e\u0432 \u043d\u0430 Python"},"content":{"rendered":"\n<div class=\"post__text post__text-html post__text_v1\" id=\"post-content-body\" data-io-article-url=\"https:\/\/habr.com\/ru\/post\/491552\/\">\n<p>\u0414\u0435\u0442\u0435\u043a\u0442\u0438\u0440\u043e\u0432\u0430\u043d\u0438\u0435 \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0439 \u2014 \u0438\u043d\u0442\u0435\u0440\u0435\u0441\u043d\u0430\u044f \u0437\u0430\u0434\u0430\u0447\u0430 \u043c\u0430\u0448\u0438\u043d\u043d\u043e\u0433\u043e \u043e\u0431\u0443\u0447\u0435\u043d\u0438\u044f. \u041d\u0435 \u0441\u0443\u0449\u0435\u0441\u0442\u0432\u0443\u0435\u0442 \u043a\u0430\u043a\u043e\u0433\u043e-\u0442\u043e \u043e\u043f\u0440\u0435\u0434\u0435\u043b\u0435\u043d\u043d\u043e\u0433\u043e \u0441\u043f\u043e\u0441\u043e\u0431\u0430 \u0435\u0435 \u0440\u0435\u0448\u0435\u043d\u0438\u044f, \u0442\u0430\u043a \u043a\u0430\u043a \u043a\u0430\u0436\u0434\u044b\u0439 \u043d\u0430\u0431\u043e\u0440 \u0434\u0430\u043d\u043d\u044b\u0445 \u0438\u043c\u0435\u0435\u0442 \u0441\u0432\u043e\u0438 \u043e\u0441\u043e\u0431\u0435\u043d\u043d\u043e\u0441\u0442\u0438. \u041d\u043e \u0432 \u0442\u043e \u0436\u0435 \u0432\u0440\u0435\u043c\u044f \u0435\u0441\u0442\u044c \u043d\u0435\u0441\u043a\u043e\u043b\u044c\u043a\u043e \u043f\u043e\u0434\u0445\u043e\u0434\u043e\u0432, \u043a\u043e\u0442\u043e\u0440\u044b\u0435 \u043f\u043e\u043c\u043e\u0433\u0430\u044e\u0442 \u0434\u043e\u0431\u0438\u0442\u044c\u0441\u044f \u0443\u0441\u043f\u0435\u0445\u0430. \u042f \u0445\u043e\u0447\u0443 \u0440\u0430\u0441\u0441\u043a\u0430\u0437\u0430\u0442\u044c \u043f\u0440\u043e \u043e\u0434\u0438\u043d \u0438\u0437 \u0442\u0430\u043a\u0438\u0445 \u043f\u043e\u0434\u0445\u043e\u0434\u043e\u0432 \u2014 \u0430\u0432\u0442\u043e\u0435\u043d\u043a\u043e\u0434\u0435\u0440\u044b.<\/p>\n<p><a name=\"habracut\"><\/a>  <\/p>\n<h2 id=\"kakoy-dataset-vybrat\">\u041a\u0430\u043a\u043e\u0439 \u0434\u0430\u0442\u0430\u0441\u0435\u0442 \u0432\u044b\u0431\u0440\u0430\u0442\u044c?<\/h2>\n<p>  <\/p>\n<p>\u0421\u0430\u043c\u044b\u0439 \u0430\u043a\u0442\u0443\u0430\u043b\u044c\u043d\u044b\u0439 \u0432\u043e\u043f\u0440\u043e\u0441 \u0432 \u0436\u0438\u0437\u043d\u0438 \u043b\u044e\u0431\u043e\u0433\u043e \u0434\u0430\u0442\u0430-\u0441\u0430\u0435\u043d\u0442\u0438\u0441\u0442\u0430. \u0427\u0442\u043e\u0431\u044b \u0443\u043f\u0440\u043e\u0441\u0442\u0438\u0442\u044c \u043f\u043e\u0432\u0435\u0441\u0442\u0432\u043e\u0432\u0430\u043d\u0438\u0435, \u044f \u0431\u0443\u0434\u0443 \u0438\u0441\u043f\u043e\u043b\u044c\u0437\u043e\u0432\u0430\u0442\u044c \u043f\u0440\u043e\u0441\u0442\u043e\u0439 \u043f\u043e \u0441\u0442\u0440\u0443\u043a\u0442\u0443\u0440\u0435 \u0434\u0430\u0442\u0430\u0441\u0435\u0442, \u043a\u043e\u0442\u043e\u0440\u044b\u0439 \u0441\u0433\u0435\u043d\u0435\u0440\u0438\u0440\u0443\u0435\u043c \u0437\u0434\u0435\u0441\u044c \u0436\u0435.<\/p>\n<p>  <\/p>\n<pre><code class=\"python\"># \u0438\u043c\u043f\u043e\u0440\u0442\u0438\u0440\u0443\u0435\u043c \u0431\u0438\u0431\u043b\u0438\u043e\u0442\u0435\u043a\u0438 import os import numpy as np from sklearn.model_selection import train_test_split import matplotlib.pyplot as plt from matplotlib.colors import Normalize<\/code><\/pre>\n<p>  <\/p>\n<pre><code class=\"python\"># \u0444\u0443\u043d\u043a\u0446\u0438\u044f \u0434\u043b\u044f \u0433\u0435\u043d\u0435\u0440\u0430\u0446\u0438\u0438 \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u043e\u0433\u043e \u0440\u0430\u0441\u043f\u0440\u0435\u0434\u0435\u043b\u0435\u043d\u0438\u044f \u0441 \u0437\u0430\u0434\u0430\u043d\u043d\u044b\u043c\u0438 \u043f\u0430\u0440\u0430\u043c\u0435\u0442\u0440\u0430\u043c\u0438 def gen_normal_distribution(mu, sigma, size, range=(0, 1), max_val=1):   bins = np.linspace(*range, size)   result = 1 \/ (sigma * np.sqrt(2*np.pi)) * np.exp(-(bins - mu)**2 \/ (2*sigma**2))    cur_max_val = result.max()   k = max_val \/ cur_max_val    result *= k    return result<\/code><\/pre>\n<p>  <\/p>\n<p>\u0420\u0430\u0441\u0441\u043c\u043e\u0442\u0440\u0438\u043c \u043f\u0440\u0438\u043c\u0435\u0440 \u0440\u0430\u0431\u043e\u0442\u044b \u0444\u0443\u043d\u043a\u0446\u0438\u0438. \u0421\u043e\u0437\u0434\u0430\u0434\u0438\u043c \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u043e\u0435 \u0440\u0430\u0441\u043f\u0440\u0435\u0434\u0435\u043b\u0435\u043d\u0438\u0435 \u0441 \u03bc = 0.3 \u0438 \u03c3 = 0.05:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">dist = gen_normal_distribution(0.3, 0.05, 256, max_val=1) print(dist.max()) &gt;&gt;&gt; 1.0 plt.plot(np.linspace(0, 1, 256), dist)<\/code><\/pre>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/g_\/yf\/dk\/g_yfdkkwd3z97n9kpek3k-6mjwk.png\"><\/p>\n<p>  <\/p>\n<p>\u041e\u0431\u044a\u044f\u0432\u0438\u043c \u043f\u0430\u0440\u0430\u043c\u0435\u0442\u0440\u044b \u043d\u0430\u0448\u0435\u0433\u043e \u0434\u0430\u0442\u0430\u0441\u0435\u0442\u0430:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">in_distribution_size = 2000 out_distribution_size = 200 val_size = 100 sample_size = 256  random_generator = np.random.RandomState(seed=42) # \u0434\u043b\u044f \u0432\u043e\u0441\u043f\u0440\u043e\u0438\u0437\u0432\u043e\u0434\u0438\u043c\u043e\u0441\u0442\u0438 \u0438\u0441\u043f\u043e\u043b\u044c\u0437\u0443\u0435\u0442\u0441\u044f seed<\/code><\/pre>\n<p>  <\/p>\n<p>\u0418 \u0444\u0443\u043d\u043a\u0446\u0438\u0438 \u0434\u043b\u044f \u0433\u0435\u043d\u0435\u0440\u0438\u0440\u043e\u0432\u0430\u043d\u0438\u044f \u043f\u0440\u0438\u043c\u0435\u0440\u043e\u0432 \u2014 \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u0438 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0445. \u041d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u044b\u043c\u0438 \u0431\u0443\u0434\u0443\u0442 \u0441\u0447\u0438\u0442\u0430\u0442\u044c\u0441\u044f \u0440\u0430\u0441\u043f\u0440\u0435\u0434\u0435\u043b\u0435\u043d\u0438\u044f \u0441 \u043e\u0434\u043d\u0438\u043c \u043c\u0430\u043a\u0441\u0438\u043c\u0443\u043c\u043e\u043c, \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u043c\u0438 \u2014 \u0441 \u0434\u0432\u0443\u043c\u044f:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">def generate_in_samples(size, sample_size):   global random_generator    in_samples = np.zeros((size, sample_size))    in_mus = random_generator.uniform(0.1, 0.9, size)   in_sigmas = random_generator.uniform(0.05, 0.5, size)    for i in range(size):     in_samples[i] = gen_normal_distribution(in_mus[i], in_sigmas[i], sample_size, max_val=1)    return in_samples  def generate_out_samples(size, sample_size):   global random_generator    # \u0441\u043e\u0437\u0434\u0430\u0435\u043c \u0440\u0430\u0441\u043f\u0440\u0435\u0434\u0435\u043b\u0435\u043d\u0438\u0435 \u0441 \u043e\u0434\u043d\u0438\u043c \u043f\u0438\u043a\u043e\u043c   out_samples = generate_in_samples(size, sample_size)    # \u0438 \u043d\u0430\u043a\u043b\u0430\u0434\u044b\u0432\u0430\u0435\u043c \u043f\u043e\u0432\u0435\u0440\u0445 \u043d\u0435\u0433\u043e \u0435\u0449\u0435 \u043e\u0434\u0438\u043d \u043d\u0435\u0431\u043e\u043b\u044c\u0448\u043e\u0439 \u043c\u0430\u043a\u0441\u0438\u043c\u0443\u043c   out_additional_mus = random_generator.uniform(0.1, 0.9, size)   out_additional_sigmas = random_generator.uniform(0.01, 0.05, size)    for i in range(size):     anomaly = gen_normal_distribution(out_additional_mus[i], out_additional_sigmas[i], sample_size, max_val=0.12)     out_samples[i] += anomaly    return out_samples<\/code><\/pre>\n<p>  <\/p>\n<p>\u0422\u0430\u043a \u0432\u044b\u0433\u043b\u044f\u0434\u0438\u0442 \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u044b\u0439 \u043f\u0440\u0438\u043c\u0435\u0440:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">in_samples = generate_in_samples(in_distribution_size, sample_size) plt.plot(np.linspace(0, 1, sample_size), in_samples[42])<\/code><\/pre>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/f2\/zs\/zo\/f2zszowdiozndeblbr9y174adgi.png\"><\/p>\n<p>  <\/p>\n<p>\u0410 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0439 \u2014 \u0442\u0430\u043a:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">out_samples = generate_out_samples(out_distribution_size, sample_size) plt.plot(np.linspace(0, 1, sample_size), out_samples[42])<\/code><\/pre>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/em\/bd\/7b\/embd7b623honrwk-7ofxbvk6_yg.png\"><\/p>\n<p>  <\/p>\n<p>\u0421\u043e\u0437\u0434\u0430\u0434\u0438\u043c \u043c\u0430\u0441\u0441\u0438\u0432\u044b \u0441 \u043f\u0440\u0438\u0437\u043d\u0430\u043a\u0430\u043c\u0438 \u0438 \u043c\u0435\u0442\u043a\u0430\u043c\u0438:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">x = np.concatenate((in_samples, out_samples)) # \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u044b\u0435 \u043f\u0440\u0438\u043c\u0435\u0440\u044b \u0438\u043c\u0435\u044e\u0442 \u043c\u0435\u0442\u043a\u0443 0, \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0435 -- 1 y = np.concatenate((np.zeros(in_distribution_size), np.ones(out_distribution_size)))  # \u0440\u0430\u0437\u0434\u0435\u043b\u0435\u043d\u0438\u0435 \u043d\u0430 \u043e\u0431\u0443\u0447\u0430\u044e\u0449\u0443\u044e\/\u0442\u0440\u0435\u043d\u0438\u0440\u043e\u0432\u043e\u0447\u043d\u0443\u044e \u0432\u044b\u0431\u043e\u0440\u043a\u0438 x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.2, shuffle=True, random_state=42)<\/code><\/pre>\n<p>  <\/p>\n<p>\u0418 \u043f\u043e\u0437\u0432\u043e\u043b\u044e \u0441\u0435\u0431\u0435 \u043d\u0435\u043c\u043d\u043e\u0433\u043e \u0441\u0445\u0438\u0442\u0440\u0438\u0442\u044c. \u0414\u043b\u044f \u0430\u043b\u0433\u043e\u0440\u0438\u0442\u043c\u043e\u0432, \u0440\u0430\u0431\u043e\u0442\u0430\u044e\u0449\u0438\u0445 \u0432 \u0434\u0432\u0435 \u0441\u0442\u0430\u0434\u0438\u0438, \u043f\u043e\u0442\u0440\u0435\u0431\u0443\u0435\u0442\u0441\u044f 2 \u043e\u0431\u0443\u0447\u0430\u044e\u0449\u0438\u0445 \u0432\u044b\u0431\u043e\u0440\u043a\u0438 \u0438 1 \u0442\u0435\u0441\u0442\u043e\u0432\u0430\u044f (\u0432\u0430\u043b\u0438\u0434\u0430\u0446\u0438\u043e\u043d\u043d\u0430\u044f). \u041f\u0440\u0438 \u0440\u0430\u0431\u043e\u0442\u0435 \u0441 \u0440\u0435\u0430\u043b\u044c\u043d\u044b\u043c\u0438 \u0434\u0430\u043d\u043d\u044b\u043c\u0438 \u043f\u0440\u0438\u0448\u043b\u043e\u0441\u044c \u0431\u044b \u0443\u0440\u0435\u0437\u0430\u0442\u044c \u0438\u043c\u0435\u044e\u0449\u0438\u0435\u0441\u044f \u0442\u0440\u0435\u043d\u0438\u0440\u043e\u0432\u043e\u0447\u043d\u0443\u044e \u0438 \u0442\u0435\u0441\u0442\u043e\u0432\u0443\u044e \u0432\u044b\u0431\u043e\u0440\u043a\u0438, \u043d\u043e \u043c\u044b \u043c\u043e\u0436\u0435\u0448\u044c \u0434\u043e\u0433\u0435\u043d\u0435\u0440\u0438\u0440\u043e\u0432\u0430\u0442\u044c \u0434\u0430\u043d\u043d\u044b\u0435 (\u0432 \u0431\u043e\u043b\u044c\u0448\u0438\u0445 \u043a\u043e\u043b\u0438\u0447\u0435\u0441\u0442\u0432\u0430\u0445), \u0447\u0442\u043e\u0431\u044b \u0431\u043e\u043b\u0435\u0435 \u043e\u0431\u044a\u0435\u043a\u0442\u0438\u0432\u043d\u043e \u043e\u0446\u0435\u043d\u0438\u0442\u044c \u043a\u0430\u0447\u0435\u0441\u0442\u0432\u043e \u043c\u043e\u0434\u0435\u043b\u0438:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\"># \u0441\u043e\u0437\u0434\u0430\u0435\u043c \u0432\u0430\u043b\u0438\u0434\u0430\u0446\u0438\u043e\u043d\u043d\u0443\u044e \u0432\u044b\u0431\u043e\u0440\u043a\u0443 \u0438\u0437 100 \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u0438 100 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u043f\u0440\u0438\u043c\u0435\u0440\u043e\u0432 x_val_out = generate_out_samples(val_size, sample_size) x_val_in = generate_in_samples(val_size, sample_size)  x_val = np.concatenate((x_val_out, x_val_in)) y_val = np.concatenate((np.ones(val_size), np.zeros(val_size)))<\/code><\/pre>\n<p>  <\/p>\n<h2 id=\"modeli\">\u041c\u043e\u0434\u0435\u043b\u0438<\/h2>\n<p>  <\/p>\n<p>\u041f\u0435\u0440\u0435\u0434 \u0442\u0435\u043c, \u043a\u0430\u043a \u043f\u0435\u0440\u0435\u0439\u0442\u0438 \u043a \u0442\u0435\u043c\u0435 \u0441\u0442\u0430\u0442\u044c\u0438, \u043f\u0440\u043e\u0432\u0435\u0440\u0438\u043c \u043d\u0435\u0441\u043a\u043e\u043b\u044c\u043a\u043e \u0430\u043b\u0433\u043e\u0440\u0438\u0442\u043c\u043e\u0432 \u0434\u0435\u0442\u0435\u043a\u0442\u0438\u0440\u043e\u0432\u0430\u043d\u0438\u044f \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0439 \u0438\u0437 Sklearn: \u043e\u0434\u043d\u043e\u043a\u043b\u0430\u0441\u0441\u043e\u0432\u044b\u0439 SVM \u0438 \u0438\u0437\u043e\u043b\u0438\u0440\u0443\u044e\u0449\u0438\u0439 \u043b\u0435\u0441. \u042d\u0442\u043e \u0430\u043b\u0433\u043e\u0440\u0438\u0442\u043c\u044b \u043e\u0431\u0443\u0447\u0435\u043d\u0438\u044f \u0431\u0435\u0437 \u0443\u0447\u0438\u0442\u0435\u043b\u044f, \u0442\u043e \u0435\u0441\u0442\u044c \u043e\u043d\u0438 \u043e\u0440\u0438\u0435\u043d\u0442\u0438\u0440\u0443\u044e\u0442\u0441\u044f \u0442\u043e\u043b\u044c\u043a\u043e \u043d\u0430 \u043f\u0440\u0435\u0434\u0441\u0442\u0430\u0432\u043b\u0435\u043d\u0438\u0435 \u0434\u0430\u043d\u043d\u044b\u0445, \u043d\u043e \u043d\u0435 \u043d\u0430 \u043e\u0431\u0443\u0447\u0430\u044e\u0449\u0438\u0435 \u043c\u0435\u0442\u043a\u0438.<\/p>\n<p>  <\/p>\n<pre><code class=\"python\"># \u0444\u0443\u043d\u043a\u0446\u0438\u0438 \u0434\u043b\u044f \u043e\u0446\u0435\u043d\u043a\u0438 \u043a\u0430\u0447\u0435\u0441\u0442\u0432\u0430 \u043c\u043e\u0434\u0435\u043b\u0435\u0439 from sklearn.metrics import classification_report from sklearn.metrics import f1_score<\/code><\/pre>\n<p>  <\/p>\n<h3 id=\"one-class-svm\">One class SVM<\/h3>\n<p>  <\/p>\n<pre><code class=\"python\">from sklearn.svm import OneClassSVM<\/code><\/pre>\n<p>  <\/p>\n<p>OneClassSVM \u043f\u043e\u0437\u0432\u043e\u043b\u044f\u0435\u0442 \u0437\u0430\u0434\u0430\u0442\u044c \u043f\u0430\u0440\u0430\u043c\u0435\u0442\u0440 <code>nu<\/code> \u2014 \u0434\u043e\u043b\u044e \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u043e\u0431\u044a\u0435\u043a\u0442\u043e\u0432 \u0432 \u0432\u044b\u0431\u043e\u0440\u043a\u0435.<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">out_dist_part = out_distribution_size \/ (out_distribution_size + in_distribution_size) svm = OneClassSVM(nu=out_dist_part) svm.fit(x_train, y_train) &gt;&gt;&gt; OneClassSVM(cache_size=200, coef0=0.0, degree=3, gamma='scale', kernel='rbf',             max_iter=-1, nu=0.09090909090909091, shrinking=True, tol=0.001,             verbose=False)<\/code><\/pre>\n<p>  <\/p>\n<p>\u0414\u0435\u043b\u0430\u0435\u043c \u043f\u0440\u0435\u0434\u0441\u043a\u0430\u0437\u0430\u043d\u0438\u044f \u043d\u0430 \u0442\u0435\u0441\u0442\u043e\u0432\u043e\u043c \u043d\u0430\u0431\u043e\u0440\u0435:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">svm_prediction = svm.predict(x_val) svm_prediction[svm_prediction == 1] = 0 svm_prediction[svm_prediction == -1] = 1<\/code><\/pre>\n<p>  <\/p>\n<p>\u0412 sklearn \u0435\u0441\u0442\u044c \u043e\u0447\u0435\u043d\u044c \u0443\u0434\u043e\u0431\u043d\u0430\u044f \u0444\u0443\u043d\u043a\u0446\u0438\u044f \u2014 <code>classification_report<\/code>, \u043e\u043d\u0430 \u043f\u043e\u0437\u0432\u043e\u043b\u044f\u0435\u0442 \u043e\u0446\u0435\u043d\u0438\u0442\u044c \u0442\u0430\u043a\u0438\u0435 \u0432\u0430\u0436\u043d\u044b\u0435 \u0434\u043b\u044f Anomaly detection \u043c\u0435\u0442\u0440\u0438\u043a\u0438, \u043a\u0430\u043a precision \u0438 recall, \u043f\u0440\u0438\u0447\u0435\u043c \u0434\u043b\u044f \u043a\u0430\u0436\u0434\u043e\u0433\u043e \u043a\u043b\u0430\u0441\u0441\u0430:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">print(classification_report(y_val, svm_prediction))  &gt;&gt;&gt;           precision    recall  f1-score   support           0.0       0.57      0.93      0.70       100          1.0       0.81      0.29      0.43       100      accuracy                           0.61       200    macro avg       0.69      0.61      0.57       200 weighted avg       0.69      0.61      0.57       200 <\/code><\/pre>\n<p>  <\/p>\n<p>\u041d\u0443, \u0442\u0430\u043a\u043e\u0435. \u0414\u043e\u0432\u043e\u043b\u044c\u043d\u043e \u043d\u0438\u0437\u043a\u0438\u0439 f1-score \u0441\u0432\u0438\u0434\u0435\u0442\u0435\u043b\u044c\u0441\u0442\u0432\u0443\u0435\u0442 \u043e \u0442\u043e\u043c, \u0447\u0442\u043e \u043c\u043e\u0434\u0435\u043b\u044c \u043f\u043b\u043e\u0445\u043e \u0441\u043f\u0440\u0430\u0432\u043b\u044f\u0435\u0442\u0441\u044f \u0441 \u0437\u0430\u0434\u0430\u0447\u0435\u0439.<\/p>\n<p>  <\/p>\n<h3 id=\"isolation-forest\">Isolation forest<\/h3>\n<p>  <\/p>\n<p>\u041e\u043a\u0435\u0439, \u043c\u043e\u0436\u0435\u0442 \u0431\u044b\u0442\u044c, \u0431\u0440\u0430\u0442-\u0431\u043b\u0438\u0437\u043d\u0435\u0446 \u0441\u043b\u0443\u0447\u0430\u0439\u043d\u043e\u0433\u043e \u043b\u0435\u0441\u0430 \u0441\u043c\u043e\u0436\u0435\u0442 \u043b\u0443\u0447\u0448\u0435 \u0441\u043f\u0440\u0430\u0432\u0438\u0442\u044c\u0441\u044f \u0441 \u0437\u0430\u0434\u0430\u0447\u0435\u0439?<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">from sklearn.ensemble import IsolationForest<\/code><\/pre>\n<p>  <\/p>\n<p>\u0423 \u0438\u0437\u043e\u043b\u0438\u0440\u0443\u044e\u0449\u0435\u0433\u043e \u043b\u0435\u0441\u0430 \u0442\u043e\u0436\u0435 \u0435\u0441\u0442\u044c \u043f\u0430\u0440\u0430\u043c\u0435\u0442\u0440, \u043e\u0442\u0432\u0435\u0447\u0430\u044e\u0449\u0438\u0439 \u0437\u0430 \u0434\u043e\u043b\u044e \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u043e\u0431\u044a\u0435\u043a\u0442\u043e\u0432 \u0432 \u0432\u044b\u0431\u043e\u0440\u043a\u0435. \u0417\u0430\u0434\u0430\u0434\u0438\u043c \u0435\u0433\u043e:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">out_dist_part = out_distribution_size \/ (out_distribution_size + in_distribution_size)  iso_forest = IsolationForest(n_estimators=100, contamination=out_dist_part, max_features=100, n_jobs=-1) iso_forest.fit(x_train) &gt;&gt;&gt; IsolationForest(behaviour='deprecated', bootstrap=False,                 contamination=0.09090909090909091, max_features=100,                 max_samples='auto', n_estimators=100, n_jobs=-1,                 random_state=None, verbose=0, warm_start=False)<\/code><\/pre>\n<p>  <\/p>\n<p>Classification report? \u2014 Classification report!<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">iso_forest_prediction = iso_forest.predict(x_val) iso_forest_prediction[iso_forest_prediction == 1] = 0 iso_forest_prediction[iso_forest_prediction == -1] = 1  print(classification_report(y_val, iso_forest_prediction)) &gt;&gt;&gt;            precision    recall  f1-score   support           0.0       0.50      0.91      0.65       100          1.0       0.53      0.10      0.17       100      accuracy                           0.51       200    macro avg       0.51      0.51      0.41       200 weighted avg       0.51      0.51      0.41       200<\/code><\/pre>\n<p>  <\/p>\n<h3 id=\"randomforestclassifier\">RandomForestClassifier<\/h3>\n<p>  <\/p>\n<p>\u041a\u043e\u043d\u0435\u0447\u043d\u043e, \u0432\u0441\u0435\u0433\u0434\u0430 \u0441\u0442\u043e\u0438\u0442 \u043f\u0440\u043e\u0432\u0435\u0440\u044f\u0442\u044c \u043a\u0430\u043a\u043e\u0439-\u043d\u0438\u0431\u0443\u0434\u044c \u043f\u0440\u043e\u0441\u0442\u043e\u0439 \u0432\u0430\u0440\u0438\u0430\u043d\u0442 \u0442\u0438\u043f\u0430 &quot;\u0430 \u0432\u0434\u0440\u0443\u0433 \u0437\u0430\u0434\u0430\u0447\u0430 \u043a\u043b\u0430\u0441\u0441\u0438\u0444\u0438\u043a\u0430\u0446\u0438\u0438 \u043e\u043a\u0430\u0436\u0435\u0442\u0441\u044f \u043f\u043e\u0441\u0438\u043b\u044c\u043d\u043e\u0439 \u0434\u043b\u044f \u0441\u043b\u0443\u0447\u0430\u0439\u043d\u043e\u0433\u043e \u043b\u0435\u0441\u0430?&quot; \u0427\u0442\u043e \u0436, \u043f\u043e\u043f\u0440\u043e\u0431\u0443\u0435\u043c:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">from sklearn.ensemble import RandomForestClassifier  random_forest = RandomForestClassifier(n_estimators=100, max_features=100, n_jobs=-1) random_forest.fit(x_train, y_train) &gt;&gt;&gt; RandomForestClassifier(bootstrap=True, ccp_alpha=0.0, class_weight=None,                        criterion='gini', max_depth=None, max_features=100,                        max_leaf_nodes=None, max_samples=None,                        min_impurity_decrease=0.0, min_impurity_split=None,                        min_samples_leaf=1, min_samples_split=2,                        min_weight_fraction_leaf=0.0, n_estimators=100,                        n_jobs=-1, oob_score=False, random_state=None, verbose=0,                        warm_start=False)<\/code><\/pre>\n<p>  <\/p>\n<pre><code class=\"python\">random_forest_prediction = random_forest.predict(x_val) print(classification_report(y_val, random_forest_prediction)) &gt;&gt;&gt;            precision    recall  f1-score   support           0.0       0.57      0.99      0.72       100          1.0       0.96      0.25      0.40       100      accuracy                           0.62       200    macro avg       0.77      0.62      0.56       200 weighted avg       0.77      0.62      0.56       200<\/code><\/pre>\n<p>  <\/p>\n<h3 id=\"autoencoder\">Autoencoder<\/h3>\n<p>  <\/p>\n<p>\u0412 \u043e\u0431\u0449\u0435\u043c, \u0441\u043b\u0443\u0447\u0438\u043b\u0430\u0441\u044c \u0434\u043e\u0432\u043e\u043b\u044c\u043d\u043e \u043e\u0431\u044b\u0434\u0435\u043d\u043d\u0430\u044f \u0434\u043b\u044f \u0434\u0435\u0442\u0435\u043a\u0442\u0438\u0440\u043e\u0432\u0430\u043d\u0438\u044f \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0439 \u0448\u0442\u0443\u043a\u0430: \u043d\u0438\u0447\u0435\u0433\u043e \u043d\u0435 \u0441\u0440\u0430\u0431\u043e\u0442\u0430\u043b\u043e. \u041f\u0435\u0440\u0435\u0445\u043e\u0434\u0438\u043c \u043a \u0430\u0432\u0442\u043e\u043a\u043e\u0434\u0438\u0440\u043e\u0432\u0449\u0438\u043a\u0430\u043c.<\/p>\n<p>  <\/p>\n<p>\u041f\u0440\u0438\u043d\u0446\u0438\u043f \u0440\u0430\u0431\u043e\u0442\u044b \u0430\u0432\u0442\u043e\u0435\u043d\u043a\u043e\u0434\u0435\u0440\u0430 \u0441\u043e\u0441\u0442\u043e\u0438\u0442 \u0432 \u0442\u043e\u043c, \u0447\u0442\u043e \u043c\u043e\u0434\u0435\u043b\u044c \u043f\u044b\u0442\u0430\u0435\u0442\u0441\u044f \u0441\u043d\u0430\u0447\u0430\u043b\u0430 &quot;\u0441\u0436\u0430\u0442\u044c&quot; \u0434\u0430\u043d\u043d\u044b\u0435, \u0430 \u043f\u043e\u0442\u043e\u043c \u0432\u043e\u0441\u0441\u0442\u0430\u043d\u043e\u0432\u0438\u0442\u044c \u0438\u0445. \u041e\u043d\u0430 \u0441\u043e\u0441\u0442\u043e\u0438\u0442 \u0438\u0437 2 \u0447\u0430\u0441\u0442\u0435\u0439: Encoder&#8217;\u0430 \u0438 Decoder&#8217;\u0430, \u043a\u043e\u0442\u043e\u0440\u044b\u0435 \u0437\u0430\u043d\u0438\u043c\u0430\u044e\u0442\u0441\u044f \u0441\u0436\u0430\u0442\u0438\u0435\u043c \u0438 \u0440\u0430\u0441\u0448\u0438\u0444\u0440\u043e\u0432\u043a\u043e\u0439 \u0441\u043e\u043e\u0442\u0432\u0435\u0442\u0441\u0442\u0432\u0435\u043d\u043d\u043e.<\/p>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/kv\/bx\/mt\/kvbxmtkqeauexgdruqi52u6dhs8.png\"><br \/>  <em>\u0418\u0437\u043e\u0431\u0440\u0430\u0436\u0435\u043d\u0438\u0435 \u0432\u0437\u044f\u0442\u043e \u0438\u0437 <a href=\"https:\/\/towardsdatascience.com\/applied-deep-learning-part-3-autoencoders-1c083af4d798#a265\" rel=\"nofollow\">\u0441\u0442\u0430\u0442\u044c\u0438<\/a><\/em><\/p>\n<p>  <\/p>\n<p>\u0427\u0442\u043e\u0431\u044b \u0441\u043e\u0437\u0434\u0430\u0442\u044c \u044d\u0444\u0444\u0435\u043a\u0442 &quot;\u0441\u0436\u0430\u0442\u0438\u044f&quot; \u0434\u0430\u043d\u043d\u044b\u0445, \u043d\u0430\u0434\u043e \u0443\u043c\u0435\u043d\u044c\u0448\u0438\u0442\u044c \u0441\u043a\u0440\u044b\u0442\u044b\u0435 \u0441\u043b\u043e\u0438, \u0437\u0430\u0441\u0442\u0430\u0432\u0438\u0432 \u0438\u0445 \u043e\u0431\u0440\u0430\u0431\u0430\u0442\u044b\u0432\u0430\u0442\u044c \u0431\u043e\u043b\u044c\u0448\u0435\u0435 \u043a\u043e\u043b\u0438\u0447\u0435\u0441\u0442\u0432\u043e \u0438\u043d\u0444\u043e\u0440\u043c\u0430\u0446\u0438\u0438 \u043c\u0435\u043d\u044c\u0448\u0438\u043c \u043a\u043e\u043b\u0438\u0447\u0435\u0441\u0442\u0432\u043e\u043c \u043d\u0435\u0439\u0440\u043e\u043d\u043e\u0432.<\/p>\n<p>  <\/p>\n<p>\u041a\u0430\u043a \u044d\u0442\u043e\u0442 \u044d\u0444\u0444\u0435\u043a\u0442 \u043f\u043e\u043c\u043e\u0436\u0435\u0442 \u043d\u0430\u043c? \u0422\u0435\u043e\u0440\u0438\u044f \u0438\u043d\u0444\u043e\u0440\u043c\u0430\u0446\u0438\u0438 \u0433\u043e\u0432\u043e\u0440\u0438\u0442 \u043e \u0442\u043e\u043c, \u0447\u0442\u043e \u0447\u0435\u043c \u0431\u043e\u043b\u0435\u0435 \u0432\u0435\u0440\u043e\u044f\u0442\u043d\u043e \u0441\u043e\u0431\u044b\u0442\u0438\u0435, \u0442\u0435\u043c \u043c\u0435\u043d\u044c\u0448\u0435\u0435 \u043a\u043e\u043b\u0438\u0447\u0435\u0441\u0442\u0432\u043e \u0438\u043d\u0444\u043e\u0440\u043c\u0430\u0446\u0438\u0438 \u043f\u043e\u0442\u0440\u0435\u0431\u0443\u0435\u0442\u0441\u044f, \u0447\u0442\u043e\u0431\u044b \u043e\u043f\u0438\u0441\u0430\u0442\u044c \u044d\u0442\u043e \u0441\u043e\u0431\u044b\u0442\u0438\u0435. \u0412\u0441\u043f\u043e\u043c\u043d\u0438\u043c, \u0447\u0442\u043e \u0443 \u043d\u0430\u0441 \u0432\u0441\u0435\u0433\u043e \u043b\u0438\u0448\u044c 9% \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0439 \u0438 91% \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u043e\u0431\u044a\u0435\u043a\u0442\u043e\u0432. \u0422\u043e\u0433\u0434\u0430 \u0434\u043b\u044f \u0445\u0440\u0430\u043d\u0435\u043d\u0438\u044f \u0438\u043d\u0444\u043e\u0440\u043c\u0430\u0446\u0438\u0438 \u043e\u0431 \u043e\u0431\u044b\u0447\u043d\u044b\u0445 \u043e\u0431\u044a\u0435\u043a\u0442\u0430\u0445 \u043f\u043e\u0442\u0440\u0435\u0431\u0443\u0435\u0442\u0441\u044f \u043c\u0435\u043d\u044c\u0448\u0435 \u0438\u043d\u0444\u043e\u0440\u043c\u0430\u0446\u0438\u0438, \u0447\u0435\u043c \u0434\u043b\u044f \u0437\u0430\u043f\u043e\u043c\u0438\u043d\u0430\u043d\u0438\u044f \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0445. \u041d\u043e \u0442\u043e\u0433\u0434\u0430, \u0435\u0441\u043b\u0438 \u043c\u044b \u043f\u043e\u0434\u0431\u0435\u0440\u0435\u043c \u043f\u0440\u0430\u0432\u0438\u043b\u044c\u043d\u044b\u0435 \u043f\u0430\u0440\u0430\u043c\u0435\u0442\u0440\u044b \u043d\u0435\u0439\u0440\u043e\u043d\u043d\u043e\u0439 \u0441\u0435\u0442\u0438, \u0442\u043e \u043e\u043d\u0430 \u0441\u043c\u043e\u0436\u0435\u0442 \u0437\u0430\u043f\u043e\u043c\u0438\u043d\u0430\u0442\u044c \u0438 \u0432\u043e\u0441\u0441\u0442\u0430\u043d\u0430\u0432\u043b\u0438\u0432\u0430\u0442\u044c <strong>\u0442\u043e\u043b\u044c\u043a\u043e<\/strong> \u043e\u0431\u044b\u0447\u043d\u044b\u0435 \u043e\u0431\u044a\u0435\u043a\u0442\u044b: \u043d\u0430 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0435 \u0435\u0439 \u043f\u0440\u043e\u0441\u0442\u043e \u043d\u0435 \u0431\u0443\u0434\u0435\u0442 \u0445\u0432\u0430\u0442\u0430\u0442\u044c \u043e\u0431\u043e\u0431\u0449\u0430\u044e\u0449\u0435\u0439 \u0441\u043f\u043e\u0441\u043e\u0431\u043d\u043e\u0441\u0442\u0438.<\/p>\n<p>  <\/p>\n<p>\u041f\u043e\u044d\u0442\u043e\u043c\u0443 \u0432\u043e\u0441\u0441\u0442\u0430\u043d\u043e\u0432\u043b\u0435\u043d\u043d\u044b\u0435 \u043c\u043e\u0434\u0435\u043b\u044c\u044e \u0434\u0430\u043d\u043d\u044b\u0435 \u0431\u0443\u0434\u0443\u0442 <em>\u0437\u043d\u0430\u0447\u0438\u0442\u0435\u043b\u044c\u043d\u043e<\/em> \u043e\u0442\u043b\u0438\u0447\u0430\u0442\u044c\u0441\u044f \u043e\u0442 \u0438\u0441\u0445\u043e\u0434\u043d\u044b\u0445.<\/p>\n<p>  <\/p>\n<p>\u041c\u043e\u0434\u0435\u043b\u044c \u043d\u0430\u043f\u0438\u0448\u0435\u043c \u043d\u0430 PyTorch, \u043f\u043e\u044d\u0442\u043e\u043c\u0443 \u0438\u043c\u043f\u043e\u0440\u0442\u0438\u0440\u0443\u0435\u043c \u043d\u0443\u0436\u043d\u044b\u0435 \u043c\u043e\u0434\u0443\u043b\u0438:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">import torch from torch import nn from torch.utils.data import TensorDataset, DataLoader from torch.optim import Adam<\/code><\/pre>\n<p>  <\/p>\n<p>\u0417\u0430\u0434\u0430\u0435\u043c \u0433\u0438\u043f\u0435\u0440\u043f\u0430\u0440\u0430\u043c\u0435\u0442\u0440\u044b \u043c\u043e\u0434\u0435\u043b\u0438:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">batch_size = 32 lr = 1e-3<\/code><\/pre>\n<p>  <\/p>\n<p>\u041e\u0447\u0435\u043d\u044c \u0432\u0430\u0436\u043d\u0430\u044f \u0447\u0430\u0441\u0442\u044c. \u0422\u0440\u0435\u043d\u0438\u0440\u043e\u0432\u043e\u0447\u043d\u044b\u0439 \u0434\u0430\u0442\u0430\u0441\u0435\u0442 \u0434\u043e\u043b\u0436\u0435\u043d \u0441\u043e\u0441\u0442\u043e\u044f\u0442\u044c \u0442\u043e\u043b\u044c\u043a\u043e \u0438\u0437 \u043e\u0431\u044b\u0447\u043d\u044b\u0445 \u043e\u0431\u044a\u0435\u043a\u0442\u043e\u0432, \u043f\u043e\u0442\u043e\u043c\u0443 \u0447\u0442\u043e \u0435\u0441\u043b\u0438 \u043c\u043e\u0434\u0435\u043b\u044c \u0443\u0432\u0438\u0434\u0438\u0442 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0435, \u0442\u043e \u043e\u043d\u0430 \u043c\u043e\u0436\u0435\u0442 &quot;\u043f\u043e\u0434\u0441\u0442\u0440\u043e\u0438\u0442\u044c\u0441\u044f&quot; \u043f\u043e\u0434 \u0438\u0445 \u0432\u043e\u0441\u0441\u0442\u0430\u043d\u043e\u0432\u043b\u0435\u043d\u0438\u0435 \u0438 \u043c\u044b \u043d\u0435 \u0441\u043c\u043e\u0436\u0435\u043c \u043d\u0430\u0439\u0442\u0438 \u0437\u043d\u0430\u0447\u0438\u0442\u0435\u043b\u044c\u043d\u044b\u0435 \u043e\u0442\u043b\u0438\u0447\u0438\u044f \u0432 \u0438\u0441\u0445\u043e\u0434\u043d\u044b\u0445 \u0438 \u0432\u043e\u0441\u0441\u0442\u0430\u043d\u043e\u0432\u043b\u0435\u043d\u043d\u044b\u0445 \u0434\u0430\u043d\u043d\u044b\u0445. \u041a\u043e\u043d\u0435\u0447\u043d\u043e, \u043a\u043e\u0433\u0434\u0430 \u0433\u0438\u043f\u0435\u0440\u043f\u0430\u0440\u0430\u043c\u0435\u0442\u0440\u044b \u043c\u043e\u0434\u0435\u043b\u0438 (\u043a\u043e\u043b\u0438\u0447\u0435\u0441\u0442\u0432\u043e \u043d\u0435\u0439\u0440\u043e\u043d\u043e\u0432 \u0432 \u0441\u043a\u0440\u044b\u0442\u044b\u0445 \u0441\u043b\u043e\u044f\u0445, learning rate, \u0440\u0430\u0437\u043c\u0435\u0440 \u0431\u0430\u0442\u0447\u0430) \u043f\u043e\u0434\u043e\u0431\u0440\u0430\u043d\u044b \u043f\u0440\u0430\u0432\u0438\u043b\u044c\u043d\u043e, \u0442\u043e \u0448\u0430\u043d\u0441 \u043d\u0435\u0443\u0434\u0430\u0447\u0438 \u043f\u043e\u043d\u0438\u0436\u0430\u0435\u0442\u0441\u044f, \u043d\u043e \u043b\u0443\u0447\u0448\u0435 \u043f\u0435\u0440\u0435\u0441\u0442\u0440\u0430\u0445\u043e\u0432\u0430\u0442\u044c\u0441\u044f.<\/p>\n<p>  <\/p>\n<pre><code class=\"python\"># \u0431\u0435\u0440\u0435\u043c \u0442\u043e\u043b\u044c\u043a\u043e \u0442\u0435 \u043f\u0440\u0438\u043c\u0435\u0440\u044b \u0438\u0437 x_train, \u043a\u043e\u0442\u043e\u0440\u044b\u0435 \u043d\u0435 \u044f\u0432\u043b\u044f\u044e\u0442\u0441\u044f \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u043c\u0438 train_in_distribution = x_train[y_train == 0] train_in_distribution = torch.tensor(train_in_distribution.astype(np.float32))  train_in_dataset = TensorDataset(train_in_distribution) train_in_loader = DataLoader(train_in_dataset, batch_size=batch_size, shuffle=True)  # \u0434\u043b\u044f \u0442\u0435\u0441\u0442\u0430 \u0438 \u0432\u0430\u043b\u0438\u0434\u0430\u0446\u0438\u0438 \u0432\u043e\u0437\u044c\u043c\u0435\u043c \u0432\u0441\u0435 \u043f\u0440\u0438\u043c\u0435\u0440\u044b, \u0447\u0442\u043e\u0431\u044b \u043c\u043e\u0436\u043d\u043e \u0431\u044b\u043b\u043e \u0441\u0440\u0430\u0432\u043d\u0438\u0432\u0430\u0442\u044c \u0440\u0430\u0431\u043e\u0442\u0443 \u043c\u043e\u0434\u0435\u043b\u0438 \u043d\u0430 \u043e\u0431\u044b\u0447\u043d\u044b\u0445 \u0438 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u0434\u0430\u043d\u043d\u044b\u0445 test_dataset = TensorDataset(     torch.tensor(x_test.astype(np.float32)),     torch.tensor(y_test.astype(np.long)) ) test_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=False)  val_dataset = TensorDataset(torch.tensor(x_val.astype(np.float32))) val_loader = DataLoader(val_dataset, batch_size=batch_size, shuffle=False)<\/code><\/pre>\n<p>  <\/p>\n<p>\u041c\u043e\u0434\u0435\u043b\u044c\u043a\u0430. \u0421\u043a\u0440\u044b\u0442\u044b\u0439 \u0441\u043b\u043e\u0439 \u0431\u0443\u0434\u0435\u0442 \u0438\u043c\u0435\u0442\u044c \u0440\u0430\u0437\u043c\u0435\u0440 4 \u043d\u0435\u0439\u0440\u043e\u043d\u0430, \u0447\u0442\u043e\u0431\u044b \u0441\u044b\u043c\u0438\u0442\u0438\u0440\u043e\u0432\u0430\u0442\u044c \u043d\u0435\u043e\u043f\u0442\u0438\u043c\u0430\u043b\u044c\u043d\u044b\u0435 \u043f\u0430\u0440\u0430\u043c\u0435\u0442\u0440\u044b \u043c\u043e\u0434\u0435\u043b\u0438 (\u0434\u043b\u044f \u043e\u0431\u044b\u0447\u043d\u044b\u0445 \u043f\u0440\u0438\u043c\u0435\u0440\u043e\u0432 \u043d\u0430\u043c \u0434\u043e\u0441\u0442\u0430\u0442\u043e\u0447\u043d\u043e 2 \u043d\u0435\u0439\u0440\u043e\u043d\u043e\u0432: \u043e\u0434\u0438\u043d \u0434\u043b\u044f \u0437\u0430\u0434\u0430\u043d\u0438\u044f \u03bc, \u0434\u0440\u0443\u0433\u043e\u0439 \u2014 \u0434\u043b\u044f \u03c3, \u0442\u0430\u043a \u043c\u044b \u043f\u043e\u043b\u043d\u043e\u0441\u0442\u044c\u044e \u043e\u043f\u0438\u0448\u0435\u043c \u0440\u0430\u0441\u043f\u0440\u0435\u0434\u0435\u043b\u0435\u043d\u0438\u0435; \u043d\u043e \u0432\u043e\u0442 4 \u043d\u0435\u0439\u0440\u043e\u043d\u0430 \u043c\u043e\u0433\u0443\u0442 \u0432\u043e\u0441\u0441\u0442\u0430\u043d\u043e\u0432\u0438\u0442\u044c \u0440\u0430\u0441\u043f\u0440\u0435\u0434\u0435\u043b\u0435\u043d\u0438\u0435 \u0441 \u0434\u0432\u0443\u043c\u044f \u043c\u0430\u043a\u0441\u0438\u043c\u0443\u043c\u0430\u043c\u0438, \u043a\u0430\u043a \u0432 \u043d\u0430\u0448\u0438\u0445 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u043f\u0440\u0438\u043c\u0435\u0440\u0430\u0445).<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">class Autoencoder(nn.Module):   def __init__(self, input_size):     super(Autoencoder, self).__init__()     self.encoder = nn.Sequential(       nn.Linear(input_size, 128),       nn.LeakyReLU(0.2),       nn.Linear(128, 64),       nn.LeakyReLU(0.2),       nn.Linear(64, 16),       nn.LeakyReLU(0.2),       nn.Linear(16, 4),       nn.LeakyReLU(0.2),     )     self.decoder = nn.Sequential(       nn.Linear(4, 16),       nn.LeakyReLU(0.2),       nn.Linear(16, 64),       nn.LeakyReLU(0.2),       nn.Linear(64, 128),       nn.LeakyReLU(0.2),       nn.Linear(128, 256),       nn.LeakyReLU(0.2),     )    def forward(self, x):     x = self.encoder(x)     x = self.decoder(x)     return x<\/code><\/pre>\n<p>  <\/p>\n<pre><code class=\"python\">model = Autoencoder(sample_size).cuda() criterion = nn.MSELoss() per_sample_criterion = nn.MSELoss(reduction=&quot;none&quot;) # loss \u0434\u043b\u044f \u043a\u0430\u0436\u0434\u043e\u0433\u043e \u043f\u0440\u0438\u043c\u0435\u0440\u0430, \u043f\u0440\u043e\u043f\u0443\u0449\u0435\u043d\u043d\u043e\u0433\u043e \u0447\u0435\u0440\u0435\u0437 \u043c\u043e\u0434\u0435\u043b\u044c # \u0431\u0435\u0437 \u043f\u0430\u0440\u0430\u043c\u0435\u0442\u0440\u0430 reduction=&quot;none&quot; pytorch \u0443\u0441\u0440\u0435\u0434\u043d\u044f\u0435\u0442 loss'\u044b \u043f\u043e \u0432\u0441\u0435\u043c \u043e\u0431\u044a\u0435\u043a\u0442\u0430\u043c optimizer = Adam(model.parameters(), lr=lr, weight_decay=1e-5)<\/code><\/pre>\n<p>  <\/p>\n<p>\u0427\u0442\u043e\u0431\u044b \u043a\u0430\u043a-\u0442\u043e \u0441\u0440\u0430\u0432\u043d\u0438\u0432\u0430\u0442\u044c loss&#8217;\u044b \u043d\u0430 \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u0438 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u043f\u0440\u0438\u043c\u0435\u0440\u0430\u0445, \u0437\u0430\u0434\u0430\u0434\u0438\u043c \u0444\u0443\u043d\u043a\u0446\u0438\u044e, \u0441\u0442\u0440\u043e\u044f\u0449\u0443\u044e boxplot&#8217;\u044b \u043e\u0448\u0438\u0431\u043e\u043a:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">def save_score_distribution(model, data_loader, criterion, save_to, figsize=(8, 6)):   losses = [] # \u0437\u0434\u0435\u0441\u044c \u0431\u0443\u0434\u0435\u043c \u0445\u0440\u0430\u043d\u0438\u0442\u044c loss \u0434\u043b\u044f \u043a\u0430\u0436\u0434\u043e\u0433\u043e \u043e\u0431\u044a\u0435\u043a\u0442\u0430 \u0432\u044b\u0431\u043e\u0440\u043a\u0438   labels = [] # \u0437\u0434\u0435\u0441\u044c -- \u043c\u0435\u0442\u043a\u0438 \u043a\u043b\u0430\u0441\u0441\u0430   for (x_batch, y_batch) in data_loader:     x_batch = x_batch.cuda()      output = model(x_batch)     loss = criterion(output, x_batch)      loss = torch.mean(loss, dim=1) # \u0443\u0441\u0440\u0435\u0434\u043d\u044f\u0435\u043c loss \u043f\u043e \u043f\u0430\u0440\u0430\u043c\u0435\u0442\u0440\u0430\u043c \u043c\u043e\u0434\u0435\u043b\u0438 (\u0442\u043e \u0435\u0441\u0442\u044c \u043e\u0441\u0442\u0430\u0432\u043b\u044f\u0435\u043c \u0443\u0441\u0440\u0435\u0434\u043d\u0435\u043d\u043d\u0443\u044e \u043e\u0448\u0438\u0431\u043a\u0443 \u0434\u043b\u044f \u043a\u0430\u0436\u0434\u043e\u0433\u043e \u043e\u0431\u044a\u0435\u043a\u0442\u0430 \u0432\u044b\u0431\u043e\u0440\u043a\u0438)     loss = loss.detach().cpu().numpy().flatten()     losses.append(loss)      labels.append(y_batch.detach().cpu().numpy().flatten())    losses = np.concatenate(losses)   labels = np.concatenate(labels)    losses_0 = losses[labels == 0] # \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u044b\u0439 \u043a\u043b\u0430\u0441\u0441   losses_1 = losses[labels == 1] # \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0439 \u043a\u043b\u0430\u0441\u0441    fig, ax = plt.subplots(1, figsize=figsize)    ax.boxplot([losses_0, losses_1])   ax.set_xticklabels(['normal', 'anomaly'])    plt.savefig(save_to)   plt.close(fig)<\/code><\/pre>\n<p>  <\/p>\n<p>\u041d\u0430 \u043d\u0435\u043e\u0431\u0443\u0447\u0435\u043d\u043d\u043e\u0439 \u043c\u043e\u0434\u0435\u043b\u0438 \u0444\u0443\u043d\u043a\u0446\u0438\u044f \u0432\u044b\u0434\u0430\u0435\u0442 \u0441\u043b\u0435\u0434\u0443\u044e\u0449\u0438\u0439 \u0440\u0435\u0437\u0443\u043b\u044c\u0442\u0430\u0442:<\/p>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/zd\/bk\/vd\/zdbkvdiaq0ogs63d8w2wvrzkgey.jpeg\"><\/p>\n<p>  <\/p>\n<p>\u041e\u0431\u0443\u0447\u0435\u043d\u0438\u0435:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">experiment_path = &quot;ood_detection&quot; # \u043f\u0430\u043f\u043a\u0430 \u0434\u043b\u044f \u0441\u043e\u0445\u0440\u0430\u043d\u0435\u043d\u0438\u044f \u043a\u0430\u0440\u0442\u0438\u043d\u043e\u043a !rm -rf $experiment_path os.makedirs(experiment_path, exist_ok=True)<\/code><\/pre>\n<p>  <\/p>\n<pre><code class=\"python\">epochs = 100  for epoch in range(epochs):   running_loss = 0   for (x_batch, ) in train_in_loader:     x_batch = x_batch.cuda()      output = model(x_batch)     loss = criterion(output, x_batch)      optimizer.zero_grad()     loss.backward()     optimizer.step()      running_loss += loss.item()    print(&quot;epoch [{}\/{}], train loss:{:.4f}&quot;.format(epoch+1, epochs, running_loss))    # \u0441\u043e\u0445\u0440\u0430\u043d\u0435\u043d\u0438\u0435 \u0440\u0430\u0441\u043f\u0440\u0435\u0434\u0435\u043b\u0435\u043d\u0438\u0439 \u043e\u0448\u0438\u0431\u043e\u043a   plot_path = os.path.join(experiment_path, &quot;{}.jpg&quot;.format(epoch+1))   save_score_distribution(model, test_loader, per_sample_criterion, plot_path)  &gt;&gt;&gt;  epoch [1\/100], train loss:8.5728 epoch [2\/100], train loss:4.2405 epoch [3\/100], train loss:4.0852 epoch [4\/100], train loss:1.7578 epoch [5\/100], train loss:0.8543 ... epoch [96\/100], train loss:0.0147 epoch [97\/100], train loss:0.0154 epoch [98\/100], train loss:0.0117 epoch [99\/100], train loss:0.0105 epoch [100\/100], train loss:0.0097<\/code><\/pre>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/b2\/j_\/p3\/b2j_p34rxntvfl1zjcuq4kpomkw.jpeg\"><br \/>  <em>\u042d\u043f\u043e\u0445\u0430 50<\/em><\/p>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/do\/za\/bo\/dozabo7dq8ffd-5ahaoe4h29tka.jpeg\"><br \/>  <em>\u042d\u043f\u043e\u0445\u0430 100<\/em><\/p>\n<p>  <\/p>\n<p>\u0412\u0438\u0434\u0438\u043c, \u0447\u0442\u043e \u043e\u0448\u0438\u0431\u043a\u0438 \u043d\u0430 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u0438 \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u043a\u043b\u0430\u0441\u0441\u0430\u0445 \u043e\u0442\u043b\u0438\u0447\u0430\u044e\u0442\u0441\u044f \u0434\u043e\u0441\u0442\u0430\u0442\u043e\u0447\u043d\u043e \u0441\u0438\u043b\u044c\u043d\u043e. \u0421\u0440\u0430\u0432\u043d\u0438\u043c, \u043a\u0430\u043a \u0432\u044b\u0433\u043b\u044f\u0434\u044f\u0442 \u0440\u0435\u0430\u043b\u044c\u043d\u044b\u0435 \u0438 \u0432\u043e\u0441\u0441\u0442\u0430\u043d\u043e\u0432\u043b\u0435\u043d\u043d\u044b\u0435 \u0434\u0430\u043d\u043d\u044b\u0435:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\"># \u0444\u0443\u043d\u043a\u0446\u0438\u044f, \u043a\u043e\u0442\u043e\u0440\u0430\u044f \u0432\u043e\u0437\u0432\u0440\u0430\u0449\u0430\u0435\u0442 \u0432\u043e\u0441\u0441\u0442\u0430\u043d\u043e\u0432\u043b\u0435\u043d\u043d\u044b\u0435 \u043e\u0431\u044a\u0435\u043a\u0442\u044b def get_prediction(model, x):   global batch_size    dataset = TensorDataset(torch.tensor(x.astype(np.float32)))   data_loader = DataLoader(dataset, batch_size=batch_size, shuffle=False)    predictions = []   for batch in data_loader:     x_batch = batch[0].cuda()     pred = model(x_batch) # x -&gt; encoder -&gt; decoder -&gt; x_pred     predictions.append(pred.detach().cpu().numpy())    predictions = np.concatenate(predictions)   return predictions  # \u0432\u0438\u0437\u0443\u0430\u043b\u0438\u0437\u0430\u0446\u0438\u044f \u043e\u0431\u044a\u0435\u043a\u0442\u043e\u0432 \u0438\u0437 \u043d\u0435\u0441\u043a\u043e\u043b\u044c\u043a\u0438\u0445 \u0432\u044b\u0431\u043e\u0440\u043e\u043a (\u0440\u0435\u0430\u043b\u044c\u043d\u043e\u0439 \u0438 \u0432\u043e\u0441\u0441\u0442\u0430\u043d\u043e\u0432\u043b\u0435\u043d\u043d\u043e\u0439) def compare_data(xs, sample_num, data_range=(0, 1), labels=None):   fig, axes = plt.subplots(len(xs))   sample_size = len(xs[0][sample_num])    for i in range(len(xs)):     axes[i].plot(np.linspace(*data_range, sample_size), xs[i][sample_num])    if labels:     for i, label in enumerate(labels):       axes[i].set_ylabel(label)<\/code><\/pre>\n<p>  <\/p>\n<p>\u041a\u0430\u043a \u043c\u043e\u0434\u0435\u043b\u044c \u043e\u0442\u0440\u0430\u0431\u043e\u0442\u0430\u0435\u0442 \u043d\u0430 \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u043e\u043c \u043f\u0440\u0438\u043c\u0435\u0440\u0435:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">x_test_pred = get_prediction(model, x_test) compare_data([x_test[y_test == 0], x_test_pred[y_test == 0]], 10, labels=[&quot;real&quot;, &quot;encoded&quot;])<\/code><\/pre>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/__\/pk\/l2\/__pkl2tmqin1hyvfz6oafuskooa.png\"><\/p>\n<p>  <\/p>\n<p>\u0418 \u043d\u0430 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u043e\u043c:<\/p>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/mo\/jn\/js\/mojnjsp5pbw8up0gifbnerzgrbu.png\"><\/p>\n<p>  <\/p>\n<p>\u0427\u0442\u043e \u0436\u0435 \u0434\u0435\u043b\u0430\u0442\u044c \u0441 \u0432\u043e\u0441\u0441\u0442\u0430\u043d\u043e\u0432\u043b\u0435\u043d\u043d\u044b\u043c\u0438 <code>X<\/code>? \u041f\u043e\u043f\u0440\u043e\u0431\u0443\u0435\u043c \u043d\u0435\u043a\u043e\u0442\u043e\u0440\u044b\u0435 \u0438\u0434\u0435\u0438.<\/p>\n<p>  <\/p>\n<h4 id=\"difference-score\">Difference score<\/h4>\n<p>  <\/p>\n<p>\u0421\u0430\u043c\u043e\u0435 \u043e\u0447\u0435\u0432\u0438\u0434\u043d\u043e\u0435 \u2014 \u0432\u044b\u0447\u0435\u0441\u0442\u044c \u0438\u0437 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u043e\u0433\u043e \u043f\u0440\u0438\u043c\u0435\u0440\u0430 \u0440\u0435\u0430\u043b\u044c\u043d\u044b\u0439. \u0422\u043e\u0433\u0434\u0430 \u0432 \u0442\u043e\u043c \u043c\u0435\u0441\u0442\u0435, \u0433\u0434\u0435 \u0431\u044b\u043b \u0432\u0442\u043e\u0440\u043e\u0439 \u043f\u0438\u043a, \u0431\u0443\u0434\u0435\u0442 \u0437\u043d\u0430\u0447\u0435\u043d\u0438\u0435, \u0431\u043e\u043b\u044c\u0448\u0435\u0435 \u043d\u0443\u043b\u044f. \u0418 \u043c\u043e\u0434\u0435\u043b\u044c \u0441\u043c\u043e\u0436\u0435\u0442 \u043b\u0435\u0433\u043a\u043e \u044d\u0442\u043e \u0434\u0435\u0442\u0435\u043a\u0442\u0438\u0440\u043e\u0432\u0430\u0442\u044c.<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">def get_difference_score(model, x):   global batch_size    dataset = TensorDataset(torch.tensor(x.astype(np.float32)))   data_loader = DataLoader(dataset, batch_size=batch_size, shuffle=False)    predictions = []   for (x_batch, ) in data_loader:     x_batch = x_batch.cuda()     preds = model(x_batch)     predictions.append(preds.detach().cpu().numpy())    predictions = np.concatenate(predictions)    # \u0432\u044b\u0447\u0438\u0442\u0430\u043d\u0438\u0435 \u043f\u0440\u0435\u0434\u0441\u043a\u0430\u0437\u0430\u043d\u043d\u044b\u0445 \u043f\u0440\u0438\u043c\u0435\u0440\u043e\u0432 \u0438\u0437 \u0440\u0435\u0430\u043b\u044c\u043d\u044b\u0445   return (x - predictions)<\/code><\/pre>\n<p>  <\/p>\n<pre><code class=\"python\">from sklearn.ensemble import RandomForestClassifier  test_score = get_difference_score(model, x_test)  score_forest = RandomForestClassifier(max_features=100) score_forest.fit(test_score, y_test) &gt;&gt;&gt; RandomForestClassifier(bootstrap=True, ccp_alpha=0.0, class_weight=None,                        criterion='gini', max_depth=None, max_features=100,                        max_leaf_nodes=None, max_samples=None,                        min_impurity_decrease=0.0, min_impurity_split=None,                        min_samples_leaf=1, min_samples_split=2,                        min_weight_fraction_leaf=0.0, n_estimators=100,                        n_jobs=None, oob_score=False, random_state=None,                        verbose=0, warm_start=False)<\/code><\/pre>\n<p>  <\/p>\n<p>\u041a\u0441\u0442\u0430\u0442\u0438, \u0443 \u0447\u0438\u0442\u0430\u0442\u0435\u043b\u044f \u043c\u043e\u0433 \u0432\u043e\u0437\u043d\u0438\u043a\u043d\u0443\u0442\u044c \u0432\u043e\u043f\u0440\u043e\u0441: \u043f\u043e\u0447\u0435\u043c\u0443 \u0431\u044b \u043d\u0435 \u0441\u0434\u0435\u043b\u0430\u0442\u044c \u043b\u0438\u0448\u044c 2 \u0432\u044b\u0431\u043e\u0440\u043a\u0438 \u2014 \u043e\u0431\u0443\u0447\u0430\u044e\u0449\u0443\u044e \u0438 \u0442\u0435\u0441\u0442\u043e\u0432\u0443\u044e. \u0422\u0430\u043a \u0432\u043e\u0442, \u0432 \u043d\u0430\u0448\u0435\u043c \u043f\u0430\u0439\u043f\u043b\u0430\u0439\u043d\u0435 \u0442\u0435\u043f\u0435\u0440\u044c 2 \u043c\u043e\u0434\u0435\u043b\u0438: \u0441\u043d\u0430\u0447\u0430\u043b\u0430 \u0434\u0430\u043d\u043d\u044b\u0435 \u043f\u0440\u043e\u043f\u0443\u0441\u043a\u0430\u044e\u0442\u0441\u044f \u0447\u0435\u0440\u0435\u0437 \u0430\u0432\u0442\u043e\u0435\u043d\u043a\u043e\u0434\u0435\u0440, \u0437\u0430\u0442\u0435\u043c \u2014 \u0447\u0435\u0440\u0435\u0437 \u0444\u0443\u043d\u043a\u0446\u0438\u044e <code>difference_score<\/code>, \u0438 \u043f\u043e\u0441\u043b\u0435 \u2014 \u0447\u0435\u0440\u0435\u0437 \u0441\u043b\u0443\u0447\u0430\u0439\u043d\u044b\u0439 \u043b\u0435\u0441. \u041d\u043e \u0435\u0441\u043b\u0438 \u043c\u044b \u043f\u0440\u043e\u043f\u0443\u0441\u0442\u0438\u043c \u0442\u0440\u0435\u043d\u0438\u0440\u043e\u0432\u043e\u0447\u043d\u0443\u044e \u0432\u044b\u0431\u043e\u0440\u043a\u0443 \u0447\u0435\u0440\u0435\u0437 \u0430\u0432\u0442\u043e\u0435\u043d\u043a\u043e\u0434\u0435\u0440, \u0430 \u0437\u0430\u0442\u0435\u043c \u043f\u043e\u0434\u0430\u0434\u0438\u043c \u043d\u0430 \u043e\u0431\u0443\u0447\u0435\u043d\u0438\u0435 \u0432 \u0441\u043b\u0443\u0447\u0430\u0439\u043d\u044b\u0439 \u043b\u0435\u0441, \u0442\u043e \u0434\u0430\u043d\u043d\u044b\u0435 <em>\u0437\u0430\u0433\u0440\u044f\u0437\u043d\u044f\u0442\u0441\u044f<\/em> \u0438 \u0440\u0430\u0431\u043e\u0442\u0430 \u043c\u043e\u0434\u0435\u043b\u0438 \u0431\u0443\u0434\u0435\u0442 \u043d\u0435\u043a\u043e\u0440\u0440\u0435\u043a\u0442\u043d\u043e\u0439.<\/p>\n<p>  <\/p>\n<p>\u041a\u0430\u043a \u044d\u0442\u043e \u043f\u0440\u043e\u0438\u0441\u0445\u043e\u0434\u0438\u0442? \u041f\u0440\u0438 \u043e\u0431\u0443\u0447\u0435\u043d\u0438\u0438 \u0430\u0432\u0442\u043e\u0435\u043d\u043a\u043e\u0434\u0435\u0440 \u0441\u0442\u0440\u0435\u043c\u0438\u0442\u0441\u044f \u043c\u0438\u043d\u0438\u043c\u0438\u0437\u0438\u0440\u043e\u0432\u0430\u0442\u044c \u043e\u0448\u0438\u0431\u043a\u0443 \u043d\u0430 <strong>\u043e\u0431\u0443\u0447\u0430\u044e\u0449\u0435\u0439<\/strong> \u0432\u044b\u0431\u043e\u0440\u043a\u0435. \u041f\u043e\u044d\u0442\u043e\u043c\u0443 <code>difference_score<\/code> \u043d\u0430 \u043d\u0435\u0439 \u0431\u0443\u0434\u0435\u0442 \u043e\u0447\u0435\u043d\u044c \u043d\u0438\u0437\u043e\u043a, \u0438 \u0441\u043b\u0443\u0447\u0430\u0439\u043d\u044b\u0439 \u043b\u0435\u0441 (\u0432\u0442\u043e\u0440\u0430\u044f \u043c\u043e\u0434\u0435\u043b\u044c) \u0431\u0443\u0434\u0435\u0442 \u0434\u0443\u043c\u0430\u0442\u044c, \u0447\u0442\u043e \u043d\u0443\u0436\u043d\u043e \u043f\u043e\u0441\u0442\u0430\u0432\u0438\u0442\u044c \u043d\u0438\u0437\u043a\u0438\u0439 \u043f\u043e\u0440\u043e\u0433 \u0437\u043d\u0430\u0447\u0435\u043d\u0438\u0439, \u0447\u0442\u043e\u0431\u044b \u0441\u0434\u0435\u0442\u0435\u043a\u0442\u0438\u0440\u043e\u0432\u0430\u0442\u044c \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u044e. \u041d\u043e \u043a\u043e\u0433\u0434\u0430 \u043c\u044b \u043d\u0430\u0447\u043d\u0435\u043c \u0442\u0435\u0441\u0442\u0438\u0440\u043e\u0432\u0430\u0442\u044c \u0432\u0430\u043b\u0438\u0434\u0430\u0446\u0438\u043e\u043d\u043d\u0443\u044e \u0432\u044b\u0431\u043e\u0440\u043a\u0443, \u043c\u043e\u0436\u0435\u0442 \u0432\u044b\u044f\u0441\u043d\u0438\u0442\u044c\u0441\u044f, \u0447\u0442\u043e \u0441\u0440\u0435\u0434\u043d\u0438\u0439 \u0443\u0440\u043e\u0432\u0435\u043d\u044c <code>difference_score<\/code> \u043d\u0430 \u043d\u0435\u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u043e\u0431\u044a\u0435\u043a\u0442\u0430\u0445 \u0432\u044b\u0448\u0435, \u0447\u0435\u043c \u043f\u0440\u0438 \u043e\u0431\u0443\u0447\u0435\u043d\u0438\u0438 (\u043f\u043e\u0442\u043e\u043c\u0443 \u0447\u0442\u043e \u0430\u0432\u0442\u043e\u0435\u043d\u043a\u043e\u0434\u0435\u0440 \u043d\u0435 \u0432\u0438\u0434\u0435\u043b \u0442\u0430\u043a\u0438\u0435 \u043e\u0431\u044a\u0435\u043a\u0442\u044b \u043f\u0440\u0438 \u043e\u0431\u0443\u0447\u0435\u043d\u0438\u0438). \u0418 \u0441\u043b\u0443\u0447\u0430\u0439\u043d\u044b\u0439 \u043b\u0435\u0441 \u043d\u0435 \u0441\u043c\u043e\u0436\u0435\u0442 \u0430\u0434\u0435\u043a\u0432\u0430\u0442\u043d\u043e \u043e\u0431\u043d\u0430\u0440\u0443\u0436\u0438\u0442\u044c \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0438.<\/p>\n<p>  <\/p>\n<p>\u041e\u0446\u0435\u043d\u0438\u043c \u043c\u043e\u0434\u0435\u043b\u044c \u043d\u0430 \u0432\u0430\u043b\u0438\u0434\u0430\u0446\u0438\u043e\u043d\u043d\u043e\u0439 \u0432\u044b\u0431\u043e\u0440\u043a\u0435:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">val_score = get_difference_score(model, x_val) prediction = score_forest.predict(val_score) print(classification_report(y_val, prediction)) &gt;&gt;&gt;            precision    recall  f1-score   support           0.0       0.76      1.00      0.87       100          1.0       1.00      0.69      0.82       100      accuracy                           0.84       200    macro avg       0.88      0.84      0.84       200 weighted avg       0.88      0.84      0.84       200<\/code><\/pre>\n<p>  <\/p>\n<p>\u0423\u0436\u0435 \u043b\u0443\u0447\u0448\u0435. \u041f\u043e\u0441\u043c\u043e\u0442\u0440\u0438\u043c \u043d\u0430 \u043d\u0430\u0448\u0438 \u0434\u0430\u043d\u043d\u044b\u0435:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">indices = np.arange(len(prediction)) # \u0432 \u043f\u0435\u0440\u0432\u0443\u044e \u043e\u0447\u0435\u0440\u0435\u0434\u044c \u043d\u0430\u0441 \u0438\u043d\u0442\u0435\u0440\u0435\u0441\u0443\u044e\u0442 \u0442\u0435 \u0441\u043b\u0443\u0447\u0430\u0438, \u043a\u043e\u0433\u0434\u0430 \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0438 \u043d\u0435 \u0431\u044b\u043b\u0438 \u0440\u0430\u0441\u043f\u043e\u0437\u043d\u0430\u043d\u044b wrong_indices = indices[(prediction == 0) &amp; (y_val == 1)]  x_val_pred = get_prediction(model, x_val) compare_data([x_val, x_val_pred], wrong_indices[0])<\/code><\/pre>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/ia\/23\/md\/ia23mdeirk0yph9z4daawza1axw.png\"><\/p>\n<p>  <\/p>\n<p>\u0410 \u0447\u0442\u043e \u043f\u0440\u043e\u0438\u0441\u0445\u043e\u0434\u0438\u0442 \u0441 \u0441\u0430\u043c\u0438\u043c\u0438 \u0441\u043a\u043e\u0440\u0430\u043c\u0438? \u041d\u0435\u0440\u0430\u0441\u043f\u043e\u0437\u043d\u0430\u043d\u043d\u044b\u0435 \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0438:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">plt.imshow(val_score[wrong_indices], norm=Normalize(0, 1, clip=True))<\/code><\/pre>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/cg\/oy\/eq\/cgoyeqykuf6bpows4sd8tlky-gg.png\"><\/p>\n<p>  <\/p>\n<p>\u0420\u0430\u0441\u043f\u043e\u0437\u043d\u0430\u043d\u043d\u044b\u0435 \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0438:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">plt.imshow(val_score[(prediction == 1) &amp; (y_val == 1)], norm=Normalize(0, 1, clip=True))<\/code><\/pre>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/nk\/hw\/th\/nkhwthlp5-_pvq_6xbcy2cqn20c.png\"><\/p>\n<p>  <\/p>\n<p>\u0418 \u043f\u0440\u043e\u0441\u0442\u043e \u0432\u0435\u0440\u043d\u043e \u043f\u0440\u0435\u0434\u0441\u043a\u0430\u0437\u0430\u043d\u043d\u044b\u0435 \u043d\u0435\u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0435 \u043e\u0431\u044a\u0435\u043a\u0442\u044b:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">plt.imshow(val_score[(prediction == 0) &amp; (y_val == 0)], norm=Normalize(0, 1, clip=True))<\/code><\/pre>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/cu\/xw\/od\/cuxwod1ophr8rs43agecaz_sawm.png\"><\/p>\n<p>  <\/p>\n<p>\u0425\u043c, \u044d\u0442\u043e \u043d\u0430\u0442\u0430\u043b\u043a\u0438\u0432\u0430\u0435\u0442 \u043d\u0430 \u043e\u043f\u0440\u0435\u0434\u0435\u043b\u0435\u043d\u043d\u0443\u044e \u043c\u044b\u0441\u043b\u044c: \u0443 \u0431\u043e\u043b\u044c\u0448\u0435\u0439 \u0447\u0430\u0441\u0442\u0438 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u043e\u0431\u044a\u0435\u043a\u0442\u043e\u0432 \u0432 \u0441\u043a\u043e\u0440\u0435 \u0445\u043e\u0440\u043e\u0448\u043e \u043f\u0440\u043e\u0441\u043b\u0435\u0436\u0438\u0432\u0430\u0435\u0442\u0441\u044f \u043f\u0438\u043a.<\/p>\n<p>  <\/p>\n<h4 id=\"difference-histograms\">Difference histograms<\/h4>\n<p>  <\/p>\n<p>\u0422\u043e\u0433\u0434\u0430 \u043c\u043e\u0436\u043d\u043e \u043f\u043e\u0441\u0442\u0440\u043e\u0438\u0442\u044c \u0433\u0438\u0441\u0442\u043e\u0433\u0440\u0430\u043c\u043c\u044b \u0437\u043d\u0430\u0447\u0435\u043d\u0438\u0439 \u0438 \u043d\u0430 \u043d\u0438\u0445 \u0434\u0435\u043b\u0430\u0442\u044c \u043f\u0440\u0435\u0434\u0441\u043a\u0430\u0437\u0430\u043d\u0438\u044f. \u0423 \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u043e\u0431\u044a\u0435\u043a\u0442\u043e\u0432 \u0441\u0442\u043e\u043b\u0431\u0446\u044b \u0431\u0443\u0434\u0443\u0442 \u043d\u0430\u0445\u043e\u0434\u0438\u0442\u044c\u0441\u044f \u043e\u043a\u043e\u043b\u043e \u043d\u0443\u043b\u044f \u0438 \u0432\u0441\u0442\u0440\u0435\u0447\u0430\u0442\u044c\u0441\u044f \u043e\u0442\u043d\u043e\u0441\u0438\u0442\u0435\u043b\u044c\u043d\u043e \u0447\u0430\u0441\u0442\u043e. \u0423 \u0430\u043d\u043e\u043c\u0430\u043b\u044c\u043d\u044b\u0445 \u0436\u0435 \u2014 \u043f\u043e\u043c\u0438\u043c\u043e \u043e\u0441\u043d\u043e\u0432\u043d\u044b\u0445 \u0441\u0442\u043e\u043b\u0431\u0446\u043e\u0432, \u0431\u0443\u0434\u0443\u0442 \u043d\u0435\u0431\u043e\u043b\u044c\u0448\u0438\u0435 &quot;\u043f\u043e\u0431\u043e\u0447\u043d\u044b\u0435&quot;, \u0441\u043e\u043e\u0442\u0432\u0435\u0442\u0441\u0442\u0432\u0443\u044e\u0449\u0438\u0435 \u0431\u043e\u043b\u044c\u0448\u0435\u0439 \u0440\u0430\u0437\u043d\u0438\u0446\u0435 \u043c\u0435\u0436\u0434\u0443 \u0440\u0435\u0430\u043b\u044c\u043d\u044b\u043c \u0438 \u043f\u0440\u0435\u0434\u0441\u043a\u0430\u0437\u0430\u043d\u043d\u044b\u043c \u043e\u0431\u044a\u0435\u043a\u0442\u043e\u043c.<\/p>\n<p>  <\/p>\n<p>\u041f\u043e\u0441\u043c\u043e\u0442\u0440\u0438\u043c, \u0432 \u043a\u0430\u043a\u043e\u043c \u0434\u0438\u0430\u043f\u0430\u0437\u043e\u043d\u0435 \u0438\u0437\u043c\u0435\u043d\u044f\u0435\u0442\u0441\u044f <code>difference score<\/code><\/p>\n<p>  <\/p>\n<pre><code class=\"python\">print(&quot;test score: [{}; {}]&quot;.format(test_score.min(), test_score.max())) &gt;&gt;&gt; test score: [-0.2260764424351479; 0.26339245919832344]<\/code><\/pre>\n<p>  <\/p>\n<p>\u0418\u0441\u0445\u043e\u0434\u044f \u0438\u0437 \u0437\u043d\u0430\u0447\u0435\u043d\u0438\u0439 \u0432 \u0442\u0435\u0441\u0442\u043e\u0432\u043e\u0439 \u0432\u044b\u0431\u043e\u0440\u043a\u0435, \u0431\u0443\u0434\u0435\u043c \u0441\u0442\u0440\u043e\u0438\u0442\u044c \u0433\u0438\u0441\u0442\u043e\u0433\u0440\u0430\u043c\u043c\u044b \u0442\u043e\u043b\u044c\u043a\u043e \u043d\u0430 \u0434\u0430\u043d\u043d\u043e\u043c \u043f\u0440\u043e\u043c\u0435\u0436\u0443\u0442\u043a\u0435:<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">def score_to_histograms(scores, bins=10, data_range=(-0.3, 0.3)):   result_histograms = np.zeros((len(scores), bins))    for i in range(len(scores)):     hist, bins = np.histogram(scores[i], bins=bins, range=data_range)     result_histograms[i] = hist    return result_histograms<\/code><\/pre>\n<p>  <\/p>\n<pre><code class=\"python\">test_histogram = score_to_histograms(test_score, bins=10, data_range=(-0.3, 0.3)) val_histogram = score_to_histograms(val_score, bins=10, data_range=(-0.3, 0.3))<\/code><\/pre>\n<p>  <\/p>\n<pre><code class=\"python\">plt.title(&quot;normal histogram&quot;) plt.bar(np.linspace(-0.3, 0.3, 10), test_histogram[y_test == 0][0])<\/code><\/pre>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/3v\/vg\/y0\/3vvgy00sui4mr7xamlvdnfmxzos.png\"><\/p>\n<p>  <\/p>\n<pre><code class=\"python\">plt.title(&quot;anomaly histogram&quot;) plt.bar(np.linspace(-0.3, 0.3, 10), test_histogram[y_test == 1][0])<\/code><\/pre>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/b4\/lh\/r-\/b4lhr-eiytahuwak954gzmstkbq.png\"><\/p>\n<p>  <\/p>\n<p>\u0414\u0435\u0439\u0441\u0442\u0432\u0438\u0442\u0435\u043b\u044c\u043d\u043e, \u043f\u043e\u044f\u0432\u043b\u044f\u0435\u0442\u0441\u044f \u0434\u043e\u043f\u043e\u043b\u043d\u0438\u0442\u0435\u043b\u044c\u043d\u043e\u0435 &quot;\u043a\u0440\u044b\u043b\u043e&quot;, \u043a\u043e\u0442\u043e\u0440\u043e\u0435 \u0431\u0443\u0434\u0435\u0442 \u043f\u0440\u043e\u0441\u0442\u043e \u043e\u0431\u043d\u0430\u0440\u0443\u0436\u0438\u0442\u044c.<\/p>\n<p>  <\/p>\n<pre><code class=\"python\">histogram_forest = RandomForestClassifier(n_estimators=10) histogram_forest.fit(test_histogram, y_test) &gt;&gt;&gt; RandomForestClassifier(bootstrap=True, ccp_alpha=0.0, class_weight=None,                        criterion='gini', max_depth=None, max_features='auto',                        max_leaf_nodes=None, max_samples=None,                        min_impurity_decrease=0.0, min_impurity_split=None,                        min_samples_leaf=1, min_samples_split=2,                        min_weight_fraction_leaf=0.0, n_estimators=10,                        n_jobs=None, oob_score=False, random_state=None,                        verbose=0, warm_start=False)<\/code><\/pre>\n<p>  <\/p>\n<pre><code class=\"python\">val_prediction = histogram_forest.predict(val_histogram) print(classification_report(y_val, val_prediction)) &gt;&gt;&gt;            precision    recall  f1-score   support           0.0       0.83      0.99      0.90       100          1.0       0.99      0.80      0.88       100      accuracy                           0.90       200    macro avg       0.91      0.90      0.89       200 weighted avg       0.91      0.90      0.89       200<\/code><\/pre>\n<p>  <\/p>\n<h2 id=\"vyvody\">\u0412\u044b\u0432\u043e\u0434\u044b<\/h2>\n<p>  <\/p>\n<p>\u0414\u0435\u0442\u0435\u043a\u0442\u0438\u0440\u043e\u0432\u0430\u043d\u0438\u0435 \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0439 \u2014 \u0441\u043b\u043e\u0436\u043d\u0430\u044f \u0438 \u0438\u043d\u0442\u0435\u0440\u0435\u0441\u043d\u0430\u044f \u0437\u0430\u0434\u0430\u0447\u0430. \u041f\u043e\u0440\u043e\u0439 \u0441\u043f\u043e\u0441\u043e\u0431\u044b, \u043a\u043e\u0442\u043e\u0440\u044b\u0435 \u0440\u0430\u0431\u043e\u0442\u0430\u044e\u0442 \u043d\u0430 \u043e\u0434\u043d\u043e\u043c \u043d\u0430\u0431\u043e\u0440\u0435 \u0434\u0430\u043d\u043d\u044b\u043c, \u043d\u0435 \u0431\u0443\u0434\u0443\u0442 \u0440\u0430\u0431\u043e\u0442\u0430\u0442\u044c \u043d\u0430 \u0434\u0440\u0443\u0433\u043e\u043c, \u043f\u0443\u0441\u0442\u044c \u0438 \u043e\u0447\u0435\u043d\u044c \u043f\u043e\u0445\u043e\u0436\u0435\u043c, \u0434\u0430\u0442\u0430\u0441\u0435\u0442\u0435. \u041f\u043e\u044d\u0442\u043e\u043c\u0443 \u043d\u0443\u0436\u043d\u043e \u0441\u043c\u043e\u0442\u0440\u0435\u0442\u044c \u043d\u0430 \u0446\u0435\u043b\u044b\u0435 \u0433\u0440\u0443\u043f\u043f\u044b \u0441\u043f\u043e\u0441\u043e\u0431\u043e\u0432 (\u043f\u043e\u0434\u0445\u043e\u0434\u044b) \u0438 \u0442\u0435\u0441\u0442\u0438\u0440\u043e\u0432\u0430\u0442\u044c \u043c\u043d\u043e\u0433\u043e \u0433\u0438\u043f\u043e\u0442\u0435\u0437.<\/p>\n<p>  <\/p>\n<p>\u0410\u0432\u0442\u043e\u043a\u043e\u0434\u0438\u0440\u043e\u0432\u0449\u0438\u043a\u0438 \u2014 \u043e\u0447\u0435\u043d\u044c \u043c\u043e\u0449\u043d\u044b\u0439 \u0438\u043d\u0441\u0442\u0440\u0443\u043c\u0435\u043d\u0442 \u0434\u043b\u044f \u0434\u0435\u0442\u0435\u043a\u0442\u0438\u0440\u043e\u0432\u0430\u043d\u0438\u044f \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0439. \u0427\u0442\u043e \u0435\u0449\u0435 \u043c\u043e\u0436\u043d\u043e \u043f\u043e\u043f\u0440\u043e\u0431\u043e\u0432\u0430\u0442\u044c \u0441 \u043d\u0438\u043c\u0438? \u0415\u0441\u0442\u044c \u0432\u0430\u0440\u0438\u0430\u043d\u0442 \u0438\u0441\u043f\u043e\u043b\u044c\u0437\u043e\u0432\u0430\u0442\u044c VAE, \u0443 \u043a\u043e\u0442\u043e\u0440\u043e\u0433\u043e \u043b\u0430\u0442\u0435\u043d\u0442\u043d\u043e\u0435 \u043f\u0440\u043e\u0441\u0442\u0440\u0430\u043d\u0441\u0442\u0432\u043e (\u0442\u0435 \u0441\u0430\u043c\u044b\u0435 4 \u043d\u0435\u0439\u0440\u043e\u043d\u0430) \u044f\u0432\u043b\u044f\u0435\u0442\u0441\u044f \u043d\u043e\u0440\u043c\u0430\u043b\u044c\u043d\u044b\u043c \u0440\u0430\u0441\u043f\u0440\u0435\u0434\u0435\u043b\u0435\u043d\u0438\u0435\u043c. \u0422\u043e\u0433\u0434\u0430 \u043c\u043e\u0436\u043d\u043e \u043a\u043b\u0430\u0441\u0441\u0438\u0444\u0438\u0446\u0438\u0440\u043e\u0432\u0430\u0442\u044c \u043f\u0440\u0438\u043c\u0435\u0440\u044b \u043f\u043e \u044d\u0442\u043e\u043c\u0443 \u043f\u0440\u043e\u0441\u0442\u0440\u0430\u043d\u0441\u0442\u0432\u0443. \u0422\u0430\u043a\u0436\u0435 \u0441\u0443\u0449\u0435\u0441\u0442\u0432\u0443\u044e\u0442 CVAE, \u043a\u043e\u0442\u043e\u0440\u044b\u0435 \u0438\u0441\u043f\u043e\u043b\u044c\u0437\u0443\u044e\u0442 \u043a\u0430\u043a\u043e\u0439-\u0442\u043e \u0434\u043e\u043f\u043e\u043b\u043d\u0438\u0442\u0435\u043b\u044c\u043d\u044b\u0439 \u043f\u0440\u0438\u0437\u043d\u0430\u043a \u0434\u043b\u044f \u0433\u0435\u043d\u0435\u0440\u0430\u0446\u0438\u0438 \u0438\u0437 \u0441\u043a\u0440\u044b\u0442\u043e\u0433\u043e \u043f\u0440\u0438\u0437\u043d\u0430\u043a\u043e\u0432\u043e\u0433\u043e \u043f\u0440\u043e\u0441\u0442\u0440\u0430\u043d\u0441\u0442\u0432\u0430. \u041d\u0430\u043f\u0440\u0438\u043c\u0435\u0440, \u044d\u0442\u043e \u043c\u043e\u0436\u0435\u0442 \u0431\u044b\u0442\u044c \u043c\u0435\u0442\u043a\u0430 \u043a\u043b\u0430\u0441\u0441\u0430, \u0440\u0430\u0437\u043c\u0435\u0440 \u0446\u0438\u0444\u0440\u044b, \u043c\u0430\u043a\u0441\u0438\u043c\u0430\u043b\u044c\u043d\u043e\u0435 \u0437\u043d\u0430\u0447\u0435\u043d\u0438\u0435 \u0432 \u0440\u0430\u0441\u043f\u0440\u0435\u0434\u0435\u043b\u0435\u043d\u0438\u0438 \u0438 \u0442.\u0434.<\/p>\n<p>  <\/p>\n<p>\u041e\u0447\u0435\u043d\u044c \u0431\u043b\u0438\u0437\u043a\u043e \u043a \u0430\u0432\u0442\u043e\u0435\u043d\u043a\u043e\u0434\u0435\u0440\u0430\u043c \u043d\u0430\u0445\u043e\u0434\u044f\u0442\u0441\u044f GAN&#8217;\u044b, \u0433\u0434\u0435 \u0435\u0441\u0442\u044c \u0446\u0435\u043b\u0430\u044f \u043c\u043e\u0434\u0435\u043b\u044c (\u0434\u0438\u0441\u043a\u0440\u0438\u043c\u0438\u043d\u0430\u0442\u043e\u0440), \u043f\u043e\u0437\u0432\u043e\u043b\u044f\u044e\u0449\u0430\u044f \u043e\u0431\u043d\u0430\u0440\u0443\u0436\u0438\u0442\u044c \u0441\u0435\u043c\u043f\u043b\u044b \u043d\u0435 \u0438\u0437 \u043e\u0431\u0443\u0447\u0430\u044e\u0449\u0435\u0439 \u0432\u044b\u0431\u043e\u0440\u043a\u0438. \u0422\u043e\u0433\u0434\u0430 \u0434\u043e\u0441\u0442\u0430\u0442\u043e\u0447\u043d\u043e \u0445\u043e\u0440\u043e\u0448\u043e \u043d\u0430\u0442\u0440\u0435\u043d\u0438\u0440\u043e\u0432\u0430\u0442\u044c \u043f\u0430\u0440\u0443 \u0433\u0435\u043d\u0435\u0440\u0430\u0442\u043e\u0440\u0430 \u0438 \u0434\u0438\u0441\u043a\u0440\u0438\u043c\u0438\u043d\u0430\u0442\u043e\u0440\u0430 (\u0447\u0442\u043e \u0432 \u0434\u0435\u0439\u0441\u0442\u0432\u0438\u0442\u0435\u043b\u044c\u043d\u043e\u0441\u0442\u0438 \u043d\u0435\u043f\u0440\u043e\u0441\u0442\u043e).<\/p>\n<p>  <\/p>\n<p>\u041a\u043e\u0433\u0434\u0430 \u0443 \u043c\u0435\u043d\u044f \u0432\u043f\u0435\u0440\u0432\u044b\u0435 \u043f\u043e\u044f\u0432\u0438\u043b\u0430\u0441\u044c \u0437\u0430\u0434\u0430\u0447\u0430 \u0434\u0435\u0442\u0435\u043a\u0442\u0438\u0440\u043e\u0432\u0430\u0442\u044c \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0438, \u044f \u0441\u043e\u0441\u0442\u0430\u0432\u0438\u043b \u043d\u0435\u0431\u043e\u043b\u044c\u0448\u043e\u0439 \u0441\u043f\u0438\u0441\u043e\u043a \u0442\u043e\u0433\u043e, \u0447\u0442\u043e \u043c\u043e\u0436\u043d\u043e \u043f\u043e\u043f\u0440\u043e\u0431\u043e\u0432\u0430\u0442\u044c.<\/p>\n<p>  <\/p>\n<div class=\"spoiler\"><b class=\"spoiler_title\">\u041f\u043e\u0434\u0445\u043e\u0434\u044b \u0434\u043b\u044f \u0440\u0430\u0431\u043e\u0442\u044b \u0441 \u043a\u0430\u0440\u0442\u0438\u043d\u043a\u0430\u043c\u0438<\/b><\/p>\n<div class=\"spoiler_text\">\n<h4 id=\"statistical-parametric\">Statistical parametric<\/h4>\n<p>  <\/p>\n<ul>\n<li>GMM \u2014 Gaussian mixture modelling + Akaike or Bayesian Information Criterion<\/li>\n<li>HMM \u2014 Hidden Markov models<\/li>\n<li>MRF \u2014 Markov random fields<\/li>\n<li>CRF \u2014 conditional random fields<\/li>\n<\/ul>\n<p>  <\/p>\n<h4 id=\"robust-statistic\">Robust statistic<\/h4>\n<p>  <\/p>\n<ul>\n<li>minimum volume estimation<\/li>\n<li>PCA<\/li>\n<li>estimation maximisation (EM) + deterministic annealing<\/li>\n<li>K-means<\/li>\n<\/ul>\n<p>  <\/p>\n<h4 id=\"non-parametric-statistics\">Non-parametric statistics<\/h4>\n<p>  <\/p>\n<ul>\n<li>histogram analysis with density estimation on KNN<\/li>\n<li>local kernel models (Parzen windowing)<\/li>\n<li>vector of feature matching with similarity distance (between train and test)<\/li>\n<li>wavelets + MMRF<\/li>\n<li>histogram-based measures features<\/li>\n<li>texture features<\/li>\n<li>shape features<\/li>\n<li>features from VGG-16<\/li>\n<li>HOG<\/li>\n<\/ul>\n<p>  <\/p>\n<h4 id=\"neural-networks\">Neural networks<\/h4>\n<p>  <\/p>\n<ul>\n<li>self organisation maps (SOM) or Kohonen&#8217;s<\/li>\n<li>Radial Basis Functions (RBF) (Minhas, 2005)<\/li>\n<li>LearningVector Quantisation (LVQ)<\/li>\n<li>ProbabilisticNeural Networks (PNN)<\/li>\n<li>Hopfieldnetworks<\/li>\n<li>SupportVector Machines (SVM)<\/li>\n<li>AdaptiveResonance Theory (ART)<\/li>\n<li>Relevance vector machine (RVM)<\/li>\n<\/ul>\n<\/div>\n<\/div>\n<p>  <\/p>\n<h4 id=\"nemnogo-obo-mne\">\u041d\u0435\u043c\u043d\u043e\u0433\u043e \u043e\u0431\u043e \u043c\u043d\u0435<\/h4>\n<p>  <\/p>\n<p>\u041c\u0435\u043d\u044f \u0437\u043e\u0432\u0443\u0442 \u0415\u0432\u0433\u0435\u043d\u0438\u0439, Data science&#8217;\u043e\u043c \u044f \u0437\u0430\u043d\u0438\u043c\u0430\u044e\u0441\u044c \u0443\u0436\u0435 \u043f\u043e\u043b\u0442\u043e\u0440\u0430 \u0433\u043e\u0434\u0430. \u0421\u0435\u0439\u0447\u0430\u0441 \u0432 \u0431\u043e\u043b\u044c\u0448\u0435\u0439 \u0441\u0442\u0435\u043f\u0435\u043d\u0438 \u043f\u043e\u0433\u0440\u0443\u0436\u0430\u044e\u0441\u044c \u0432 Computer vision, \u043d\u043e \u0442\u0430\u043a\u0436\u0435 \u0438\u043d\u0442\u0435\u0440\u0435\u0441\u0443\u044e\u0441\u044c \u043d\u0435\u0439\u0440\u043e\u0442\u0435\u0445\u043d\u043e\u043b\u043e\u0433\u0438\u044f\u043c\u0438. (\u0421\u043e\u0435\u0434\u0438\u043d\u0438\u0442\u044c \u0438\u0441\u043a\u0443\u0441\u0441\u0442\u0432\u0435\u043d\u043d\u044b\u0435 \u0438 \u043d\u0430\u0442\u0443\u0440\u0430\u043b\u044c\u043d\u044b\u0435 \u043d\u0435\u0439\u0440\u043e\u043d\u043d\u044b\u0435 \u0441\u0435\u0442\u0438 \u2014 \u043c\u043e\u044f \u043c\u0435\u0447\u0442\u0430!) \u042d\u0442\u043e\u0442 \u043f\u043e\u0441\u0442 \u0441\u043e\u0437\u0434\u0430\u043d \u0431\u043b\u0430\u0433\u043e\u0434\u0430\u0440\u044f \u043d\u0430\u0448\u0435\u0439 \u043a\u043e\u043c\u0430\u043d\u0434\u0435 \u2014 FARADAY Lab. \u041c\u044b \u2014 \u043d\u0430\u0447\u0438\u043d\u0430\u044e\u0449\u0438\u0435 \u0440\u043e\u0441\u0441\u0438\u0439\u0441\u043a\u0438\u0435 \u0441\u0442\u0430\u0440\u0442\u0430\u043f\u0435\u0440\u044b \u0438 \u0433\u043e\u0442\u043e\u0432\u044b \u0434\u0435\u043b\u0438\u0442\u044c\u0441\u044f \u0441 \u0412\u0430\u043c\u0438 \u0442\u0435\u043c, \u0447\u0442\u043e \u0443\u0437\u043d\u0430\u0435\u043c \u0441\u0430\u043c\u0438.<\/p>\n<p>  <\/p>\n<p>\u0423\u0434\u0430\u0447\u0438 c:<\/p>\n<p>  <\/p>\n<p><img decoding=\"async\" src=\"https:\/\/habrastorage.org\/webt\/ms\/1u\/a8\/ms1ua8wsr4u5h1opv3ylaonfq2k.png\"><\/p>\n<p>  <\/p>\n<h3 id=\"poleznye-ssylki\">\u041f\u043e\u043b\u0435\u0437\u043d\u044b\u0435 \u0441\u0441\u044b\u043b\u043a\u0438<\/h3>\n<p>  <\/p>\n<ul>\n<li><a href=\"https:\/\/github.com\/evjeny\/ood_detection_autoencoders\" rel=\"nofollow\">\u0420\u0435\u043f\u043e\u0437\u0438\u0442\u043e\u0440\u0438\u0439 \u0441 \u043a\u043e\u0434\u043e\u043c<\/a><\/li>\n<li><a href=\"https:\/\/www.researchgate.net\/publication\/202974904_Anomaly_Detection_in_Medical_Image_Analysis\" rel=\"nofollow\">Anomaly Detection in Medical Image Analysis<\/a><\/li>\n<li><a href=\"https:\/\/towardsdatascience.com\/anomaly-detection-for-dummies-15f148e559c1\" rel=\"nofollow\">Anomaly detection for dummies<\/a><\/li>\n<li><a href=\"http:\/\/www.machinelearning.ru\/wiki\/images\/5\/54\/AnomalyDetectionMethods.pdf\" rel=\"nofollow\">\u041c\u0435\u0442\u043e\u0434\u044b \u0434\u0435\u0442\u0435\u043a\u0442\u0438\u0440\u043e\u0432\u0430\u043d\u0438\u044f \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0439<\/a><\/li>\n<li><a href=\"https:\/\/ai.googleblog.com\/2019\/12\/improving-out-of-distribution-detection.html?m=1\" rel=\"nofollow\">Improving OOD detection<\/a><\/li>\n<\/ul>\n<\/div>\n<p> \u0441\u0441\u044b\u043b\u043a\u0430 \u043d\u0430 \u043e\u0440\u0438\u0433\u0438\u043d\u0430\u043b \u0441\u0442\u0430\u0442\u044c\u0438 <a href=\"https:\/\/habr.com\/ru\/post\/491552\/\"> https:\/\/habr.com\/ru\/post\/491552\/<\/a><\/p>\n","protected":false},"excerpt":{"rendered":"\n<div class=\"post__text post__text-html post__text_v1\" id=\"post-content-body\" data-io-article-url=\"https:\/\/habr.com\/ru\/post\/491552\/\">\n<p>\u0414\u0435\u0442\u0435\u043a\u0442\u0438\u0440\u043e\u0432\u0430\u043d\u0438\u0435 \u0430\u043d\u043e\u043c\u0430\u043b\u0438\u0439 \u2014 \u0438\u043d\u0442\u0435\u0440\u0435\u0441\u043d\u0430\u044f \u0437\u0430\u0434\u0430\u0447\u0430 \u043c\u0430\u0448\u0438\u043d\u043d\u043e\u0433\u043e \u043e\u0431\u0443\u0447\u0435\u043d\u0438\u044f. \u041d\u0435 \u0441\u0443\u0449\u0435\u0441\u0442\u0432\u0443\u0435\u0442 \u043a\u0430\u043a\u043e\u0433\u043e-\u0442\u043e \u043e\u043f\u0440\u0435\u0434\u0435\u043b\u0435\u043d\u043d\u043e\u0433\u043e \u0441\u043f\u043e\u0441\u043e\u0431\u0430 \u0435\u0435 \u0440\u0435\u0448\u0435\u043d\u0438\u044f, \u0442\u0430\u043a \u043a\u0430\u043a \u043a\u0430\u0436\u0434\u044b\u0439 \u043d\u0430\u0431\u043e\u0440 \u0434\u0430\u043d\u043d\u044b\u0445 \u0438\u043c\u0435\u0435\u0442 \u0441\u0432\u043e\u0438 \u043e\u0441\u043e\u0431\u0435\u043d\u043d\u043e\u0441\u0442\u0438. \u041d\u043e \u0432 \u0442\u043e \u0436\u0435 \u0432\u0440\u0435\u043c\u044f \u0435\u0441\u0442\u044c \u043d\u0435\u0441\u043a\u043e\u043b\u044c\u043a\u043e \u043f\u043e\u0434\u0445\u043e\u0434\u043e\u0432, \u043a\u043e\u0442\u043e\u0440\u044b\u0435 \u043f\u043e\u043c\u043e\u0433\u0430\u044e\u0442 \u0434\u043e\u0431\u0438\u0442\u044c\u0441\u044f \u0443\u0441\u043f\u0435\u0445\u0430. \u042f \u0445\u043e\u0447\u0443 \u0440\u0430\u0441\u0441\u043a\u0430\u0437\u0430\u0442\u044c \u043f\u0440\u043e \u043e\u0434\u0438\u043d \u0438\u0437 \u0442\u0430\u043a\u0438\u0445 \u043f\u043e\u0434\u0445\u043e\u0434\u043e\u0432 \u2014 \u0430\u0432\u0442\u043e\u0435\u043d\u043a\u043e\u0434\u0435\u0440\u044b.<\/p>\n","protected":false},"author":1,"featured_media":0,"comment_status":"open","ping_status":"open","sticky":false,"template":"","format":"standard","meta":{"footnotes":""},"categories":[],"tags":[],"class_list":["post-299869","post","type-post","status-publish","format-standard","hentry"],"_links":{"self":[{"href":"https:\/\/savepearlharbor.com\/index.php?rest_route=\/wp\/v2\/posts\/299869","targetHints":{"allow":["GET"]}}],"collection":[{"href":"https:\/\/savepearlharbor.com\/index.php?rest_route=\/wp\/v2\/posts"}],"about":[{"href":"https:\/\/savepearlharbor.com\/index.php?rest_route=\/wp\/v2\/types\/post"}],"author":[{"embeddable":true,"href":"https:\/\/savepearlharbor.com\/index.php?rest_route=\/wp\/v2\/users\/1"}],"replies":[{"embeddable":true,"href":"https:\/\/savepearlharbor.com\/index.php?rest_route=%2Fwp%2Fv2%2Fcomments&post=299869"}],"version-history":[{"count":0,"href":"https:\/\/savepearlharbor.com\/index.php?rest_route=\/wp\/v2\/posts\/299869\/revisions"}],"wp:attachment":[{"href":"https:\/\/savepearlharbor.com\/index.php?rest_route=%2Fwp%2Fv2%2Fmedia&parent=299869"}],"wp:term":[{"taxonomy":"category","embeddable":true,"href":"https:\/\/savepearlharbor.com\/index.php?rest_route=%2Fwp%2Fv2%2Fcategories&post=299869"},{"taxonomy":"post_tag","embeddable":true,"href":"https:\/\/savepearlharbor.com\/index.php?rest_route=%2Fwp%2Fv2%2Ftags&post=299869"}],"curies":[{"name":"wp","href":"https:\/\/api.w.org\/{rel}","templated":true}]}}