%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
%%                                                                 %%
%% Please do not use \input{...} to include other tex files.       %%
%% Submit your LaTeX manuscript as one .tex document.              %%
%%                                                                 %%
%% All additional figures and files should be attached             %%
%% separately and not embedded in the \TeX\ document itself.       %%
%%                                                                 %%
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%

%%\documentclass[referee,sn-basic]{sn-jnl}% referee option is meant for double line spacing

%%=======================================================%%
%% to print line numbers in the margin use lineno option %%
%%=======================================================%%

%%\documentclass[lineno,sn-basic]{sn-jnl}% Basic Springer Nature Reference Style/Chemistry Reference Style

%%======================================================%%
%% to compile with pdflatex/xelatex use pdflatex option %%
%%======================================================%%

%%\documentclass[pdflatex,sn-basic]{sn-jnl}% Basic Springer Nature Reference Style/Chemistry Reference Style

%%\documentclass[sn-basic]{sn-jnl}% Basic Springer Nature Reference Style/Chemistry Reference Style
\documentclass[sn-mathphys]{sn-jnl}% Math and Physical Sciences Reference Style
%%\documentclass[sn-aps]{sn-jnl}% American Physical Society (APS) Reference Style
%%\documentclass[sn-vancouver]{sn-jnl}% Vancouver Reference Style
%%\documentclass[sn-apa]{sn-jnl}% APA Reference Style
%%\documentclass[sn-chicago]{sn-jnl}% Chicago-based Humanities Reference Style
%%\documentclass[sn-standardnature]{sn-jnl}% Standard Nature Portfolio Reference Style
%%\documentclass[default]{sn-jnl}% Default
%%\documentclass[default,iicol]{sn-jnl}% Default with double column layout

%%%% Standard Packages
%%<additional latex packages if required can be included here>
%%%%

%%%%%=============================================================================%%%%
%%%%  Remarks: This template is provided to aid authors with the preparation
%%%%  of original research articles intended for submission to journals published 
%%%%  by Springer Nature. The guidance has been prepared in partnership with 
%%%%  production teams to conform to Springer Nature technical requirements. 
%%%%  Editorial and presentation requirements differ among journal portfolios and 
%%%%  research disciplines. You may find sections in this template are irrelevant 
%%%%  to your work and are empowered to omit any such section if allowed by the 
%%%%  journal you intend to submit to. The submission guidelines and policies 
%%%%  of the journal take precedence. A detailed User Manual is available in the 
%%%%  template package for technical guidance.
%%%%%=============================================================================%%%%

\usepackage{caption}
\usepackage{framed}
\usepackage{multirow}
\jyear{2021}%

%% as per the requirement new theorem styles can be included as shown below
\theoremstyle{thmstyleone}%
\newtheorem{theorem}{Theorem}%  meant for continuous numbers
%%\newtheorem{theorem}{Theorem}[section]% meant for sectionwise numbers
%% optional argument [theorem] produces theorem numbering sequence instead of independent numbers for Proposition
\newtheorem{proposition}[theorem]{Proposition}% 
%%\newtheorem{proposition}{Proposition}% to get separate numbers for theorem and proposition etc.

\theoremstyle{thmstyletwo}%
\newtheorem{example}{Example}%
\newtheorem{remark}{Remark}%

\theoremstyle{thmstylethree}%
\newtheorem{definition}{Definition}%

\raggedbottom
%%\unnumbered% uncomment this for unnumbered level heads

\begin{document}

\title[Article Title]{Identification of Block Cipher Algorithms Using Multi-Layer Perception Algorithm}

%%=============================================================%%
%% Prefix	-> \pfx{Dr}
%% GivenName	-> \fnm{Joergen W.}
%% Particle	-> \spfx{van der} -> surname prefix
%% FamilyName	-> \sur{Ploeg}
%% Suffix	-> \sfx{IV}
%% NatureName	-> \tanm{Poet Laureate} -> Title after name
%% Degrees	-> \dgr{MSc, PhD}
%% \author*[1,2]{\pfx{Dr} \fnm{Joergen W.} \spfx{van der} \sur{Ploeg} \sfx{IV} \tanm{Poet Laureate} 
%%                 \dgr{MSc, PhD}}\email{iauthor@gmail.com}
%%=============================================================%%

\author[1,2]{Ke Yuan}\email{yuanke@henu.edu.cn}

\author[1]{Daoming Yu}\email{yudaoming@henu.edu.cn}


\author[1]{Wei Yang}\email{yangwei2286@henu.edu.cn}


\author[1]{Zhanfei Du}\email{duzhanfei@henu.edu.cn}



\author[1]{Lin Shen}\email{shenlin@henu.edu.cn}


\author*[1]{Zheng Li}\email{lizheng@henu.edu.cn}


\affil*[1]{School of Computer and Information Engineering, Henan University, Henan Kaifeng, 475004, China}

\affil[2]{Henan Province Engineering Research Center of Spatial Information Processing, Henan University, Kaifeng 475004, China}

%%==================================%%
%% sample for unstructured abstract %%
%%==================================%%

\abstract{Cryptographic algorithm identification is the process of distinguishing or identifying the cryptographic algorithm by analysing the potential information of various features in the ciphertext under the condition of known ciphertext, which is the basis of cryptanalysis work. To solve the worse identification accuracy and stability problem of the single-layer scheme in cryptographic algorithm identification work as the complexity of ciphertext enhances, the interference between ciphertext data and the number of encryption algorithms increases, we propose a cryptographic algorithm identification scheme of block cipher algorithm using deep learning algorithm Multi-Layer Perception (MLP) in this paper. In this scheme, 15 NIST randomness test methods are used to extract ciphertext feature, and 10 meaningful features are selected as input to MLP classifier to make predictions. In the ciphertext-only scenario, five block cipher algorithms, AES, 3DES, Blowfish, CAST, and RC2, are selected for the study of cryptographic algorithm identification, then binary classification and multiclassification identification are carried out. The experimental results demonstrate that, when the size of the ciphertext files and other experimental conditions are the same as for the five conventional machine learning models, the proposed scheme has superior accuracy and stability. Among them, the average identification accuracy of binary classification is 76.5\% for ciphertext files of 1kB to 512kB size, which is 35.3\%, 37.5\%, 34.2\%, 38.5\%, and 41.6\% higher than that of the traditional classical machine learning models SVM, GNB, KNN, RF, and LR respectively. The average identification accuracy of the multiclassification is 36.2\%, which is significantly higher than the other five classical machine learning algorithms. When the ciphertext file sizes are different, among the identification rates of the six models fluctuate, the MLP model has the greatest stability and the least amount of influence. }

%%================================%%
%% Sample for structured abstract %%
%%================================%%

% \abstract{\textbf{Purpose:} The abstract serves both as a general introduction to the topic and as a brief, non-technical summary of the main results and their implications. The abstract must not include subheadings (unless expressly permitted in the journal's Instructions to Authors), equations or citations. As a guide the abstract should not exceed 200 words. Most journals do not set a hard limit however authors are advised to check the author instructions for the journal they are submitting to.
% 
% \textbf{Methods:} The abstract serves both as a general introduction to the topic and as a brief, non-technical summary of the main results and their implications. The abstract must not include subheadings (unless expressly permitted in the journal's Instructions to Authors), equations or citations. As a guide the abstract should not exceed 200 words. Most journals do not set a hard limit however authors are advised to check the author instructions for the journal they are submitting to.
% 
% \textbf{Results:} The abstract serves both as a general introduction to the topic and as a brief, non-technical summary of the main results and their implications. The abstract must not include subheadings (unless expressly permitted in the journal's Instructions to Authors), equations or citations. As a guide the abstract should not exceed 200 words. Most journals do not set a hard limit however authors are advised to check the author instructions for the journal they are submitting to.
% 
% \textbf{Conclusion:} The abstract serves both as a general introduction to the topic and as a brief, non-technical summary of the main results and their implications. The abstract must not include subheadings (unless expressly permitted in the journal's Instructions to Authors), equations or citations. As a guide the abstract should not exceed 200 words. Most journals do not set a hard limit however authors are advised to check the author instructions for the journal they are submitting to.}


%%\pacs[JEL Classification]{D8, H51}

%%\pacs[MSC Classification]{35A01, 65L10, 65L12, 65L20, 65L70}

\maketitle

\section{Introduction}\label{sec1}

Cryptanalysis is the study of ciphertext, ciphers and cryptosystems with the aim of understanding how to obtain a plaintext from a ciphertext without knowing the key and encryption algorithm. At present, most cryptanalysis techniques perform the analysis of various cryptographic algorithms based on the premise that the cryptographic algorithms are known. However, in general practice, researchers are only able to obtain ciphertext data without knowing the specific cryptographic algorithm used. In this case, the researchers could only perform ciphertext analysis. Therefore, efficient and accurate identification of cryptographic algorithms used in ciphertext has become an important prerequisite for cryptanalysis. There are major theoretical and practical ramifications to the growth of research on cryptographic algorithm identification.

Cryptographic algorithm identification is initially studied with classical cryptographic algorithms, which uses statistical methods to identify the ciphertext. The number of cryptographic algorithms has gradually risen with the advancement of contemporary cryptographic technology, the relationship between ciphertext data has become more complex, and traditional cryptographic algorithm identification technology has gradually become invalid. A number of machine learning-based methods for identifying cryptographic algorithms have been developed by researchers\cite{1}. The Cryptographic algorithm identification scheme based on machine learning techniques treat features as a set of attributes, equate the identification task with a classification task, and use a trained classifier to identify the data in the test set after training it on a training data set with features and algorithm labels \cite{2,3,4}.

At present, the cryptographic algorithm identification scheme using machine learning mainly focuses on the ECB and CBC modes of block cipher algorithm. In 2006, Dileep and Sekhar proposed a block cipher algorithms identification scheme based on support vector machine (SVM) which conducted a comparison experiment on the encrypted ciphertext of five cipher algorithms, including AES, DES, 3DES, RC5, and Blowfish in ECB and CBC modes \cite{5}. The outcomes demonstrated that the Gaussian kernel support vector machine model performs best, and that ECB mode has a higher identification accuracy than CBC mode. In 2011, Sharif et al. concluded that the Rotation Forest model had a greater effect by evaluating the identification impacts of several classification models in the process of block cipher algorithm identification\cite{6}. In 2012, Chou et al. use support vector machine (SVM) to identify cryptographic algorithms in a document\cite{7}. The scheme extracts 12 kinds of ciphertext features and makes a distinction experiment between encrypted ciphertext in ECB and CBC mode using AES and DES cipher algorithms for three different plaintexts of text, picture and video. Experiments show that the two cryptographic algorithms can successfully distinguish the ciphertext in ECB mode, but not in CBC mode, and the identification performance is poor when the plaintext is a picture file. In 2015, Wu et al. proposed a two-layer identification scheme based on the distribution characteristics of ciphertext randomness measures\cite{8}, and carried out a cluster analysis on the number of randomness values of five block cipher algorithms, AES, Camellia, DES, 3DES and SMS4, to effectively identify representative block ciphers. In 2018, Mello and Jam found that the proposed algorithm can almost completely identify all selected ciphers by analysing ciphertext files encrypted by seven cipher algorithms, ARC4, Blowfish, DES, Rijdael, RSA, Serpent and Twofish under the two modes of ECB and CBC, indicating that the ECB mode has higher identification accuracy than CBC \cite{9}. In 2019, Zhao et al. proposed a cryptographic algorithm identification scheme based on randomness detection \cite{10}, which proved that randomness detection can effectively extract ciphertext features. In the same year, Arvind and Ram proposed a method for identifying and isolating image-encrypted communication traffic using bit-plane image features and fuzzy decision criteria that use the same key encryption \cite({11}. The experiment proves that the scheme can successfully identify  both ordinary images and images encrypted using the same key.

The above analysis shows that the current cryptographic algorithm identification scheme has the following three characteristics:

(1) Machine learning classifier models are the foundation of the majority of identification schemes. For example, Mello and Xexeo use a single support vector machine classifier identification scheme to identify cryptographic algorithms \cite{12}. Fan and Zhao used multiple classification models of random forest, logistic regression, and support vector machine to identify eight block cipher schemes \cite{13}. In this paper, we proposed a deep learning MLP classification model to identify the cryptographic algorithm. The results show that the deep learning method has higher stability and accuracy.

(2) The factors involved in the identification scheme are relatively single. At present, the identification scheme of cryptography algorithm only considers the influence of encryption mode on the identification effect, which is not comprehensive. Other factors should also have a certain impact on the accuracy of identification. In this experiment, we compared the ciphertext identification results of different sizes and found that when the encryption mode is the same and the ciphertext size is different, identification accuracy is different.

(3) Poor stability of identification scheme. At present, traditional machine learning classification models are applied to most cryptographic algorithm identification schemes.. Decision trees and support vectors are prone to overfitting when dealing with classification problems, which results in only local optimal classification results. Although the support vector machine classifier has strong generalization ability, its training cost is high, and the adjustment of parameters and selection of kernel function will greatly affect the classification results. By contrast, the deep learning MLP algorithm can process large data sets, overcome the overfitting phenomenon, and has strong learning ability and high identification accuracy. By contrast, the current mainstream deep learning algorithms are based on neural networks. Depth means the level of neural networks. Its advantage is that it can mine the internal hidden laws of data and learn abstract knowledge \cite{14}. Therefore, the identification scheme of cryptographic algorithm using deep learning model has a great development space in the task of cryptographic algorithm identification.

The majority of currently used identification schemes are built around block ciphers and utilize statistical techniques and conventional machine learning algorithms. As the complexity of cryptographic algorithms increases, the data sets need to be processed, and the accuracy and stability of cryptographic algorithms have higher requirements. Therefore, in order to increase the accuracy and stability of the cipher algorithm identification scheme, this work offers a identification scheme of block cipher algorithm using MLP algorithm and applies it to the ciphertext files of varied file sizes. The two components of the technique are the extraction of ciphertext features and the creation of a classifier for identifying ciphertext algorithms. Based on the randomness test standard of NIST, an improved ciphertext feature extraction method was designed. A deep learning MLP algorithm was utilized to build a classifier, and the retrieved ciphertext feature data served as its input. The the cryptographic algorithm identification task is finally finished after the classification model has been trained and tested \cite{15}.

The order of the remaining text is as follows. Section 1 introduces the related concepts of cryptographic algorithm identification and the research context and achievements in recent years, and makes a brief analysis of the existing main work and methods. Section 2 introduces the basic principle of cryptographic algorithm identification. Section 3 mainly introduces the improved method based on the NIST randomness test for ciphertext feature extraction. Section 4 gives a complete scheme of cryptographic algorithm identification. The experimental results are presented and analysed in Section 5. Finally, section 6 summarizes the paper and looks forward to the follow-up work.


\section{Basic Principles of Cryptographic Algorithm Identification}\label{sec2}
\textbf{Definition 1 (Ciphertext)}Consider a set of cryptographic algorithms 
\begin{equation}
                 A=\{a_1,a_2,\ldots,a_k\} 
\end{equation}
where k is the number of cipher algorithms. For any given cryptographic algorithm $a_i,1\le i\le k$, the ciphertext file generated using plaintext encryption is
\begin{equation}
               F=\{b_1,b_2,\ldots,b_m\} 
\end{equation}
where $b_i$ is the   character of the ciphertext file and can be represented as different types of characters.

\textbf{Definition 2 (Ciphertext features)}Extract the feature of the ciphertext file F to obtain the feature with dimension d
\begin{equation}
              fea=\{x_1,x_2,\ldots,x_d\} 
\end{equation}
where fea represents d-dimensional features extracted from the ciphertext by the scheme.

\textbf{Definition 3 (Cryptographic algorithm identification)}
For the cipher algorithm set A and the ciphertext file F, when the cipher algorithm $a_i$ is unknown, the process of using a certain identification scheme I to crack the ciphertext only and identify its cipher algorithm $a_i$ with a certain accuracy h is called cipher algorithm identification, which is recorded as a triple
\begin{equation}
              \triangle=(A,I,h)
\end{equation}
where the accuracy rate h is considered as a natural measure for evaluating the identification technology and identification scheme of cryptographic algorithms.

\textbf{Definition 4 (Cryptographic algorithm identification scheme) }n cryptographic algorithm identification $\triangle=(A,I,{h})$, the cryptographic algorithm identification scheme is represented by triples:
\begin{equation}
              P=(oper,fea,CA)
\end{equation}
where oper is a workflow directly identifying a particular encryption algorithm; fea is a series of features extracted from ciphertext file F; CA is the identification model adopted by the cryptographic algorithm identification scheme P, which is called MLP model in this paper.
The specific operconsists of two stages: training and testing. The specific process is as follows:

\textbf{(1)}Collect a group of ciphertext files $C_{\mathrm{Tr}1}$,${C_{\mathrm{Tr}}}_2,...,C_{\mathrm{Tr}n}$ with a known cryptosystem, where n is the number of files;

\textbf{(2)}NIST randomness test method extracts the ciphertext features of this group of ciphertext files and collects a set of ciphertext features $FeaTr={feaTr_i^j\|i=1,2,...,n,j=1,2,...,d}$; Any ciphertext $featurefeaTr_i$ is the feature vector of d dimension.

\textbf{(3)}The cryptographic algorithm of ciphertext files (known) is written down as an n-dimensional vector Lab, which is used as the label of feature data, $Lab=(lab_1,lab_2,...,lab_n)$, where n is the number of ciphertext files. Binary groups (FeaTr,Lab) consisting of feature set FeaTr and label set Lab are referred to as tagged ciphertext feature sets.

\textbf{(4)}The labeled feature sets $\left(FeaTr,Lab\right)$ are submitted to the classification algorithm CA for the training of the classification model.

\textbf{Test phase:}
\textbf{(1)}Using the same feature extraction method, feature extraction is carried out on the file F to be recognized with an unknown cryptographic algorithm label, and the d-dimensional feature FeaTe, $FeaTe={feaTe^j\|j=1,2,...,d} $is obtained;

\textbf{(2)}Input the feature data FeaTe into the classifier CA trained in the training stage, and the classifier gives the identification result of the cryptographic algorithm of ciphertext F based on the feature data, that is, the label Lab of the cryptographic algorithm. Its working flow chart is shown in Figure 1.
\begin{figure}[h!]
	\begin{center}
		\includegraphics[width=12cm]{Figure1}% This is a *.eps file
	\end{center}
\centering
	\caption{Workflow of cryptographic algorithm identification}
	\label{fig:1}
\end{figure}


\section{Feature Extraction}\label{sec3}
\subsection{Mathematical Basis of Randomness Test}\label{subsec1}
Randomness testing usually checks whether or not it satisfies certain features of random sequences such as periodicity, correlation, property distribution and so on, so as to determine whether it is random \cite{16}. Among them, many randomness test schemes for cryptographic system security measurement defined in NIST FIPS140-2 \cite{17} published by the National Institute of Standards and Technology (NIST) in May 2001 are highly recognized. The theoretical basis of the randomness test is the hypothesis test \cite{18}. In a hypothesis test, the proposition that researchers want to test its correctness is generally called the zero hypothesis, which is usually recorded as $H_0$. Accordingly, the proposition that makes the original hypothesis untenable is called the alternative hypothesis, which is usually recorded as $H_1$. For example, in the randomness test, it is generally set to assume that $H_0$: the sequence is random; the alternative hypothesis $H_1$: the sequence is not random. The test results of the data are obtained through the randomness test. For each test, if the result supports the original hypothesis, then the sequence is considered random; otherwise, if the alternative hypothesis is supported, then the sequence is considered non-random. The specific steps of inspection are as follows:

\textbf{(1)}Propose a original hypothesis to be tested, the symbol is $H_0$; and an alternative hypothesis with the symbol $H_1$.

\textbf{(2)}Select the statistical method and calculate the size of the statistic according to the corresponding formula from the sample observation value. The chi-square test was selected in this paper according to the type and features of the data, and the statistic was X.

\textbf{(3)}Preset significant$ level \alpha$. Find critical and rejection regions in the distribution quantifier table corresponding to statistics X;

\textbf{(4)}The statistical values X calculated from the samples are compared with the detected criticality values. That is, if the statistical value X enters the rejection field, $H_0$ is rejected, and $H_0$ is accepted.

Hypothesis testing produces two types of errors. Table 1 shows two possible errors and probabilities in the assumption test.

% Please add the following required packages to your document preamble:
% \usepackage{multirow}

\begin{table}[]

\caption{Types and probability of hypothesis testing errors}\label{tab1}
 \centering
\begin{tabular}{ccclllllll}

\cline{1-3}
\multirow{2}{*}{Indicators  Original hypothesis} & \multirow{2}{*}{Sample test results} & \multirow{2}{*}{Wrong type} &  &  &  &  &  &  &  \\
                                                 &                                      &                             &  &  &  &  &  &  &  \\ \cline{1-3}
\multirow{2}{*}{true}                            & accept                               & 0                           &  &  &  &  &  &  &  \\
                                                 & refused                              & $\alpha$                      &  &  &  &  &  &  &  \\
\multirow{2}{*}{false}                           & accept                               &  $\beta$                           &  &  &  &  &  &  &  \\
                                                 & refused                              & 0                           &  &  &  &  &  &  & \\\cline{1-3}

\end{tabular}

\end{table}



In practical applications, one commonly used method to measure randomness is the P-valuemethod. We assume that the sample statistic Xfollows the chi-square distribution. Figure 2 depicts the distribution's probability density curve. To decide whether to accept the original hypothesis, first obtain the statistic X using a specific method, then compute the integral from X to infinity, then compare the integral result $(P-\mathrm{value}$, the area of the shaded section in the figure 2) $with \alpha$. If P-value is 1, we can say the sequence is a completely random sequence; if P-value is 0, we can say the sequence is a completely non-random sequence. So, if $P-value\geq\alpha$, the original hypothesis is accepted and the sequence is random. Otherwise, the sequence is non-random and the result is rejected. Generally, the value range $of \alpha$ is $\left(0.001, 0.01\right)$.
\begin{figure}[h!]
	\begin{center}
		\includegraphics[width=12cm]{Figure2}% This is a *.eps file
	\end{center}
\centering
	\caption{Probability density curve of chi-square distribution and the P-value}
	\label{fig:2}
\end{figure}

\subsection{Ciphertext Feature Extraction Method}\label{subsec2}
In the identification problem of cryptographic algorithms, we use ciphertext feature as input of the identification model, which has a direct influence on the identification result. Therefore, whether the ciphertext features can be reasonably extracted to more accurately describe the information features is crucial to the success of the identification task.

We use 15 randomness testing methods, which are chosen from the NIST randomness detection package \cite{19,20}, to extract a total of 40 ciphertext features, and then 10 meaningful features are selected by the filtering feature selection method (variance selection method and correlation coefficient method). (Top 10 in Table 2), and carry out the block cipher identification task based on the deep learning MLP algorithm. After two stages of training and testing, the identification of the encryption algorithm, which the ciphertext to be tested belongs to, is finally realized. The 41 specific features can be found in Table 2.

\begin{table}[]
\caption{List of 40 extracted features}\label{tab1}
\begin{tabular}{llllllllll}
\cline{1-2}
Feature extraction method                       & Feature                                                                                                                            & \multicolumn{1}{c}{} &  &  &  &  &  &  &  \\ \cline{1-2}
The\_Runs\_Test                                 & The\_Runs\_Test                                                                                                                    & \multicolumn{1}{c}{} &  &  &  &  &  &  &  \\
The\_longest\_run\_ones\_in\_a\_block\_test     & The\_longest\_run\_ones\_in\_a\_block\_test                                                                                        & \multicolumn{1}{c}{} &  &  &  &  &  &  &  \\
The\_binary\_matrix\_rank\_test                 & The\_binary\_matrix\_rank\_test                                                                                                    &                      &  &  &  &  &  &  &  \\
The\_Non\_Overlapping\_Template\_Matching\_Test & The\_Non\_Overlapping\_Template\_Matching\_Test                                                                                    &                      &  &  &  &  &  &  &  \\
The\_Maurers\_Universal\_Test                   & The\_Maurers\_Universal\_Test                                                                                                      &                      &  &  &  &  &  &  &  \\
The\_Serial\_Test                               & \begin{tabular}[c]{@{}l@{}}The\_Serial\_Test\_1\\    \\ The\_Serial\_Test\_2\end{tabular}                                          &                      &  &  &  &  &  &  &  \\
The\_Approximate\_Entropy\_Test                 & The\_Approximate\_Entropy\_Test                                                                                                    &                      &  &  &  &  &  &  &  \\
The\_Cumulative\_Sums\_Test                     & \begin{tabular}[c]{@{}l@{}}The\_Cumulative\_Sums\_Test\_1\\    \\ The\_Cumulative\_Sums\_Test\_2\end{tabular}                      &                      &  &  &  &  &  &  &  \\
The\_Random\_Excursions\_Test                   & \begin{tabular}[c]{@{}l@{}}The\_Random\_Excursions\_Test\_1\\   \\ The\_Random\_Excursions\_Test\_8\end{tabular}                   &                      &  &  &  &  &  &  &  \\
The\_Random\_Excursions\_Variant\_Test          & \begin{tabular}[c]{@{}l@{}}The\_Random\_Excursions\_Variant\_Test\_1\\  \\ The\_Random\_Excursions\_Variant\_Test\_18\end{tabular} &                      &  &  &  &  &  &  &  \\
The Overlapping Template Matchings              & Test The Overlapping Template Matchings   Test                                                                                     &                      &  &  &  &  &  &  &  \\
The Linear Complexity Test                      & The Linear Complexity Test                                                                                                         &                      &  &  &  &  &  &  &  \\
The binary matrix rank test                     & The binary matrix rank test                                                                                                        &                      &  &  &  &  &  &  &  \\
The Non Overlapping Template Matching Test      & The Non Overlapping Template Matching   Test                                                                                       &                      &  &  &  &  &  &  &  \\
The Maurers Universal Test                      & The Maurers Universal Test                                                                                                         &                      &  &  &  &  &  &  &  \\ \cline{1-2}
\end{tabular}
\end{table}

\section{Identification Scheme}\label{sec4}
\subsection{MLP Algorithm}\label{subsec1}
MLP is based on a single-layer neural network with one or more hidden layers introduced. And it is fully connected from layer to layer. Firstly, the hidden layer is fully connected with the input layer. Assuming a vector X to present the input of the input layer, then the output of the following hidden layer is $f\left(W_1X+b_1\right)$, in which W1 is the connection coefficient, b1 is the offset, and the function f can be a commonly used sigmoid function or tanh function:

sigmoid function:
\begin{equation}
             g(x)=\frac{1}{1+e^{-x}} 
\end{equation}

tanh function:
\begin{equation}
            g(x)=\frac{e^x-e^{-x}}{e^x+e^{-x}}
\end{equation}

Secondly, the hidden layer and the following output layer are also fully connected. The hidden layer to the output layer can be regarded as a multicategory logistic regression, which is, softmax regression.

Finally, output of the output layer is $\mathrm{softmax}(W_2X_1+b_2)$, and X1 represents the output $f(W_1X+b_1)$ of the hidden layer.
\subsection{Encryption Algorithm Identification System Using the MLP Algorithm}\label{subsec1}
In this subsection, the validity test of ciphertext data in the classification task is completed by the improved method of NIST randomness test. Based on this, five grouped ciphers, AES, 3DES, CAST, Blowfish and RC2, are selected as the objects of encryption algorithm identification study. Considering the components of our method, the entire procedure can be divided into two key phases: extract ciphertext features and build encryption algorithm identification classifier. 
\begin{figure}[h!]
	\begin{center}
		\includegraphics[width=12cm]{Figure3}% This is a *.eps file
	\end{center}
\centering
	\caption{Flow chart for identifying block cipher algorithm using the MLP method}
	\label{fig:3}
\end{figure}

We use 15 randomness tests in the NIST randomness test package, and choose 10 meaningful features to categorize the ciphertext. In this paper, we use MLP method to build an encryption algorithm identification classifier, and then apply those extracted ciphertext features as input of the classifier. Then follows the training and testing phase of the algorithm, which marks completion of our method. Flow chart for these procedures is shown in Figure 3. \\\\\\




\begin{framed}
\begin{center}
\textbf{Training phase}
\end{center}

\textbf{Input:}A set of ciphertext files with labels $C=C_1,C_2,...,C_n$, where $n=k\ast F$, (k is the amount of selected encryption algorithms in a set of encryption algorithms, and F is the total amount of ciphertexts that each encryption algorithm encrypts).

\textbf{Output:}The trained model MLP.

\textbf{Step 1.}Input n ciphertexts altogether, and extract features for each ciphertext file according to the improved NIST randomness test method to obtain n sets of ciphertext feature sets $FeaTr={feaTr_i^j\|i=1,2,...,n,j=1,2,...,d}$, where $feaTr_i^j $represents the j-th feature of the i-th training ciphertext file.

\textbf{Step 2.}The feature set and ciphertext labels form the sample set $\left(FeaTr,Lab\right) $as the original data set, denoted as T.

\textbf{Step 3.}Input the feature set T through the input layer. There are n ciphertexts, each ciphertext file represents a sample and each sample has d features.

\textbf{Step 4.}After calculating the input weighted sum of each neuron in the hidden layer, a nonlinear division is performed by the Relu function in the activation layer, and finally, the prediction results are output through the output layer.

\textbf{Step 5.}The weights of the output layers are adjusted to update according to the total error of the prediction results, and then the weights between the implied layers are dynamically updated according to the chain rule of derivation.

\textbf{Step 6.}Recalculate the updated weights and iterate continuously to output the updated prediction results until the total error is below a set threshold.

\textbf{Step 7.}Repeat steps 3-6 to dynamically adjust the parameters such as the neural network’s amount of layers, the amount of neurons per layer, the maximum value of iterations, and the learning rate to construct a good model.
\end{framed}

\begin{framed}
\begin{center}
\textbf{Testing phase}
\end{center}

\textbf{Input:}A set of ciphertext files without labels$ C_{\mathrm{Te}}=C_{\mathrm{Te}1},{C_{\mathrm{Te}}}_2,...,C_{\mathrm{Te}s}$.

\textbf{OutPut:}Classification results for each ciphertext file $at_1,at_2,...,at_s$.

\textbf{Step 1.}Perform feature extraction on the set of ciphertext documents to be identified CT to acquire the set of ciphertext features $FeaTe={feaTe_i^j \|i=1,2,...,d}$, where $feaTe_i^jrepresents$ the j-th feature of the i-th test ciphertext file.

\textbf{Step 2.}The well-trained classification model takes the ciphertext feature FeaTe as input, and generates the identification result, which is presented by the encryption algorithm label Lab corresponding to that ciphertext.
\end{framed}




\section{Experimental Environment}\label{sec5}

\subsection{Data Preparation}\label{subsec1}
In this paper, we improve on an open-source tool $sp800_22_testsmaster$ to implement ciphertext feature extraction. As for the whole program, we use the main program to read ciphertext files from disk, and every randomness test algorithm is packaged into a submodule,  each submodule is independent of each other and does not interfere with each other. When performing feature extraction, each module executes in parallel and saves the results to different files.

We choose to encrypt the ciphertext file using python's Crypto cryptographic algorithm library. We use the Fortuna Accumulator method in the above library to generate plaintexts, so these plaintexts are totally random. The plaintexts include a total of 500 files of sizes 1KB, 8KB, 64KB, 256KB, and 512KB. The encryption keys and initialization vectors are generated by Crypto's cipher encryption module, and we used AES, 3DES, Blowfish, CAST, and RC2 to generate ciphertexts in ECB mode with the setting of a static 16-bit string key. During the experiment, every encryption algorithm generated 500 ciphertext files, and in total 2500 ciphertext files are generated. Using the 15 randomness tests mentioned above as the ciphertext feature extraction method, feature extraction process is performed on 2500 ciphertext files respectively, and the return value of the randomness test is used as the ciphertext feature. Corresponding to a set of features, figure the feature values of every ciphertext file with these 15 randomness methods, then filter out 10 sets of meaningful features, and save these values as the input of the classifier in the identification scheme. We perform repeated random subsampling verification [21, 22] on the data obtained by extracting features from ciphertext files, and randomly divide all samples into two parts, the share of the first part is 80\% and the second is 20\%. The former part is used as training set, and the latter part is used as test set. Then perform 10-fold repeated random subsampling validation on the test set, take the average accuracy as a measure of the identification effect, finally conduct experiments of binary classification and five classification identification on the five cryptographic algorithms mentioned above. The precise parameters of the five encryption algorithms for ciphertext data collection are displayed in Table 3.
\begin{table}[]
\centering
\caption{Specific parameters list of five block cipher algorithms}
\begin{tabular}{ccccccllll}
\cline{1-6}
Algorithm & Structure & Key   & Mode & Parameter & Implementation &  &  &  &  \\ \cline{1-6}
AES       & SP        & Fixed & ECB  & Fixed     & Crypto         &  &  &  &  \\
3DES      & Feistel   & Fixed & ECB  & Fixed     & Crypto         &  &  &  &  \\
Blowfish  & Feistel   & Fixed & ECB  & Fixed     & Crypto         &  &  &  &  \\
CAST      & Feistel   & Fixed & ECB  & Fixed     & Crypto         &  &  &  &  \\
RC2       & Feistel   & Fixed & ECB  & Fixed     & Crypto         &  &  &  &  \\ \cline{1-6}
\end{tabular}
\end{table}





\subsection{The Evaluation Standards for Results of Classification}\label{subsec1}
Commonly used evaluation methods in classification are error rate, accuracy rate, precision rate, recall rate, F1 score, Receiver Operating Characteristic Curve (ROC) and Area Under ROC Curve (AUC), etc. Here the model is evaluated using a confusion matrix. Four kinds of results are produced by the confusion matrix, these are True Positive $(X_{\mathrm{TP}})$, True Negative $(X_{\mathrm{TN}})$, False Positive $(X_{\mathrm{FP}})$, and False Negative $(X_{\mathrm{FN}})$. Let $X_{\mathrm{TP}}$, $X_{\mathrm{TN}}$, $X_{\mathrm{TN}}$, and $X_{\mathrm{FN}}$ denote the corresponding number of samples, respectively, and the three indicators listed below are used to calculate the accuracy, precision, and recall of the model.


\begin{equation}
           Y_{\mathrm{Accuracy}}=\frac{X_{\mathrm{TP}}+X_{\mathrm{TN}}}{X_{\mathrm{TP}}+X_{\mathrm{TN}}+X_{\mathrm{FP}}+X_{\mathrm{FN}}}
\end{equation}

\begin{equation}
          {Y\frac{X_{TP}}{X_{TP}+X_{FP}}}_{Precision}
\end{equation}

\begin{equation}
          {Y\frac{X_{TP}}{X_{TP}+X_{FN}}}_{Recall}
\end{equation}
\begin{equation}
          F_1=\frac{2X_{TP}}{2X_{TP}+X_{\mathrm{FP}}+X_{\mathrm{FN}}}
\end{equation}

Among them, $Y_{\mathrm{Accuracy}}$ is accuracy, which is the ratio of all correctly assessed classification model outcomes to all observations; $Y_{Precision}$ is precision, which is the ratio of correct predictions in all results of the positive prediction model; $Y_{Recall}$ is recall, which is the ratio of correct predictions in all results of the true values being positive.

The two metrics of $Y_{Precision}$ and $Y_{Recall}$are mutually exclusive. In general, while the $Y_{Precision}$ is high, the $Y_{Recall}$ is low, and vice versa when the $Y_{Recall}$ is high. To balance $Y_{Precision}$ and $Y_{Recall}$, this paper uses the harmonic mean F1-score of precision and recall to evaluate the classifier’s performance.

In the research of this paper, accuracy is used as the performance evaluation standard for the identification model in all cryptographic algorithm identification systems. This is because the classification accuracy in the identification problem of the encryption algorithm is the focus of this topic's research more than anything else.

\subsection{Experimental Results and Analysis}\label{subsec3}
\subsubsection{Binary Classification of Encryption Algorithms}\label{subsubsec1}
% Please add the following required packages to your document preamble:
% \usepackage{multirow}
% \usepackage[table,xcdraw]{xcolor}
% If you use beamer only pass "xcolor=table" option, i.e. \documentclass[xcolor=table]{beamer}
\begin{table}[]

\caption{Results of ciphertext feature binary classification based on six classification models (\%)}
\centering
\begin{tabular}{cccccccccc}
\cline{1-8}
                                       &                             & \multicolumn{6}{c}{Classifier}                                        &  &  \\ \cline{3-8}
\multirow{-2}{*}{Evaluating Indicator} & \multirow{-2}{*}{File Size} & SVM   & GNB   & KNN   & RF    & LR    & MLP                           &  &  \\ \cline{1-8}
                                       & 1KB                         & 0.5   & 44    & 0.55  & 0.525 & 0.4   & 0.725 &  &  \\
                                       & 8KB                         & 0.525 & 0.52  & 0.525 & 0.575 & 0.5   & 0.725                         &  &  \\
                                       & 64KB                        & 0.625 & 0.6   & 0.6   & 0.65  & 0.6   & 0.775                         &  &  \\
                                       & 256KB                       & 0.575 & 0.62  & 0.575 & 0.625 & 0.54  & 0.775                         &  &  \\
\multirow{-5}{*}{Accuracy}             & 512KB                       & 0.6   & 0.6   & 0.6   & 0.6   & 0.54  & 0.825 &  &  \\
                                       & 1KB                         & 0.42  & 0.44  & 0.53  & 0.615 & 0.399 & 0.775 &  &  \\
                                       & 8KB                         & 0.53  & 0.539 & 0.532 & 0.6   & 0.525 & 0.777 &  &  \\
                                       & 64KB                        & 0.62  & 0.615 & 0.594 & 0.65  & 0.625 & 0.785 &  &  \\
                                       & 256KB                       & 0.58  & 0.638 & 0.583 & 0.628 & 0.543 & 0.79  &  &  \\
\multirow{-5}{*}{Precision}            & 512KB                       & 0.58  & 0.626 & 0.601 & 0.6   & 0.543 & 0.867 &  &  \\
                                       & 1KB                         & 0.42  & 0.44  & 0.55  & 0.525 & 0.4   & 0.725 &  &  \\
                                       & 8KB                         & 0.52  & 0.52  & 0.525 & 0.575 & 0.5   & 0.725 &  &  \\
                                       & 64KB                        & 0.66  & 0.6   & 0.6   & 0.65  & 0.6   & 0.775 &  &  \\
                                       & 256KB                       & 0.58  & 0.62  & 0.575 & 0.625 & 0.54  & 0.775 &  &  \\
\multirow{-5}{*}{Recall}               & 512KB                       & 0.58  & 0.6   & 0.6   & 0.6   & 0.54  & 0.825 &  &  \\
                                       & 1KB                         & 0.5   & 0.44  & 0.55  & 0.525 & 0.4   & 0.725 &  &  \\
                                       & 8KB                         & 0.525 & 0.52  & 0.525 & 0.575 & 0.5   & 0.725                         &  &  \\
                                       & 64KB                        & 0.625 & 0.6   & 0.6   & 0.65  & 0.6   & 0.775                         &  &  \\
                                       & 256KB                       & 0.575 & 0.62  & 0.575 & 0.625 & 0.54  & 0.775                         &  &  \\
\multirow{-5}{*}{F1-score}             & 256KB                       & 0.575 & 0.6   & 0.575 & 0.625 & 0.54  & 0.775                         &  &  \\ \cline{1-8}
\end{tabular}
\end{table}




A classification model was created on the basis of the ten extracted ciphertext features mentioned above, and the classification results of the classical machine learning models SVM, Gaussian Naive Bayes (GNB), K-Nearest Neighbor (KNN), Random Forest (RF), Logistic Regression (LR), and the deep learning MLP model under ciphertext files of various sizes encrypted by AES and 3DES were calculated using 10-fold repeated random subsampling verification. The first column represents the evaluation criteria of the classification results. The second column represents the ciphertext file size, which are 1KB, 8KB, 64KB, 256KB, and 512KB, respectively. The identification outcomes from the six classification models are shown in the third column. Table 4 presents the outcomes.

Table 4 shows that on ciphertext files ranging in size from 1 KB to 512 KB, the average identification accuracy for the binary classification of the standard classical classification models SVM, GNB, KNN, RF, and LR is 0.565, 0.556, 0.57, 0.595, and 0.54 correspondingly. The MLP model's average identification accuracy is 0.765, which is greater than the average of the five machine learning models by 0.353, 0.375, 0.342, 0.385, and 0.416. As can be observed, the classification outcomes of the MLP model outperform those of the five conventional machine learning classification models for a variety of ciphertext file sizes.

With ciphertext file sizes of 1KB, 8KB, 64KB, 256KB, and 512KB, respectively, the average identification accuracy for the machine learning classification models SVM, GNB, KNN, RF, and LR is 0.483, 0.529, 0.615, 0.587, and 0.588. The identification accuracy of the deep learning model MLP is at least 0.725 on the condition that file size of the ciphertext is 1 KB and 8 KB, and it is at its highest on the condition that file size of the ciphertext is 512 KB. When the ciphertext file size is 64 KB, the MLP model has the weakest improvement effect when compared to the five classic machine learning models, although the identification accuracy improves by 26\%. The best enhancement effect occurs when the ciphertext file is 8 KB in size. At this time, it is improved by 50\%. As can be observed, when the ciphertext file size is the same, the deep learning MLP model has the highest performance. The classification outcomes outperform the average identification precision of the other five conventional machine learning classification models by a wide margin.

The MLP classification model has the highest F1-score among the five ciphertext sizes, and the accuracy has reached 0.725 and above, suggesting that it has the best effect, according to a comparison of the F1-scores of the six classification models.

\begin{figure}[h!]
	\begin{center}
		\includegraphics[width=12cm]{Figure4}% This is a *.eps file
	\end{center}
\centering
	\caption{Classification accuracy of six classification models in  binary classification identification under five ciphertext files}
	\label{fig:4}
\end{figure}
Figure 4 displays the classification accuracy for each of the six classification models in the binary identification for ciphertext of the five sizes. Figure 4 demonstrates that the MLP classification model has the highest classification accuracy, which is greater than or equal to 0.725. The graph also shows a fluctuating change in tandem with the change in text length. However, the MLP classification model has the least volatility when compared to the rest five classification models, suggesting that classification accuracy may be somewhat influenced by the ciphertext file’s size. This MLP classification model provides higher stability since the file size has a smaller impact on the classification accuracy of the model.

\subsubsection{Five classification recognition of cryptographic algorithm.}\label{subsubsec1}
This section uses SVM, GNB, KNN, RF, LR, and deep learning MLP classification algorithms to perform 10 repeated random subsampling verifications on ciphertext files that have been encrypted using the AES, 3DES, Blowfish, CAST and RC2 cipher algorithms, and perform cipher algorithm identification. The test outcomes are displayed in Table 5.
% Please add the following required packages to your document preamble:
% \usepackage{multirow}
% \usepackage[table,xcdraw]{xcolor}
% If you use beamer only pass "xcolor=table" option, i.e. \documentclass[xcolor=table]{beamer}
\begin{table}[]
\caption{Identification results of five classifications of ciphertext features based on six classification models (\%)}
\begin{tabular}{cccccccccc}
\cline{1-8}
                                       &                             & \multicolumn{6}{l}{Classifier}                                        &  &  \\ \cline{3-8}
\multirow{-2}{*}{Evaluating Indicator} & \multirow{-2}{*}{File Size} & SVM   & GNB   & KNN   & RF    & LR    & MLP                           &  &  \\ \cline{1-8}
                                       & 1KB                         & 0.176 & 0.264 & 0.192 & 0.288 & 0.256 &0.37  &  &  \\
                                       & 8KB                         & 0.176 & 0.208 & 0.192 & 0.224 & 0.184 & 0.37                          &  &  \\
                                       & 64KB                        & 0.152 & 0.176 & 0.24  & 0.168 & 0.16  & 0.34                          &  &  \\
                                       & 256KB                       & 0.184 & 0.136 & 0.216 & 0.192 & 0.168 & 0.39                          &  &  \\
\multirow{-5}{*}{Accuracy}             & 512KB                       & 0.31  & 0.17  & 0.17  & 0.24  & 0.19  &0.34  &  &  \\
                                       & 1KB                         & 0.139 & 0.249 & 0.162 & 0.281 & 0.211 & 0.451 &  &  \\
                                       & 8KB                         & 0.2   & 0.244 & 0.262 & 0.286 & 0.24  & 0.463 &  &  \\
                                       & 64KB                        & 0.144 & 0.193 & 0.264 & 0.175 & 0.147 &0.4   &  &  \\
                                       & 256KB                       & 0.184 & 0.134 & 0.21  & 0.208 & 0.19  & 0.391 &  &  \\
\multirow{-5}{*}{Precision}            & 512KB                       & 0.315 & 0.176 & 0.196 & 0.265 & 0.199 & 0.421 &  &  \\
                                       & 1KB                         & 0.176 & 0.264 & 0.192 & 0.288 & 0.256 & 0.37  &  &  \\
                                       & 8KB                         & 0.176 & 0.208 & 0.192 & 0.224 & 0.184 &0.34  &  &  \\
                                       & 64KB                        & 0.152 & 0.176 & 0.24  & 0.168 & 0.16  & 0.34  &  &  \\
                                       & 256KB                       & 0.184 & 0.136 & 0.216 & 0.192 & 0.168 & 0.39  &  &  \\
\multirow{-5}{*}{Recall}               & 512KB                       & 0.31  & 0.17  & 0.17  & 0.24  & 0.19  &0.34  &  &  \\
                                       & 1KB                         & 0.176 & 0.264 & 0.192 & 0.288 & 0.256 & 0.37  &  &  \\
                                       & 8KB                         & 0.176 & 0.208 & 0.192 & 0.224 & 0.184 & 0.34                          &  &  \\
                                       & 64KB                        & 0.152 & 0.176 & 0.24  & 0.168 & 0.16  & 0.34                          &  &  \\
                                       & 256KB                       & 0.184 & 0.136 & 0.216 & 0.192 & 0.168 & 0.39                          &  &  \\
\multirow{-5}{*}{F1-score}             & 512KB                       & 0.31  & 0.17  & 0.17  & 0.24  & 0.19  & 0.34                          &  &  \\ \cline{1-8}
\end{tabular}
\end{table}


As can be seen from Table 5, the ciphertext file’s size has a significant influence on the identification accuracy of the MLP model for the multiclassification tasks of AES, 3DES, Blowfish, CAST, and RC2, but the accuracy of the MLP identification model for the ciphertext file is not less than 0.34. The average identification accuracy of multiclassifications is 0.362, while the maximum accuracy of the MLP identification method is 0.39 for 256KB ciphertext files.

Figure 5 shows the five classification accuracies of the 6 classifiers in the ciphertext file with a size of 1KB ~ 512KB. Figure 5 shows that the classification accuracy of the five conventional machine learning algorithms' cryptographic algorithm identification scheme is low and varies around 0.2. On the 64KB ciphertext file, the KNN classifier has the lowest accuracy rate of 0.120, and the SVM classifier has the highest identification accuracy on 512KB, at 0.310. The identification accuracy of the MLP algorithm-based cryptographic algorithm is clearly superior to the random classification accuracy rate of 20\% and much greater than that of the other five classical algorithms. The identification rate of the six models also varies concurrently with changes in the ciphertext file size. The MLP model stands out among them for having the shortest fluctuation range and the greatest model stability. 
\begin{figure}[h!]
	\begin{center}
		\includegraphics[width=12cm]{Figure5}% This is a *.eps file
	\end{center}
\centering
	\caption{Classification accuracy of six classification models in five classification identifications under five ciphertext files}
	\label{fig:5}
\end{figure}



\section{Summary and prospects}\label{sec6}
This research suggests a block cipher identification model on the basis of deep learning MLP algorithm, which is in light of the benefits of deep learning. The five standard block ciphers AES, 3DES, CAST, Blowfish, and RC2 are utilized as identification objects in the ciphertext-only scenario. By using the NIST random test method, the ciphertext features are extracted, and these features serve as the general indication for the following encryption algorithm identification tasks. The experimental results establish the validity that the MLP method not only outperforms the five traditional classical classification models—SVM, GNB, KNN, RF, and LR—in terms of accuracy and stability for binary classification and five classification tasks. This deep learning-based block cipher algorithm identification method merits further study as a novel concept with promise for future research on encryption algorithm identification.

\section*{Data Availability}\label{sec7}
The information utilized to support the study's conclusions may be found at https://github.com/woxinpengpai/mlpforencryption.


\section*{Conflict of Interest}\label{sec7}
The authors declare that there are no conflicts of interest regarding the publication of this paper.

\section*{Acknowledgments}\label{sec7}
This work was supported by the National Natural Science Foundation of China under Grant 61806074 and 61802111; the Key Research and Promotion Projects of Henan Province under Grant 222102210062; the Basic Research Plan of Key Scientific Research Projects in Colleges and Universities of Henan Province under Grant 22A413004; the National Innovation Training Program for College Students under Grant 202110475072.





\bibliography{sn-bibliography}% common bib file
%% if required, the content of .bbl file can be included here once bbl is generated
%%\input sn-article.bbl

%% Default %%
%%\input sn-sample-bib.tex%
\begin{thebibliography}{99}  

\bibitem{1} R. Manjula and R. Anitha. Identification of encryption algorithm using decision tree. Springer Berlin Heidelberg, 2011.
\bibitem{2}Liang-Tao Huang, Zhi-Cheng Zhao, and Ya-Qun Zhao. A two-stage cryptosystem recognition scheme based on random forest. CHINESE JOURNAL OF COMPUTERS, 41(2):382–399, 2018.
\bibitem{3}William AR De Souza and Allan Tomlinson. A distinguishing attack with a neural network. In 2013 IEEE 13th International Conference on Data Mining Workshops, pages 154–161. IEEE, 2013.
\bibitem{4}Suhaila Omer Sharif and Saad P Mansoor. Performance evaluation of classifiers used for identification of encryption algorithms. ACEEE Int. J. on Network Security, 2(4):42–45, 2011.
\bibitem{5}Aroor Dinesh Dileep and Chellu Chandra Sekhar. Identification of block ciphers using support vector machines. In The 2006 IEEE International Joint Conference on Neural Network Proceedings, pages 2696–2701. IEEE, 2006.
\bibitem{6}Suhaila O Sharif, LI Kuncheva, and SP Mansoor. Classifying encryption algorithms using pattern recognition techniques. In 2010 IEEE International Conference on Information Theory and Information Security, pages 1168–1172. IEEE, 2010.
\bibitem{7}Jung-Wei Chou, Shou-De Lin, and Chen-Mou Cheng. On the effectiveness of using state-of-the-art machine learning techniques to launch cryptographic distinguishing attacks. In Proceedings of the 5th ACM Workshop on Security and Artificial Intelligence, pages 105–110, 2012.
\bibitem{8}Wu Yang, Wang Tao, Xing Meng, and L Jin-dong. Block ciphers identification scheme based on the distribution character of randomness test values of ciphertext. Journal on Communications, 36(4):1–10, 2014.
\bibitem{9}Flavio Luis de Mello and Jose AM Xexeo. Identifying encryption algorithms in ecb and cbc modes using computational intelligence. J. Univers. Comput. Sci., 24(1):25–42, 2018. 
\bibitem{10}Zhi-Cheng Zhao, Ya-Qun Zhao, and Feng-Mei Liu. Scheme ofiblock ciphers recognition based on randomness test. Journal ofi Cryptologic Research, 6(2):177–190, 2019.
\bibitem{11} Arvind Ratan R (2020) Identifying traffic of same keys in cryptographic communications using fuzzy decision criteria and bit-plane measures. International Journal of System Assurance Engineering and Management, 11(2):466–480. 
\bibitem{12}Flavio Luis de Mello and Jose Antonio Moreira Xexeo. Cryptographic algorithm identification using machine learning and massive processing. IEEE Latin America Transactions, 14(11):4585–4590, 2016.
\bibitem{13}SiJie Fan and YaQun Zhao. Analysis of cryptosystem recognition scheme based on euclidean distance feature extraction in three machine learning classifiers. In Journal of Physics: Conference Series, volume 1314, page 012184. IOP Publishing, 2019.
\bibitem{14}Mo-Yu Bai, Hao Liu, Hao-Chuan Chen, and Zhen-Hua Zhang. Adaptive beamforming algorithm based on depth neural network. Journal of Telemetry, Tracking and Command, 40(6):28–36, 2019.
\bibitem{15}Yuan Ke, Huang YaBing and Li JiBao. A Block Cipher Algorithm Identification Scheme Based on Hybrid Random Forest and Logistic Regression Model. Neural Process Lett, 2022. 
\bibitem{16}Yong-Qiang Zhang, Shun-Bo Li, Shuai Qu, Kai-Lin Lv, Chan Liu, and Xiao-Ru Xu. Nist randomness detection method and application. Computer Knowledge and Technology, (9X):6064–6066, 2014.
\bibitem{17}Andrew Rukhin, Juan Soto, James Nechvatal, Miles Smid, Elaine Barker, Stefan Leigh, M Levenson, M Vangel, D Banks, Nathanael Heckert, James Dray, and S Vo. A Statistical Test Suite for Random and Pseudorandom Number Generators for Cryptographic Applications. Special Publication (NIST SP), National Institute of Standards and Technology, Gaithersburg, MD, 2001.
\bibitem{18}SHENG Z. Probability Theory and Mlathematical Statistics. Higher Education Press, Beijing, 2001.
\bibitem{19}Hong-Chao Li. Cipher-text features based cipher system recognition. 2018.
\bibitem{20}Yang Wu, Tao Wang, Meng Xing, and Jin-Dong Li. Block ciphers identification scheme based on the distribution character of randomness test values of ciphertext. Journal on Communications, 36(4):146-155, 2015.
\bibitem{21}Xi-Zhi Wu. Statistics: from data to conclusion (Fourth Edition). China-Statistics, China, 2013.
\bibitem{22}Xiao-Tai Niu. Support vector extracted algorithm based on knn and 10 fold cross-validation method. 393 ournal of Huazhong Normal University(Natural ences), 48(3):335–338, 2014.


\end{thebibliography}
\end{document}
