<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "https://jats.nlm.nih.gov/publishing/1.3/JATS-journalpublishing1-3.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" article-type="research-article" dtd-version="1.3" xml:lang="en">
<front>
<journal-meta>
  <journal-id journal-id-type="publisher-id">46</journal-id>
  <journal-id journal-id-type="short-title">gssr</journal-id>
  <journal-id journal-id-type="doi">10.31703/gssr</journal-id>
  <journal-title-group>
    <journal-title>Global Social Sciences Review</journal-title>
    <abbrev-journal-title abbrev-type="publisher">gssr</abbrev-journal-title>
  </journal-title-group>
  <issn publication-format="print">2520-0348</issn>
  <issn publication-format="electronic">2616-793X</issn>
  <self-uri xlink:href="https://gssrjournal.com"/>
  <publisher>
    <publisher-name>Humanity Publications</publisher-name>
    <publisher-loc>Pakistan</publisher-loc>
  </publisher>
</journal-meta>
<article-meta>
  <article-id pub-id-type="publisher-id">394936</article-id>
  <article-id pub-id-type="doi">10.31703/gssr.2024(IX-I).11</article-id>
  <article-id pub-id-type="other" specific-use="submission-id">5403</article-id>
  <article-version article-version-type="publisher">1.0</article-version>
  <article-categories>
    <subj-group subj-group-type="heading">
      <subject>article</subject>
    </subj-group>
  </article-categories>
  <title-group>
    <article-title xml:lang="en">Automatic Spoofing Detection Using Deep Learning</article-title>
  </title-group>
<contrib-group>
  <contrib contrib-type="author" seq="1" corresp="yes">
    <name>
      <surname>Nafees</surname>
      <given-names>Muhammad</given-names>
    </name>
    <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Conceptualization" vocab-term-identifier="https://credit.niso.org/contributor-roles/conceptualization/">Conceptualization</role>
    <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing – original draft" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-original-draft/">Writing – original draft</role>
    <xref ref-type="aff" rid="aff1"/>
    <xref ref-type="corresp" rid="cor1"/>
  </contrib>
  <contrib contrib-type="author" seq="2">
    <name>
      <surname>Rauf</surname>
      <given-names>Abid</given-names>
    </name>
    <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing – review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
    <xref ref-type="aff" rid="aff2"/>
  </contrib>
  <contrib contrib-type="author" seq="3">
    <name>
      <surname>Mahum</surname>
      <given-names>Rabbia</given-names>
    </name>
    <role vocab="credit" vocab-identifier="https://credit.niso.org/" vocab-term="Writing – review &amp; editing" vocab-term-identifier="https://credit.niso.org/contributor-roles/writing-review-editing/">Writing – review &amp; editing</role>
    <xref ref-type="aff" rid="aff3"/>
  </contrib>
  <aff id="aff1">
    <label>1</label>
    <institution-wrap>
      <institution>MSc, Department of Data Science, University of Engineering and Technology, Taxila</institution>
    </institution-wrap>
    <addr-line>Punjab</addr-line>
    <country>Pakistan</country>
  </aff>
  <aff id="aff2">
    <label>2</label>
    <institution-wrap>
      <institution>MSc, Department of Statistics, Quaid-i-Azam University</institution>
    </institution-wrap>
    <addr-line>Islamabad</addr-line>
    <country>Pakistan</country>
  </aff>
  <aff id="aff3">
    <label>3</label>
    <institution-wrap>
      <institution>MS, Department of Computer Science, University of Engineering and Technology, Taxila</institution>
    </institution-wrap>
    <addr-line>Punjab</addr-line>
    <country>Pakistan</country>
  </aff>
</contrib-group>
<author-notes>
  <corresp id="cor1">Corresponding Author: Muhammad Nafees, MSc, Department of Data Science, University of Engineering and Technology, Taxila, Punjab, Pakistan.</corresp>
<fn fn-type="COI-statement" id="fn-coi">
  <p>The authors declare that they have no conflicts of interest.</p>
</fn>
<fn fn-type="ethics-statement" id="fn-ethics">
  <p>This study did not require formal ethics approval.</p>
</fn>
<fn fn-type="data-availability-statement" id="fn-data">
  <p>Data sharing is not applicable to this article.</p>
</fn>
</author-notes>
<pub-date pub-type="epub" date-type="pub" publication-format="electronic">
  <day>31</day>
  <month>03</month>
  <year>2024</year>
</pub-date>
<pub-date pub-type="collection">
  <month>03</month>
  <year>2024</year>
</pub-date>
<pub-date date-type="pub" publication-format="print">
  <day>07</day>
  <month>04</month>
  <year>2024</year>
</pub-date>
  <volume>9</volume>
  <issue>1</issue>
  <season>Winter</season>
  <fpage>110</fpage>
  <lpage>131</lpage>
  <history>
    <date date-type="accepted">
      <day>07</day>
      <month>04</month>
      <year>2024</year>
    </date>
  </history>
<funding-group>
  <funding-statement>
<p>The authors received no specific funding for this work.</p>
  </funding-statement>
</funding-group>
<permissions>
  <copyright-year>2024</copyright-year>
  <copyright-holder>Humanity Publications</copyright-holder>
  <license license-type="open-access" xml:lang="en" xlink:href="https://creativecommons.org/licenses/by/4.0/">
    <license-p>This is an open access article distributed under the terms of the Creative Commons Attribution 4.0 International License.</license-p>
  </license>
</permissions>
<self-uri content-type="text/html" xlink:href="https://gssrjournal.com/article/automatic-spoofing-detection-using-deep-learning"/>
<self-uri content-type="pdf" xlink:href="https://gssrjournal.com/pdf/gssr/gGRM3eygEY.pdf"/>
<supplementary-material id="suppl-pdf" content-type="pdf" xlink:href="https://gssrjournal.com/pdf/gssr/gGRM3eygEY.pdf">
  <label>PDF</label>
  <caption>
    <title>Full Text PDF</title>
  </caption>
</supplementary-material>
  <abstract>
    <p>Deep fakes stand out to be the most dangerous side effects of Artificial Intelligence. AI assists to produce voice cloning of any entity which is very arduous to categorize whether it’s fake or real. The aim of the research is to impart a spoofing detection system to an automatic speaker verification (ASV) system that can perceive false voices efficiently. The goal is to perceive the unapparent audio elements with maximum precision and to develop a model that is proficient in automatically extracting audio features by utilizing the ASVspoof 2019 dataset. Hence, the proposed ML-DL SafetyNet model is designed that delicately differentiate ASVspoof 2019 dataset voice speeches into fake or bonafide. ASVspoof 2019 dataset is characterized into two segments LA and PA. The ML-DL SafetyNet model is centred on two unique processes; deep learning and machine learning classifiers. Both techniques executed strong performance by achieving an accuracy of 90%.</p>
  </abstract>
<kwd-group kwd-group-type="author-keywords">
  <kwd>Fake Audio</kwd>
  <kwd>Spoof Speech Detection</kwd>
  <kwd>Deep Learning</kwd>
</kwd-group>
  <custom-meta-group>
    <custom-meta><meta-name>views</meta-name><meta-value>659</meta-value></custom-meta>
    <custom-meta><meta-name>downloads</meta-name><meta-value>0</meta-value></custom-meta>
    <custom-meta><meta-name>html-views</meta-name><meta-value>0</meta-value></custom-meta>
    <custom-meta><meta-name>google-scholar-citations</meta-name><meta-value>0</meta-value></custom-meta>
    <custom-meta><meta-name>crossref-citations</meta-name><meta-value>0</meta-value></custom-meta>
  </custom-meta-group>
</article-meta>
</front>
<body>
<sec id="sec-1">
  <title>Introduction</title>
<p>As per researchers and UCL report conclusions, fake audio has become a serious and challenging problem nowadays and appears to be one of the most perturbing applications of artificial intelligence. Likewise, research also indicates that artificial intelligence will assist criminality in different tactics for the next 15 years. Furthermore, the architectures exercised by deep fakes are so concentrated and complex, that it is gruelling to segregate and prevent authentic and fake speeches.</p><p>In the recent era, there has been a sudden growth in the technology for speaker verification systems. But as far as its useful and beneficial aspects, instead, some serious risks exist there regarding improvement in the technology. Replay spoof attacks [1]can be easily operated with smartphones by utilizing AI tools, as they do not require any prior professional experience. Furthermore, it is quite challenging for the ASV (Automatic Speaker Verification) systems, as it is difficult to identify the authentic and fake speeches. The whole audio is altered by the same voice as the genuine speaker in audio-deep fake attacks, [Gao2021]propounding a considerable threat. For example, a hacker may get control over private information and contents by effectively developing fraudulent voices to encrypt voice-print-based security systems. Likewise, a person can also deceive a bank call centre by identifying himself as a registered user and may successfully make bank representatives transfer money to his account. Furthermore, an attacker can get access to the security system that is established on the voiceprint. With the growth and advancement in technology, it is quite challenging to handle this issue artificial intelligence technology has become advanced now. As for AI technology&apos;s negative impacts, on the contrary, AI technology assisted the researchers in contributing to protection against deep fake problems, like machine learning and deep learning tools. If we look a few years back, there is a considerable advancement in the field of artificial intelligence and strong architectures are utilized that even humans cannot distinguish genuine and fake speeches. These technologies [Yi2022], [Kinnunen2017], [Todisco2019] can be used for criminal activities or illegal activities and these technologies have the capacity to affect the credibility of frequently employed biometric identification models. It is necessary to address and develop strategies for recognizing the serious damage that false audio can cause. In the same way, artificial intelligence has a great role in the field of forensics. The research is being carried out for development in the field of forensics. The objective of the research [Mcuba2022]is to detect fake audio and their association by using various deep learning techniques so that deep fakes are identified at the earliest stage. The model performed various techniques of deep learning like Mel-spectrum, MFCC etc. to accomplish improved results. Furthermore, among these techniques architecture of VGG-18 performed best for finding real and fake speeches for forensics. The importance, role and requirement of machine learning and deep learning in current years are amplified with the evolution in technology. The research [Hamza2022] confers the technique of MFCC being carried out to achieve the information regarding the audios being fake or bonafide. As the modern approaches for deep fakes are so effective that it is very hard to categorize these attacks. The genuine and fake dataset was selected and then parted into different four datasets as per the investigation. The results of the findings [Almutairi 2023] obviously show that among the various machine learning algorithms SVM is the best for chosen dataset. The research depicts the method for automatically detecting the fake audios regarding Arabic speech, as limited work is carried out for distinctive languages spoofing detection. A dataset is created based upon the modern speech of Arabic pronunciation and speech was then tested and trained with the people who are non-Arabic speakers. By using their own model for spoofing detection, the researchers achieved a good accuracy based on EER. Moreover, our proposed research could be considered to be interdisciplinary as audios are part of linguistics.</p><p><break/></p><p>Related Work</p><p>Several research studies have been published that deal with automatic spoofing detection. MissimilianoTodisco&apos;s model shows the importance of developing techniques against threats of genuine and false audios. Jiyangyan Yi developed [Yi 2021]a dataset for half-truth audio detection. MoustafaAlzantot proposed model [Alzantot2019]goal is to discriminate genuine and spoofing speeches by establishing strong defensive structures. Galina Lavrentyevastruggles [Lavrentyeva 2017]  to perceive the spoofing by using a deep learning approach for ASV spoof 2017 by using anti spoofing system. The proposed model for the ASV spoof 2017 challenge succeeded with an accuracy of 87% by using a mixture of CNN, SVM, CNN and RNN networks. B.T Balamuralai proposed [Balamurali 2019]classical GMM-UBM model achieved the comparative results of mixed machine learning and identified audio structures. Shanshan Zhang proposed [Zhang 2022] model aimed to distinguish false speeches by using pre-training models. The investigation [Dua2022] attempts to detect speaker verification by using deep learning models. The findings of the research are the fusion of two models having time-distributed dense layers, LSTM   and deep neural networks. The fusion model in this investigation performed well for CQCC features.</p><p>The research [15] is based on spoof speech detection for fake speech, voice alteration and replay attack techniques. The dawn of this era [Wenger2021] has introduced various tools that are used to deceive the world by producing audios and speeches that sound authentic as spoken by the target speaker. Furthermore, in case if these tools fall into the wrong hands, will create great hazards for the world. The hazard can be at the personal level, organizational level or at the world level.  The research actually highlights and efforts the impacts of these tools on machines and computers. Findings in the research clearly show that machines and humans can easily be fooled by the latest tools and techniques. Therefore, it is suggested to raise awareness and to develop the latest and advanced protections to protect our machines, systems and humans. As per the researcher’s interest studies [Tan 2021], text-to-speech is the most trending topic in the field of artificial intelligence which has a variety of applications. Currently, deep learning technology has improved these TTS techniques. The survey research is based on TTS which highlights the current research in the field of deep learning by utilizing various relations. The research [18] presents a corpus named SAS which comprises nine techniques of spoofing of which two are dialogue fusion and the remaining are speech renovation. Research is designed for two protocols each performing a different duty. One protocol is used for evaluating speaker verification and the other protocol is used for creating spoofing material. The research is based on the utilization of recent ideas and in the absence of any verbal language (audio) spoofing detection technique, the system has a greater chance of being attacked. This paper [Columbia2021] highlights serious threats related to spoofing. For this purpose, main ambition of researchers is to find out the spoofed speech from the bonafide one. In this research a method is utilized named the capsule network by using ASV spoof 2019 dataset for detecting audios. Major work done is based on text to speech conversion. Moreover, replay attacks were also taken as part of research and results clearly show that the model also performed well in this case.</p><p>Communication networks  [15] are taken under consideration by using MAC addresses of different devices. The main objective of the research was to detect the MAC address in a wireless medium. In this research, the experiment was conducted by using different distances from devices. This system does not depend upon the amendments of standards and protocols of the devices. The system achieved different results on the basis of the targeted device distance. The model performed best for random forests. The recent world is progressing as fast as the speed of light. Many technologies [20] have been introduced in the arena of computer science. Machine learning has also played a vital role. Nowadays various machine learning tools can be used to automatically create different deep fakes which are very similar to the real ones. Currently, deep fake videos have created a condition of distress in the world. Using ML tools, not only the common public is threatened but celebrities, politicians, actors and many other high-profile persons have been threatened. For this purpose, a model is suggested based on the CNN which is used to automatically detect the spoofed videos. RNN technique is used in the paper which differentiates the spoofed and the bonafide videos. The dataset for this research has been gathered from the various sites. As the advancement in technology [21] is growing frequently and people are facilitated by means of this technology, their worries and security concerns are also rising. Smartphone is one of the major technologies of this era. We use various apps on our smartphones and face movement, and mouth movement features are utilized in this model for detection. The application MoviePy is utilized in which cutting and editing are done on the image data containing the mouth exposed along with the visibility of teeth. DFT and CNN techniques are used to achieve the results for the detection of fake and real videos. With the passage of time and advancements in artificial intelligence techniques Ismail2021], public privacy and security are at high risk. People use AI complex architecture techniques to threaten the public by creating various fake videos in which the face of someone else is being swapped with the targeted person. Face-swapping detection is a challenging task to identify whether the video is false or genuine. The research proposed model YOLO face detector is utilized and ResNet CNN is used to excerpt structures from video frames after getting these structures XGBOOST identify either the video is fake or real.</p>
</sec>
<sec id="sec-2">
  <title>Table 1</title>
<table-wrap id="table1"><label>Table 1</label><caption><title>Table 1</title></caption><table><thead><tr><th> <p><bold>S.
   No</bold></p> </th><th> <p><bold>Authors</bold></p> </th><th> <p><bold>Dataset
   used</bold></p> </th><th> <p><bold>Technique</bold></p> </th><th> <p><bold>Accuracy</bold></p> </th><th> <p><bold>Future
   Work</bold></p> </th></tr></thead><tbody><tr><td> <p>1</p> </td><td> <p>Massimiliano Todisco, Xin
  Wang and Ville Vestman (2019)</p> </td><td> <p>ASV spoof 2019</p> </td><td> <p>Tendon detection cost
  function (t-DCF), Gaussian Mixture Model (GMM), constant Q cepstral
  coefficient</p> </td><td> <p>EER 3.92%</p> </td><td> <p>NA</p> </td></tr><tr><td> <p>2</p> </td><td> <p>Jiangyan Yi and Ye Bai (2021)</p> </td><td> <p>AISHELL-3 corpus</p> </td><td> <p>Gaussian Mixture Model (GMM),
  Light Convolution Neural Network (LCNN)</p> </td><td> <p>82%</p> </td><td> <p>Different types of fakes and to develop
  datasets for other languages</p> </td></tr><tr><td> <p>3</p> </td><td> <p>MoustafaAlzantot and Ziqi
  Wang (2019)</p> </td><td> <p>78 human voice clips</p> </td><td> <p>Linear Frequency Cepstral Coefficients
  (LFCC), <break/>
  Gaussian mixture models (GMMs)</p> </td><td> <p>t-DCF 0.1569%<break/>
  EER 6.02 %</p> </td><td> <p>Improving model against
  unknown attacks</p> </td></tr><tr><td> <p>4</p> </td><td> <p>Galina Lavrentyeva1, Sergey
  Novoselov (2017)</p> </td><td> <p>ASV spoof 2017</p> </td><td> <p>SVM i-vector, LCNN, CNN+RNN</p> </td><td> <p>85%</p> </td><td> <p>NA</p> </td></tr><tr><td> <p>5</p> </td><td> <p>B. T. Balamur Ali, Kin Wah
  Edward Lin (2019)</p> </td><td> <p>ASV spoof 2017</p> </td><td> <p>MFCCs for audio preprocessing
  and audio feature selection, and <break/>
  CCs and Autoencoders are used for input and output matching</p> </td><td> <p>EER 12.6 %</p> </td><td> <p>The proposed architecture can
  be used with the assistance of the GMM model</p> </td></tr><tr><td> <p>6</p> </td><td> <p>Mohit Dua, Chhavi Jain (2022)</p> </td><td> <p>ASV spoof 2015</p> </td><td> <p>LSTM &amp; CNN</p> </td><td> <p>1.7% ERR</p> </td><td> <p>ASV spoof 2019 dataset to be
  used for better outputs</p> </td></tr><tr><td> <p>7</p> </td><td> <p>YanminQiana, Nanxin Chena,
  Kai Yua, (2019)</p> </td><td> <p>ASV spoof 2015</p> </td><td> <p>MFCC &amp; PLP are used as
  deep features. SGD is used to train the constraints and Square Error is used
  as a function.  LSTM &amp; BLSTM are
  based models, GMM is used to model the input structures while MAP is used to
  explain the initial model that denotes genuine and spoofed speeches</p> </td><td> <p>84% for spoofing
  discriminant- DNN, <break/>
  97% for LSTM,<break/>
  97.2 % for BLSTM</p> </td><td> <p>Need to improve EER</p> </td></tr><tr><td> <p>8</p> </td><td> <p>Emily Wenger, Max Bronckers
  (2021)</p> </td><td> <p>ASV spoof 2017</p> </td><td> <p>SVM, Light CNN &amp; Custom
  CNNs are used with various functions to achieve the results</p> </td><td> <p>88%</p> </td><td> <p>Exploring subsequent
  challenges and opportunities</p> </td></tr><tr><td> <p>9</p> </td><td> <p>Abhijit Jadhav, Abhishek
  Patange, Patil, (2022)</p> </td><td> <p>YouTube, FaceForensics++</p> </td><td> <p>ResNet CNN classifier for
  mining the features. LSTM for Sequence Processing</p> </td><td> <p>94%</p> </td><td> <p>detection of the audio deep
  fakes</p> </td></tr><tr><td> <p>10</p> </td><td> <p>Huy H. Nguyen, Junichi
  Yamagishi, and Isao Echizen (2021)</p> </td><td> <p>deep fake dataset,
  FaceForensics dataset</p> </td><td> <p>capsule network</p> </td><td> <p>95.93%</p> </td><td> <p>assessing the capability of
  the proposed method to fight argumentative machine attacks</p> </td></tr><tr><td> <p>11</p> </td><td> <p>Oscar de Lima, Sean Franklin,
  Annet George (2020)</p> </td><td> <p>Celeb-DF dataset containing
  590 real videos from YouTube and 5639 fake videos</p> </td><td> <p>DFT, RCN, R3D, ResNet Mixed
  3D-2D Convolution</p> </td><td> <p>RCN- 76.25<break/>
  R2Plus- 98.07<break/>
  I3D- 92.28<break/>
  MC3- 97.49<break/>
  R3D- 98.26</p> </td><td> <p>_</p> </td></tr></tbody></table></table-wrap>
</sec>
<sec id="sec-3">
  <title>Methodology</title>
<p>The proposed methodology is divided into categories like data gathering, training and testing. We have collected data from the ASV spoof 2019 corpus containing logical access (LA) and physical access (PA) speeches. The proposed model ML-DL SafetyNet introduced a model, basically a deep learning model and a machine learning model using different features. For the machine learning based, ML-DL SafetyNet model we used different algorithms like Naïve Bayes, and KNN, likewise, using the deep learning ML-DL SafetyNet model we have utilized features of convolutional neural networks and different optimizers and filters like sigmoid, ReLU etc. In the final stage, ML-DL SafetyNet classified the desired outputs. The classifier differentiated the data into the desired categories like spoof and bonafide. The complete methodology process is shown in the figure 1 below.</p>
</sec>
<sec id="sec-4">
  <title>Figure 1</title>
<p><break/></p>
</sec>
<fig id="fig-1"><alt-text>Figure 1</alt-text><caption><title>Figure 1</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 1.jpg"/></fig>
<sec id="sec-5">
  <title>Deep Learning</title>
<p>A deep learning model contains various layers like</p><p>input layers, hidden layers and classifications layers as shown in the below figure 2. Hidden layers are subcategorized into pooling, batch normalization, convolutional, activation and other different layers. In this model, features are to be extracted using various filters by convolution. When the convolution process is performed, a filter map is generated and dimensions are reduced with the help of pooling to avoid the computational power.</p><p><break/></p><p><break/></p><p>Input Layer</p><p>The first layer receives artificial neurons and then transfers these neurons and information to the other networks connected to it.</p><p>Hidden Layer</p><p>The layers that come after the input layers are</p><p>hidden layers. These layers vary in their number for a network as per data problem. These layers can be called the backbone of the neural network because these layers are ultimately responsible for the exceptional performance of the network. These are multi-tasking layers, performing different activities and functions at the same time. The hidden layer involved many different layers segregated by their distinct names as per their roles and tasks executed by these layers. Some of these layers are named ReLU, MaxPooling, Fully connected layer, Sigmoid etc. These layers vary in their number depending on the complexity of the network and the computational cost of the system.</p><p><break/></p><p>Output Layer</p><p>This is the final layer which highlights and illustrates the desired predictions of the network. This is the sole layer in the whole network which is tasked to provide the conclusive result.</p>
</sec>
<sec id="sec-6">
  <title>Figure 2</title>
<p><break/></p>
</sec>
<fig id="fig-2"><alt-text>Figure 2</alt-text><caption><title>Figure 2</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 2.jpg"/></fig>
<sec id="sec-7">
  <title>Convolution Layer</title>
<p>This layer is the core of the neural network. An input image is transformed for extracting the features, convolved through a kernel that has a small matrix having a height and length being smaller in size than the input image. Afterwards, the kernel slides all over in length and width across the input image, after which a dot matrix is calculated. To decrease and eradicate the non-linearity from the output ReLU, Tanh or any other activation functions are utilized.</p>
</sec>
<sec id="sec-8">
  <title>Figure 2</title>
<p><break/></p>
</sec>
<fig id="fig-3"><alt-text>Figure 2</alt-text><caption><title>Figure 2</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 3.jpg"/></fig>
<sec id="sec-9">
  <title>Pooling Layer</title>
<p>The purpose of this layer is the reduction of the input image. This reduction adds more strength to the features and makes the computations fast. This layer utilizes kernel (filter) and stride. Pooling can be of different types like max pooling, and average pooling.</p><p>Fully Connected Layer</p><p>This layer has all the neurons that are connected with one another both in the successor and predecessor layers. In this layer, the input is multiplied by the weight matrix and then bias is added to it.</p><p><break/></p><p>Activation Function</p><p>In a neural network, neurons play a vital role which taking the weighted sums of the inputs and passing the resultant scalar values to the function named as an activation function. The main purpose of the activation function is to determine whether the value of input remains the same or larger than the threshold value to activate the neuron. Whenever the value of the input is smaller than the threshold value, its output value will not be sent to the next layer as no neuron will be activated.</p><p>f(x)={?(0,&amp;x&lt;0@1,&amp;x?0)?		 (1)</p><p>Activation functions can be of various types and each performs well regardless of the problem. Here in our case, we have used ReLU, sigmoid and SoftMax.</p><p>ReLU is the most commonly and widely used activation function used in almost every deep learning network. The function works as follows:</p><p>{?(0,if  &amp;x&lt;0@x,if  &amp;x?0)?					(2)</p><p>which means that the function will deliver the value back whenever the input is positive, otherwise it will return zero.</p><p>The S-shaped classification output displays a sigmoid activation function ensuing 0 or 1. It is described as follows;</p><p>A=1/((1+e^(-x)))					 (3)</p><p>Similarly, a SoftMax activation function plots all the values of vectors into the probability vectors. This type of activation function forecasts spreading probability.</p><p><break/></p><p>ML-DL SafetyNet Network</p><p>The proposed network ML-DL SafetyNet is based on deep learning and machine learning networks. The network comprises various layers. The output vectors of a fully connected layer are:</p><p>y^t=[y_1+ y_(2  )+ y_(3  )+....+ y_n]		 (4)</p><p>y=f(Wx + b)(1)				 (5)</p><p>x^t=[x_1+ x_(2  )+ x+....+ x_m] 		(6)</p><p>b^t=[b_1+ b_(2  )+ b_(3  )+....+ b_n] 		 (7)</p><p>Where f, b and w are the activation function, input column vector and biases respectively.</p><p>The network comprises various layers having convolution, max pooling and ReLU layers. Activation functions play as the backbone of a neural network. These activation functions facilitate the neural network [23] with non-linearity as they assist the network in learning complex patterns in the system. Our ML-DL SafetyNet encompasses two portions i.e. deep learning and machine learning.</p><break/>
</sec>
<sec id="sec-10">
  <title>Deep Learning-based SafetyNet Network Figure 4</title>
<p><break/></p>
</sec>
<fig id="fig-4"><alt-text>Figure 4</alt-text><caption><title>Figure 4</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 4.jpg"/></fig>
<sec id="sec-11">
  <title>Figure 5</title>
<p><break/></p>
</sec>
<fig id="fig-5"><alt-text>Figure 5</alt-text><caption><title>Figure 5</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 5.jpg"/></fig>
<sec id="sec-12">
<p>As the network has no provision of WAV audio files and exclusively accepts only the images as input, so initially we transformed all the audio WAV files into images of spectrograms to improve our system efficiency. After auditory records are converted into spectrograms, different layers like convolution, ReLU, sigmoid, fully connected, SoftMax and classout are performed on spectrograms. In the layer of image input, we have taken the audio as spectrograms. As Mel-Spectrogram images are converted, these images are then passed through the convolution layer for convolving into a smaller image by using a kernel or a filter. After the convolution, for the activation function, ReLU is utilized which avoids the exponential development in computation for a neural network. After this layer again, we have to pass our data over the convolution layer that benefits the system to absorb features upon variance scale on the image. The sigmoid function layer is then utilized which benefits to diminish the non-linearity and additionally, it ascertains the type of values to be passed and stopped as output. Furthermore, two fully connected layers are utilized for weights and biases to achieve the maximum desired results.</p>
</sec>
<sec id="sec-13">
  <title>Figure 6</title>
<p><break/></p>
</sec>
<fig id="fig-6"><alt-text>Figure 6</alt-text><caption><title>Figure 6</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 6.jpg"/></fig>
<sec id="sec-14">
  <title>Figure 7</title>
<p>Moreover, the SoftMax activation function generates the outputs of the network vectors into probability vectors. In the final stage classification output layer easily classifies and differentiates our results into desired classes on the basis of probability vectors. The crucial parameter to be considered is the learning rate in the training of deep learning and machine learning models. The learning rates considered in the ML-DL SafetyNet model are 0.01 and 0.0001. Likewise, we have used 10-fold cross-validation. Architectural details of the layers are described in table 2 below.</p><p>We navigated through a series of consecutive steps in the preceding deep learning approach to achieve the desired outcomes. In the same way, we have used another approach to attain our desired results by using machine learning. ASV spoof 2019 dataset comprises audio files in the format of .flac files. Our initial approach aims to convert audio data files into .wav files. After the conversion of .flac files to .wav files, a tool named JAudio tool for feature extraction was utilized. JAudio tool lacks the support of .flac files and accepts only .wav files. Accordingly, .flac files were converted to .wav files. In this way, we attained the features in the XML format and furthermore, features converted in Excel format. Likewise, feature selection is applied and by using different machine learning algorithms data is analyzed with accuracies, confusion matrices, ROC curves and coordinate plots. We applied various machine learning algorithms to attain the desired results. Among seven learners we achieved the best accuracy of 90% for support vector machine classifier. Therefore, in our approach, SVM is the best classifier to achieve good results.</p>
</sec>
<fig id="fig-7"><alt-text>Figure 7</alt-text><caption><title>Figure 7</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 7.jpg"/></fig>
<sec id="sec-15">
  <title>Figure 8</title>
<p>Machine Learning-based SafetyNet Network</p><p>In the case of the machine learning model initially, the dataset was loaded comprising of audio files. Loading of these audio files created an audio Datastore (ADS) that can effectively load audio. Furthermore, the most important and necessary step in every model is feature extraction. For this purpose, our proposed model of machine learning achieved feature extraction. A variety of audio features were extracted from the loaded audio files of the ASV Spoof dataset. Mel-frequency Cepstral Coefficients, GammatoneCepstral Coefficients, flux, centroid, crest, decrease, entropy, flatness, kurtosis, roll-off point, skewness, slope, spread and energy are the features that are extracted. Moreover, the average of these extracted features was taken, an Excel file was generated and correlation analysis was carried out for these sets of features and achieved the correlation coefficients. When correlation analysis is performed then different machine learning algorithms are utilized and run to achieve the best result. After applying different machine learning algorithms like KNN, Gaussian Naive Bayes, Optimizable Tree, Logistic Regression, Linear Discriminant and Support Vector Machine. The results of these classifiers are shown in Fig 13.</p>
</sec>
<sec id="sec-16">
  <title>Table 2</title>
<table-wrap id="table2"><label>Table 2</label><caption><title>Table 2</title></caption><table><thead><tr><th> <p><bold>S.No</bold></p> </th><th> <p><bold>Name</bold></p> </th><th> <p><bold>Type</bold></p> </th><th> <p><bold>Activation</bold></p> </th><th> <p><bold>Learnable</bold></p> </th></tr></thead><tbody><tr><td> <p>1</p> </td><td> <p>Image
  Input 227x227x1</p> </td><td> <p>Image
  Input</p> </td><td> <p>227x227x1</p> </td><td> <p>-</p> </td></tr><tr><td> <p>2</p> </td><td> <p>Conv_1</p> <p>32
  3x3 convolution with stride [1 1]</p> </td><td> <p>Convolution</p> </td><td> <p>227x227x32</p> </td><td> <p>Weights
  3x3x1x32</p> <p>Bias
  1x1x32</p> </td></tr><tr><td> <p>3</p> </td><td> <p>Leaky
  ReLU_1</p> <p>Leaky
  ReLU with a scale of 0.01</p> </td><td> <p>Leaky
  ReLU</p> </td><td> <p>227x227x32</p> </td><td> <p>-</p> </td></tr><tr><td> <p>4</p> </td><td> <p>Maxpool_1</p> <p>5x5
  max pooling with stride [1 1] and padding the same</p> </td><td> <p>Max
  Pooling</p> </td><td> <p>227x227x32</p> </td><td> <p>-</p> </td></tr><tr><td> <p>5</p> </td><td> <p>Conv_2</p> <p>32
  3x3 convolution with stride [1 1]</p> </td><td> <p>Convolution</p> </td><td> <p>227x227x32</p> </td><td> <p>Weights
  3x3x1x32</p> <p>Bias
  1x1x32</p> </td></tr><tr><td> <p>6</p> </td><td> <p>Leaky
  ReLu_2</p> <p>Leaky
  ReLU with a scale of 0.01</p> </td><td> <p>Leaky
  ReLU</p> </td><td> <p>227x227x32</p> </td><td> <p>-</p> </td></tr><tr><td> <p>7</p> </td><td> <p>Avgpool2d_1</p> <p>5x5
  average pooling with stride [1 1] and padding the same</p> </td><td> <p>Average
  Pooling</p> </td><td> <p>227x227x32</p> </td><td> <p>-</p> </td></tr><tr><td> <p>8</p> </td><td> <p>Conv_3</p> <p>32
  3x3 convolution with stride [1 1]</p> </td><td> <p>Convolution</p> </td><td> <p>227x227x32</p> </td><td> <p>Weights
  3x3x1x32</p> <p>Bias
  1x1x32</p> </td></tr><tr><td> <p>9</p> </td><td> <p>Relu_1</p> <p>ReLU</p> </td><td> <p>ReLU</p> </td><td> <p>227x227x32</p> </td><td> <p>-</p> </td></tr><tr><td> <p>10</p> </td><td> <p>Conv_4</p> <p>32
  3x3 convolution with stride [1 1]</p> </td><td> <p>Convolution</p> </td><td> <p>227x227x32</p> </td><td> <p>Weights
  3x3x1x32</p> <p>Bias
  1x1x32</p> </td></tr><tr><td> <p>11</p> </td><td> <p>Leakyrelu_3</p> <p>Leaky
  ReLU with scale of 0.01</p> </td><td> <p>Leaky
  ReLU</p> </td><td> <p>227x227x32</p> </td><td> <p>-</p> </td></tr><tr><td> <p>12</p> </td><td> <p>Maxpool_2</p> <p>5x5
  max pooling with stride [1 1] and padding the same</p> </td><td> <p>Max
  Pooling</p> </td><td> <p>227x227x32</p> </td><td> <p>-</p> </td></tr><tr><td> <p>13</p> </td><td> <p>FC_1</p> <p>10
  fully connected layer</p> </td><td> <p>Fully
  Connected</p> </td><td> <p>1x1x10</p> </td><td> <p>Weights
  10x1648928</p> <p>Bias
  1x1x32</p> </td></tr><tr><td> <p>14</p> </td><td> <p>Leakyrelu_4</p> <p>Leaky
  ReLU with a scale of 0.01</p> </td><td> <p>Leaky
  ReLU</p> </td><td> <p>227x227x32</p> </td><td> <p>-</p> </td></tr><tr><td> <p>15</p> </td><td> <p>FC_2</p> <p>10
  fully connected layer</p> </td><td> <p>Fully
  Connected</p> </td><td> <p>1x1x10</p> </td><td> <p>Weights
  10x1648928</p> <p>Bias
  1x1x32</p> </td></tr><tr><td> <p>16</p> </td><td> <p>Relu_2</p> <p>ReLU</p> </td><td> <p>ReLU</p> </td><td> <p>227x227x32</p> </td><td> <p>-</p> </td></tr><tr><td> <p>17</p> </td><td> <p>Conv_5</p> <p>32
  3x3 convolution with stride [1 1]</p> </td><td> <p>Convolution</p> </td><td> <p>227x227x32</p> </td><td> <p>Weights
  3x3x1x32</p> <p>Bias
  1x1x32</p> </td></tr><tr><td> <p>18</p> </td><td> <p>Avgpool2d_2</p> <p>5x5
  average pooling with stride [1 1] and padding the same</p> </td><td> <p>Average
  Pooling</p> </td><td> <p>227x227x32</p> </td><td> <p>-</p> </td></tr><tr><td> <p>19</p> </td><td> <p>FC_3</p> <p>10
  fully connected layer</p> </td><td> <p>Fully
  Connected</p> </td><td> <p>1x1x10</p> </td><td> <p>Weights
  10x1648928</p> <p>Bias
  1x1x32</p> </td></tr><tr><td> <p>20</p> </td><td> <p>Softmax</p> <p>Softmax</p> </td><td> <p>Softmax</p> </td><td> <p>1x1x10</p> </td><td> <p>-</p> </td></tr><tr><td> <p>21</p> </td><td> <p>Classoutput</p> <p>Classentropyex</p> </td><td> <p>Classification
  Output</p> </td><td> <p>1x1x10</p> </td><td> <p>-</p> </td></tr></tbody></table></table-wrap> The above table depicts the details of the architecture of layers used for our CNN model designed for the classification First of all the entry point of our CNN model is the image input as we converted our audio files into image files that can be easily compatible and supported by the proposed network model.  32 filters of size 3x3 were applied by the first convolution layer named Conv_1 on an input image. To enhance the competence and ability of the network to apprehend the complex patterns Leaky ReLU activation function was employed with a scale of 0.01. For maintaining the latitudinal proportions, a 5x5 max pooling layer was then performed through a stride of [1,1]. Likewise, Conv_1 another layer of 32 filters with 3x3 size was applied named Conv_2, by utilizing Leaky ReLU activation. To down sample the feature maps 5x5 average pooling was performed with stride [1,1] and padding Furthermore, a fully connected layer was introduced comprising 10 output nodes involving a substantial number of learnable parameters. In the end, the softmax function was applied to generate probability distribution across the classes and a classification output layer was utilized that performed a precise entropy-based loss function for training.
</sec>
<sec id="sec-17">
  <title>Results and Experiments:</title>
<p>Experimental Set-Up</p><p>This section covers the details of the results which are carried out in the research. For the conduction of results, MATLAB software is used for the evaluation of the ASV Spoof 2019 dataset. The system used for research comprises of following specifications as shown in Table 3 below.</p>
</sec>
<sec id="sec-18">
  <title>Table 3</title>
<table-wrap id="table3"><label>Table 3</label><caption><title>Table 3</title></caption><table><tbody><tr><td colspan="3"> <p><bold>Hardware</bold></p> </td></tr><tr><td colspan="3" valign="top"> <p>§  Laptop Specification:</p> </td></tr><tr><td> <p>§  Processor</p> </td><td> <p>:</p> </td><td> <p>Core i7</p> </td></tr><tr><td> <p>§  RAM</p> </td><td> <p>:</p> </td><td> <p>16 GB</p> </td></tr><tr><td> <p>§  SSD</p> </td><td> <p>:</p> </td><td> <p>256 GB</p> </td></tr><tr><td> <p>§  Hard Drive</p> </td><td> <p>:</p> </td><td> <p>1 TB</p> </td></tr><tr><td> <p>§  GPU</p> </td><td> <p>:</p> </td><td> <p>2 GB</p> </td></tr><tr><td> <p>§  LCD</p> </td><td> <p>:</p> </td><td> <p>14´´</p> </td></tr><tr><td colspan="3" valign="top"> <p>§  Software:</p> </td></tr><tr><td colspan="3" valign="top"> <p>§  A tool which is used for achieving the
  results and analysis of the dataset is MATLAB.</p> </td></tr></tbody></table></table-wrap> Accuracy shows that a model is correct up to which level. When you want to find a specific class, accuracy tells how better this class is predicted. Mathematically,Accuracy=(True Positive + True Negative)/(True Positive + True Negative + False Positive + False Negative)		(8)Precision is basically the ratio of the predictions. Precision tells the ratio of estimates classified as confident which are appropriately classified to the number of estimates classified as confident whether they may be accurate or improper. Mathematically,Precision=(True Positive )/(True Positive + False Positive )	(9)Recall is also the ratio of the predictions. Recall displays the relation between the number of confident examples that are appropriately projected and classified as confident to the total number of confident examples.Mathematically,Recall=(True Positive )/(True Positive + False Negative )		(10)<break/>DatasetASV spoof 2019 dataset is trailed from the previous three sessions that were held during inter-speech 2013, 2015 and 2017. The first edition 2013 [Delgado2021] mainly concerned and targeted awareness about the spoofing threats. In the second edition of 2015 main target was to discriminate between real and fake speech by text-to-speech or voice alteration systems. Likewise, the next edition 2017 primarily targeted the detection of spoof attacks. The latest edition of ASV spoof 2019 [Lorenzo2018]is the best at the moment for the ASV spoof 2015 dataset consists of  TTS and VC-generated spoofing occurrences. Since remarkable progress occurred in the field of artificial intelligence, nowadays the tools and techniques which are used for the VC and TTS are so strongly developed that it is quite hard for someone to discriminate between false and genuine speech. Furthermore, as these tools provide much genuineness in speech, therefore, these threats are alarming and need to be addressed as soon as possible. Genuine speech was taken from 107 talkers among them 46 men and 61 women, having no noise effect in it. Various algorithms are utilized to create spoofed speeches from the genuine speech taken from 107 speakers. The dataset is segregated as follows:
</sec>
<sec id="sec-19">
  <title>Table 4</title>
<table-wrap id="table4"><label>Table 4</label><caption><title>Table 4</title></caption><table><thead><tr><th> <p><bold>Logical
   Access</bold></p> </th><th> <p><bold>Samples</bold></p> </th><th> <p><bold>Spoofing
   System</bold></p> </th><th> <p><bold>Algorithm
   Type</bold></p> </th><th> <p><bold>Data</bold></p> </th></tr></thead><tbody><tr><td valign="bottom"> <p>Total Samples</p> </td><td> <p>25380</p> </td><td> <p>_</p> </td><td> <p>_</p> </td><td> <p>_</p> </td></tr><tr><td> <p>A01</p> </td><td> <p>3800</p> </td><td> <p>TTS</p> </td><td> <p>neural waveform model</p> </td><td valign="bottom"> <p>Text</p> </td></tr><tr><td> <p>A02</p> </td><td> <p>3800</p> </td><td> <p>TTS</p> </td><td valign="bottom"> <p>Vocoder</p> </td><td valign="bottom"> <p>Text</p> </td></tr><tr><td> <p>A03</p> </td><td> <p>3800</p> </td><td> <p>TTS</p> </td><td valign="bottom"> <p>Vocoder</p> </td><td valign="bottom"> <p>Text</p> </td></tr><tr><td> <p>A04</p> </td><td> <p>3800</p> </td><td> <p>TTS</p> </td><td> <p>waveform concatenation</p> </td><td valign="bottom"> <p>Text</p> </td></tr><tr><td> <p>A05</p> </td><td> <p>3800</p> </td><td> <p>VC</p> </td><td valign="bottom"> <p>Vocoder</p> </td><td valign="bottom"> <p>Speech</p> </td></tr><tr><td> <p>A06</p> </td><td> <p>3800</p> </td><td> <p>VC</p> </td><td valign="bottom"> <p>spectral filtering</p> </td><td valign="bottom"> <p>Speech</p> </td></tr><tr><td> <p>A07</p> </td><td> <p>3800</p> </td><td> <p>TTS</p> </td><td valign="bottom"> <p>vocoder+GAN</p> </td><td valign="bottom"> <p>Text</p> </td></tr><tr><td> <p>A08</p> </td><td> <p>3800</p> </td><td> <p>TTS</p> </td><td valign="bottom"> <p>neural waveform</p> </td><td valign="bottom"> <p>Text</p> </td></tr><tr><td> <p>A09</p> </td><td> <p>3800</p> </td><td> <p>TTS</p> </td><td valign="bottom"> <p>Vocoder</p> </td><td valign="bottom"> <p>Text</p> </td></tr><tr><td> <p>A10</p> </td><td> <p>3800</p> </td><td> <p>TTS</p> </td><td valign="bottom"> <p>neural waveform</p> </td><td valign="bottom"> <p>Text</p> </td></tr><tr><td> <p>A11</p> </td><td> <p>3800</p> </td><td> <p>TTS</p> </td><td valign="bottom"> <p>griffin lim</p> </td><td valign="bottom"> <p>Text</p> </td></tr><tr><td> <p>A12</p> </td><td> <p>3800</p> </td><td> <p>TTS</p> </td><td valign="bottom"> <p>neural waveform</p> </td><td valign="bottom"> <p>Text</p> </td></tr><tr><td> <p>A13</p> </td><td> <p>3800</p> </td><td> <p>TTS_VC</p> </td><td valign="bottom"> <p>Waveform concatenation,
  waveform filtering</p> </td><td> <p>speech</p> </td></tr><tr><td> <p>A14</p> </td><td> <p>3800</p> </td><td> <p>TTS_VC</p> </td><td valign="bottom"> <p>Vocoder</p> </td><td valign="bottom"> <p>speech</p> </td></tr><tr><td> <p>A15</p> </td><td> <p>3800</p> </td><td> <p>TTS_VC</p> </td><td valign="bottom"> <p>neural waveform</p> </td><td valign="bottom"> <p>speech</p> </td></tr><tr><td> <p>A16</p> </td><td> <p>3800</p> </td><td> <p>TTS</p> </td><td valign="bottom"> <p>waveform concatenation</p> </td><td valign="bottom"> <p>text</p> </td></tr><tr><td> <p>A17</p> </td><td> <p>3800</p> </td><td> <p>VC</p> </td><td valign="bottom"> <p>waveform filtering</p> </td><td valign="bottom"> <p>speech</p> </td></tr><tr><td> <p>A18</p> </td><td> <p>3800</p> </td><td> <p>VC</p> </td><td valign="bottom"> <p>Vocoder</p> </td><td valign="bottom"> <p>speech</p> </td></tr><tr><td> <p>A19</p> </td><td> <p>3800</p> </td><td> <p>VC</p> </td><td valign="bottom"> <p>spectral filtering</p> </td><td valign="bottom"> <p>speech</p> </td></tr></tbody></table></table-wrap> Spoofed and genuine speeches were collected from the 20 speakers for the training of the dataset. For this purpose, some of the algorithms are used for speech conversion and some are used for speech synthesis as shown in table 5 below.
</sec>
<sec id="sec-20">
  <title>Table 5</title>
<table-wrap id="table5"><label>Table 5</label><caption><title>Table 5</title></caption><table><thead><tr><th> <p><bold>Task</bold></p> </th><th> <p><bold>Technique Used</bold></p> </th></tr></thead><tbody><tr><td rowspan="2"> <p>Voice Conversion</p> </td><td valign="top"> <p>§  Neural Network</p> </td></tr><tr><td valign="top"> <p>§  Transfer Function</p> </td></tr><tr><td rowspan="4"> <p>Speech Synthesis</p> </td><td valign="top"> <p>§  Neural network-based parametric
  speech synthesis using source filter vocoders</p> </td></tr><tr><td valign="top"> <p>§  Neural network based</p> </td></tr><tr><td valign="top"> <p>§  parametric speech synthesis using
  Wavenet</p> </td></tr><tr><td valign="top"> <p>§  Waveform concatenation</p> </td></tr></tbody></table></table-wrap>
</sec>
<sec id="sec-21">
  <title>Table 5</title>
<table-wrap id="table6"><label>Table 6</label><caption><title>Table 6</title></caption><table><tbody><tr><td colspan="10"> <p><bold>PA Samples</bold></p> </td></tr><tr><td rowspan="2">  </td><td rowspan="2"> <p><bold>Sample</bold></p> </td><td colspan="4"> <p><bold>Label</bold></p> </td><td rowspan="2"> <p><bold>Type of Attack</bold></p> </td><td colspan="2"> <p><bold>Labels</bold></p> </td><td rowspan="2"> <p><bold>Replay Device Quality</bold></p> </td></tr><tr><td>  </td><td> <p><bold>a</bold></p> </td><td> <p><bold>b</bold></p> </td><td> <p><bold>c</bold></p> </td><td> <p><bold>a</bold></p> </td><td> <p><bold>b</bold></p> </td></tr><tr><td> <p>Training</p> </td><td> <p>54,000</p> </td><td> <p>Room Size (m<sup>2</sup>)</p> </td><td> <p>2-5</p> </td><td> <p>5-10</p> </td><td> <p>10-20</p> </td><td> <p>Attacker to Talker Distance (cm)</p> </td><td> <p>10-50</p> </td><td> <p>Perfect</p> </td><td> <p>Perfect</p> </td></tr><tr><td> <p>Dev</p> </td><td> <p>33,534</p> </td><td> <p>T60 (ms)</p> </td><td> <p>50-200</p> </td><td> <p>200-600</p> </td><td> <p>600-1000</p> </td><td> <p>Attacker to Talker Distance (cm)</p> </td><td> <p>50-100</p> </td><td> <p>High</p> </td><td> <p>High</p> </td></tr><tr><td> <p>Eval</p> </td><td> <p>153,522</p> </td><td> <p>Talker to ASV Distance (cm)</p> </td><td> <p>10-50</p> </td><td> <p>50-100</p> </td><td> <p>100-150</p> </td><td> <p>Attacker to Talker Distance (cm)</p> </td><td> <p>&gt;100</p> </td><td> <p>Low</p> </td><td> <p>Low</p> </td></tr><tr><td colspan="10"> <p><bold>LA Samples</bold></p> </td></tr><tr><td>  </td><td>  </td><td colspan="2"> <p>Spoof System</p> </td><td> <p>Input</p> </td><td colspan="2"> <p>Mechanism of Input</p> </td><td colspan="3"> <p>Generator for Sound Wave</p> </td></tr><tr><td> <p>Training</p> </td><td> <p>25,380</p> </td><td colspan="2" rowspan="3"> <p>A01, A02, A03, A04, A05, A06, A07, A08, A09, A10,
  A11, A12, A13, A14, A15, A16, A17, A18, A19</p> </td><td rowspan="3"> <p>Text Human Speech</p> </td><td colspan="2" rowspan="3"> <p>TTS, RNN, NLP, WORLD, MFCC, Spectral filtering
  ASR, Conv and Bi Waveform Conc</p> </td><td colspan="3" rowspan="3"> <p>Wave Net*,
  WORLD, Waveform Concatenation,</p> <p>OLA and
  Special Filters, Griffin-Lim, STRAIGHT, MFCC Vocoder.</p> </td></tr><tr><td> <p>Dev</p> </td><td> <p>24,986</p> </td></tr><tr><td> <p>Eval</p> </td><td> <p>71,933</p> </td></tr></tbody></table></table-wrap> The main objective of ASV spoof creativity was to guard the programmed talker authentication from deceiving attacks [Kinnunen2017][27][28][29][Gomez2017] The below table highlights the LA and PA datasets for ASV spoof 2019.
</sec>
<sec id="sec-22">
  <title>Data for Experiments</title>
<p>The experimental approach for the research is</p><p>constructed on training and testing of the LA dataset. LA dataset comprised 25,380 samples including both genuine and fake samples. Among these samples 2,590 samples are bonafide and 22,900 samples are spoofed. This research study falls under the interdisciplinary approach of a study converging data science, management science and linguistics. Audios contain linguistic data in the arrangement of verbal language. The present study has discovered automatic spoofing detection in the linguistic data of audio through deep learning.</p><p>During the training of the model, different limitations for the model are to be developed. Initially, the model was accomplished using a learning rate of 0.01 and after that, we also trained the model at the learning rate of 0.0001. In both cases of using different learning rates the accuracy of the trained model remained unaltered, which means the learning rate did change the model’s accuracy. The limitations of the trained ML-DL SafetyNet model are as under:</p>
</sec>
<sec id="sec-23">
  <title>Table 6</title>
<table-wrap id="table7"><label>Table 7</label><caption><title>Table 7</title></caption><table><tbody><tr><td> <p><bold>Limitations</bold></p> </td><td> <p><bold>Value</bold></p> </td></tr><tr><td> <p>Learning Rate</p> </td><td> <p>0.01
  / 0.0001</p> </td></tr><tr><td> <p>Batch size</p> </td><td> <p>128</p> </td></tr><tr><td> <p>Confidence value</p> </td><td> <p>0.20</p> </td></tr><tr><td> <p>No. of epochs</p> </td><td> <p>30</p> </td></tr></tbody></table></table-wrap>
</sec>
<sec id="sec-24">
  <title>Figure 9</title>
<p><break/></p>
</sec>
<fig id="fig-8"><alt-text>Figure 9</alt-text><caption><title>Figure 9</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 8.jpg"/></fig>
<sec id="sec-25">
  <title>Figure 10</title>
<p><break/></p>
</sec>
<fig id="fig-9"><alt-text>Figure 10</alt-text><caption><title>Figure 10</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 9.jpg"/></fig>
<sec id="sec-26">
  <title>Evaluation of Logical Access Violation’s Performance</title>
<p>The main objective of the ML-DL SafetyNet model is to evaluate the effectiveness of the proposed model against the LA attacks. We have used classification learner algorithms by using machine learning and deep learning to differentiate between bonafide and spoof speeches. The table below indicates an equal minimum classification error and AUC in our proposed spoofing detection model. ML-DL SafetyNet model achieved different accuracies for different classification algorithms. Likewise, by using the deep learning ML-DL Safety Net model successfully extracted the features from the Mel-spectrograms. As far as our outcomes it is understood that the ML-DL SafetyNet model achieved better results.</p>
</sec>
<sec id="sec-27">
  <title>Table 8</title>
<table-wrap id="table8"><label>Table 8</label><caption><title>Table 8</title></caption><table><tbody><tr><td> <p><bold>Dataset</bold></p> </td><td> <p><bold>Accuracy%</bold></p> </td><td> <p><bold>Recall%</bold></p> </td><td> <p><bold>Precision%</bold></p> </td><td> <p><bold>Min-tDCF</bold></p> </td></tr><tr><td> <p>Eval Set</p> </td><td> <p>98.3</p> </td><td> <p>99.1</p> </td><td> <p>98.1</p> </td><td> <p>0.003</p> </td></tr><tr><td> <p>Dev Set</p> </td><td> <p>98.8</p> </td><td> <p>98.6</p> </td><td> <p>97.9</p> </td><td> <p>0.007</p> </td></tr></tbody></table></table-wrap>
</sec>
<sec id="sec-28">
  <title>Figure 11</title>
<p><break/></p>
</sec>
<fig id="fig-10"><alt-text>Figure 11</alt-text><caption><title>Figure 11</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 10.jpg"/></fig>
<sec id="sec-29">
  <title>Performance Comparison of Physical Access Attack</title>
<p>The foremost objective of the research is to dig out the usefulness of spoofing recognition for physical access attacks.  For such a reason audio samples are taken from the PA set and their Mel spectrograms are created after the spectrogram’s generation, these spectrograms are separated based on real and fake classes. The results indicate an EER of 0.62% and 3.4% for eval and dev sets respectively. Similarly, results indicate min-tDCF of 0.04 and 0.09 for eval and dev sets respectively. As far as accuracies are concerned, we have accomplished better results from the earlier research models as we have achieved 99.5% and 97.4% accuracies for eval and dev sets. Performance plots and results are shown below.</p>
</sec>
<sec id="sec-30">
  <title>Table 9</title>
<table-wrap id="table9"><label>Table 9</label><caption><title>Table 9</title></caption><table><tbody><tr><td valign="top"> <p><bold>Dataset</bold></p> </td><td> <p><bold>Accuracy%</bold></p> </td><td> <p><bold>EER%</bold></p> </td><td> <p><bold>Recall%</bold></p> </td><td> <p><bold>Precision%</bold></p> </td><td> <p><bold>Min-tDCF</bold></p> </td></tr><tr><td valign="top"> <p>Eval Set</p> </td><td> <p>99.5</p> </td><td> <p>0.62</p> </td><td> <p>95.84</p> </td><td> <p>99.2</p> </td><td> <p>0.04</p> </td></tr><tr><td valign="top"> <p>Dev Set</p> </td><td> <p>97.4</p> </td><td> <p>3.4</p> </td><td> <p>99.34</p> </td><td> <p>98.6</p> </td><td> <p>0.09</p> </td></tr></tbody></table></table-wrap>
</sec>
<sec id="sec-31">
  <title>Figure 12</title>
<p><break/></p>
</sec>
<fig id="fig-11"><alt-text>Figure 12</alt-text><caption><title>Figure 12</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 11.jpg"/></fig>
<sec id="sec-32">
  <title>Comparison of Various Machine Learning Classifiers</title>
<p>Our research is focused on audio spoofing detection which is more hazardous than video deep fakes because most of our communication is based on audio like audio phone calls, voice recordings etc. For such a reason, it is the essential need of moment to recognize fake audios. We calculated the results of our problem by using deep learning and machine learning techniques. By using deep learning, we performed the 10-fold cross-validation and used different learning rates, but the output did not change by changing the learning rate. We achieved an accuracy of 90% by using a deep learning app. Similarly, we used different classification learners on the same data to achieve the results. We performed seven classifiers namely KNN, SVM, fine tree, naïve Bayes, logistic regression, linear discriminant and optimizable discriminant. Among these classifiers support vector machine (SVM) performed well with an accuracy of 90%. The results of the different classifiers are as under:</p>
</sec>
<sec id="sec-33">
  <title>Table 10</title>
<table-wrap id="table10"><label>Table 10</label><caption><title>Table 10</title></caption><table><thead><tr><th> <p><bold>S. No</bold></p> </th><th valign="top"> <p><bold>Algorithm Name</bold></p> </th><th> <p><bold>Accuracy
   (%)</bold></p> </th></tr></thead><tbody><tr><td> <p>1</p> </td><td valign="top"> <p>Optimizable Tree</p> </td><td> <p>88</p> </td></tr><tr><td> <p>2</p> </td><td valign="top"> <p>Gaussian Naïve Bayes</p> </td><td> <p>90</p> </td></tr><tr><td> <p>3</p> </td><td valign="top"> <p>K-Nearest Neighbor (KNN)</p> </td><td> <p>90</p> </td></tr><tr><td> <p>4</p> </td><td valign="top"> <p>Logistic Regression</p> </td><td> <p>79</p> </td></tr><tr><td> <p>5</p> </td><td valign="top"> <p>Linear Discriminant</p> </td><td> <p>84</p> </td></tr><tr><td> <p>6</p> </td><td valign="top"> <p>Support Vector Machine (SVM)</p> </td><td> <p>90</p> </td></tr><tr><td> <p>7</p> </td><td valign="top"> <p>Optimizable Discriminant</p> </td><td> <p>89</p> </td></tr></tbody></table></table-wrap>
</sec>
<sec id="sec-34">
  <title>Figure 13</title>
<p><break/></p>
</sec>
<fig id="fig-12"><alt-text>Figure 13</alt-text><caption><title>Figure 13</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 12.jpg"/></fig>
<sec id="sec-35">
  <title>Recognition of Voice Copying Algorithms</title>
<p>The principal ambition of the research is to figure out which technique and algorithm applied is to be best utilized in future for such types of problems. Basically, six different kinds of algorithms were used in the ASV spoof 2019 dataset as per the LA group is concerned (i.e. A01 to A06)</p><p>In the experimental planning 25,380 samples were used for the determination of model training and to check whether the model is best and how the model will behave for the samples used. 24,986 samples were collected for the development set to test the model. The algorithms (A01 to A06) used in this experimentation accomplished several enhanced entity relationships (i.e. 0.5, 2.0, 1.09, 1.2, 1.02 and 1.6 % respectively).</p><p><break/></p><p>Artificial Voice and Voice Conversion Performance Review</p><p>This experimental setup is focused on the investigation of the results that spoofing is well identified for which of the techniques either text-to-speech or the voice cloning technique. We have implemented the Mel-spectrograms with various different characteristics for the training of our dataset samples including TTS and VC samples.  Experimentation comprises the various Voice Cloning (VC) systems including A05 and A06 for the generation of the spoofing trials. Similarly, experimentation also includes text-to-speech (TTS) systems (A01 to A04) for the generation of various spoofing trials for our model. Both these systems’ main objective was to generate the spoof trials for the LA dataset for our model to be used for training purposes. Moreover, the LA dataset includes the evaluation set which contains other different 13 algorithms (i.e. A07 to A12, A16 TTS spoofing systems, A17 to A19 VC spoofing systems and A13 to A15 VC-TTS spoofing systems). Experimental results are shown below.</p>
</sec>
<sec id="sec-36">
  <title>Table 11</title>
<table-wrap id="table11"><label>Table 11</label><caption><title>Table 11</title></caption><table><tbody><tr><td> <p><bold>Algorithm</bold></p> </td><td> <p><bold>Accuracy (%)</bold></p> </td><td> <p><bold>Precision (%)</bold></p> </td><td> <p><bold>Recall (%)</bold></p> </td></tr><tr><td> <p>A01</p> </td><td> <p>99.2</p> </td><td> <p>96.92</p> </td><td> <p>98.3</p> </td></tr><tr><td> <p>A02</p> </td><td> <p>99.7</p> </td><td> <p>99.37</p> </td><td> <p>93.2</p> </td></tr><tr><td> <p>A03</p> </td><td> <p>94.1</p> </td><td> <p>94.35</p> </td><td> <p>98.9</p> </td></tr><tr><td> <p>A04</p> </td><td> <p>94.4</p> </td><td> <p>94.81</p> </td><td> <p>99.8</p> </td></tr><tr><td> <p>A05</p> </td><td> <p>98.4</p> </td><td> <p>95.98</p> </td><td> <p>98.7</p> </td></tr><tr><td> <p>A06</p> </td><td> <p>95.9</p> </td><td> <p>96.11</p> </td><td> <p>98.2</p> </td></tr></tbody></table></table-wrap>
</sec>
<sec id="sec-37">
  <title>Figure 14</title>
<p><break/></p>
</sec>
<fig id="fig-13"><alt-text>Figure 14</alt-text><caption><title>Figure 14</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 13.jpg"/></fig>
<sec id="sec-38">
<p>TTS spoof trials generated and real from the ASVspoof 2019 dataset were used for the training of the model and at the same time the trials from the evaluation set of TTS were used to test the proposed model. Enhanced entity-relationship (EER) achieved from the model is 0.63% and min-tDCF is 0.0159.</p>
</sec>
<sec id="sec-39">
  <title>Table 12</title>
<table-wrap id="table12"><label>Table 12</label><caption><title>Table 12</title></caption><table><tbody><tr><td valign="top"> <p><bold>Spoofing Category</bold></p> </td><td> <p><bold>Min-tDCF</bold></p> </td><td> <p><bold>EER (%)</bold></p> </td><td> <p><bold>Accuracy (%)</bold></p> </td><td> <p><bold>Precision (%)</bold></p> </td><td> <p><bold>Recall (%)</bold></p> </td></tr><tr><td valign="top"> <p>Voice Cloning (VC)</p> </td><td> <p>0.39</p> </td><td> <p>18.6</p> </td><td> <p>83.10</p> </td><td> <p>98.21</p> </td><td> <p>76.95</p> </td></tr><tr><td valign="top"> <p>Text to Speech (TTS)</p> </td><td> <p>0.04</p> </td><td> <p>0.49</p> </td><td> <p>98.99</p> </td><td> <p>99.33</p> </td><td> <p>98.96</p> </td></tr><tr><td valign="top"> <p>Overall LA</p> </td><td> <p>0.03</p> </td><td> <p>0.08</p> </td><td> <p>99.45</p> </td><td> <p>99.41</p> </td><td> <p>99.29</p> </td></tr></tbody></table></table-wrap>
</sec>
<sec id="sec-40">
  <title>Disparity of Performance to Prevailing Techniques</title>
<p>The experiment compares the speech spoofing detector to other voice spoofing detection techniques. We conducted a comparison analysis with the models listed below in the table to demonstrate the viability of the ML-DL SafetyNetmodel, which is an improved   ANN-based classifier for good detection of flaws in playback samples, cloning algorithms, art facts and authentic samples’ dynamic speech qualities depending on their vocal tracts. The proposed and current approaches&apos; performance in the context of EER and min-tCDF results on PA and LA datasets of ASV spoof 2019 are shown.</p>
</sec>
<sec id="sec-41">
  <title>Table 13</title>
<table-wrap id="table13"><label>Table 13</label><caption><title>Table 13</title></caption><table><thead><tr><th rowspan="2"> <p><bold>System</bold></p> </th><th colspan="2"> <p><bold>LA Evaluation Set</bold></p> </th><th colspan="2"> <p><bold>PA Evaluation Set</bold></p> </th></tr><tr><th> <p><bold>% of EER</bold></p> </th><th> <p><bold>Min-tDCF</bold></p> </th><th> <p><bold>% of EER</bold></p> </th><th> <p><bold>Min-tDCF</bold></p> </th></tr></thead><tbody><tr><td valign="top"> <p>Baseline:
  GMM  [31]</p> </td><td> <p>2.71</p> </td><td> <p>0.0663</p> </td><td> <p>8.09</p> </td><td> <p>0.2116</p> </td></tr><tr><td valign="top"> <p>MFCC
  [31]</p> </td><td> <p>16.80</p> </td><td> <p>0.3945</p> </td><td> <p>-</p> </td><td> <p>-</p> </td></tr><tr><td valign="top"> <p>CQCC
  [31]</p> </td><td> <p>8.82</p> </td><td> <p>0.2076</p> </td><td> <p>12.06</p> </td><td> <p>0.2982</p> </td></tr><tr><td valign="top"> <p>Baseline:
  GMM [32]</p> </td><td> <p>0.43</p> </td><td> <p>0.0123</p> </td><td> <p>9.57</p> </td><td> <p>0.2366</p> </td></tr><tr><td valign="top"> <p>LFCC
  [32]</p> </td><td> <p>2.71</p> </td><td> <p>0.0663</p> </td><td> <p>8.09</p> </td><td> <p>0.2116</p> </td></tr><tr><td valign="top"> <p>CQCC
  [32]</p> </td><td> <p>0.94</p> </td><td> <p>0.08</p> </td><td> <p>0.43</p> </td><td> <p>0.03</p> </td></tr><tr><td valign="top"> <p>Baseline:
  GMM  [33]</p> </td><td> <p>2.64</p> </td><td> <p>0.0755</p> </td><td> <p>5.43</p> </td><td> <p>0.1465</p> </td></tr><tr><td valign="top"> <p>LFCC
  [33]</p> </td><td> <p>8.09</p> </td><td> <p>0.2116</p> </td><td> <p>13.54</p> </td><td> <p>0.3017</p> </td></tr><tr><td valign="top"> <p>CQCC
  [33]</p> </td><td> <p>9.57</p> </td><td> <p>0.2366</p> </td><td> <p>11.04</p> </td><td> <p>0.2454</p> </td></tr><tr><td valign="top"> <p>Baseline:
  GMM [34]</p> </td><td> <p>10.62</p> </td><td> <p>0.2401</p> </td><td> <p>5.58</p> </td><td> <p>0.1518</p> </td></tr><tr><td valign="top"> <p>LFCC
  [34]</p> </td><td> <p>0.28</p> </td><td> <p>0.0062</p> </td><td> <p>4.79</p> </td><td> <p>0.1314</p> </td></tr><tr><td valign="top"> <p>CQCC
  [34]</p> </td><td> <p>0.43</p> </td><td> <p>0.0123</p> </td><td> <p>9.87</p> </td><td> <p>0.1953</p> </td></tr><tr><td valign="top"> <p>Baseline:
  GMM  [35]</p> </td><td> <p>-</p> </td><td> <p>-</p> </td><td> <p>-</p> </td><td> <p>-</p> </td></tr><tr><td valign="top"> <p>LFCC
  [35]</p> </td><td> <p>11.04</p> </td><td> <p>0.2454</p> </td><td> <p>0.43</p> </td><td> <p>0.0123</p> </td></tr><tr><td valign="top"> <p>CQCC
  [35]</p> </td><td> <p>9.87</p> </td><td> <p>0.1953</p> </td><td> <p>9.57</p> </td><td> <p>0.2366</p> </td></tr><tr><td valign="top"> <p>Baseline:
  GMM [36]</p> </td><td> <p>5.06</p> </td><td> <p>0.1562</p> </td><td> <p>-</p> </td><td> <p>-</p> </td></tr><tr><td valign="top"> <p>LFCC
  [36]</p> </td><td> <p>4.04</p> </td><td> <p>0.1655</p> </td><td> <p>-</p> </td><td> <p>-</p> </td></tr><tr><td valign="top"> <p>CQCC
  [36]</p> </td><td> <p>2.64</p> </td><td> <p>0.1331</p> </td><td> <p>-</p> </td><td> <p>-</p> </td></tr><tr><td valign="top"> <p>Baseline:
  Parallel DDWS [37]</p> </td><td> <p>2.63</p> </td><td> <p>-</p> </td><td> <p>0.47</p> </td><td> <p>-</p> </td></tr><tr><td valign="top"> <p>Sequential
  DDWS [37]</p> </td><td> <p>2.08</p> </td><td> <p>-</p> </td><td> <p>0.63</p> </td><td> <p>-</p> </td></tr><tr><td valign="top"> <p>BC
  Res-Max [37]</p> </td><td> <p>2.59</p> </td><td> <p>-</p> </td><td> <p>0.49</p> </td><td> <p>-</p> </td></tr><tr><td valign="top"> <p>CGCNN:
  VAE log-CQT + log CQT [38]</p> </td><td> <p>1.84</p> </td><td> <p>0.056</p> </td><td> <p>0.35</p> </td><td> <p>0.0092</p> </td></tr><tr><td valign="top"> <p>CGCNN:
  Phase + log CQT [38]</p> </td><td> <p>1.09</p> </td><td> <p>0.034</p> </td><td> <p>0.31</p> </td><td> <p>0.0078</p> </td></tr><tr><td valign="top"> <p>ResNet18:
  Phase + log CQT [38]</p> </td><td> <p>1.53</p> </td><td> <p>0.051</p> </td><td> <p>1.16</p> </td><td> <p>0.0350</p> </td></tr><tr><td valign="top"> <p>Baseline:
  CLS-LBP + LSTM  [39]</p> </td><td> <p>0.06</p> </td><td> <p>0.0017</p> </td><td> <p>0.58</p> </td><td> <p>0.0160</p> </td></tr><tr><td valign="top"> <p>CQCC
  [39]</p> </td><td> <p>1.18</p> </td><td> <p>0.0520</p> </td><td> <p>11.5</p> </td><td> <p>0.2457</p> </td></tr><tr><td valign="top"> <p>Ours:DLDet</p> </td><td> <p>0.052</p> </td><td> <p>0.0028</p> </td><td> <p>0.41</p> </td><td> <p>0.0243</p> </td></tr></tbody></table></table-wrap> Receiver Operating Curves (ROC)Receiver operating curves basically assist in designing two types of factors i.e. True Positive (TP) and False Positive (FP). ROC curves for all the classifiers are plotted. Basically, these curves help to express the performance of the designed classifier models. ROC curves along with the area under the curves (AUC) are also plotted in the same window. The plots are as under:
</sec>
<sec id="sec-42">
  <title>Figure 15</title>
<p><break/></p>
</sec>
<fig id="fig-14"><alt-text>Figure 15</alt-text><caption><title>Figure 15</title></caption><graphic xlink:href="https://gssrjournal.com/gGRM3eygEY/Figure 14.jpg"/></fig>
<sec id="sec-43">
  <title>Conclusion</title>
<p>The proposed ML-DL SafetyNet model is
structured into two sections, the first section powers deep learning
techniques, while the second section uses machine learning techniques. In the first
section of the ML-DL SafetyNet model, ASV spoof 2019 dataset audio files of
logical access (LA) are utilized and converted to image spectrograms that are
supported by MATLAB. Afterwards, the ML-DL SafetyNet model is trained using
different learning rates. The proposed model achieved an accuracy of about 90%
by using a deep learning approach. In the case of the second approach of ML-DL
SafetyNet feature extraction and feature selection are performed by using
various machine learning classifiers. Likewise, the same ASV spoof 2019 dataset
was used for machine learning classifiers as in the preceding section of deep
learning and performed the tasks over seven classifiers. Among all the
classifiers, the support vector machine (SVM) performed best with an accuracy
of 90%. Our relative analysis of the existing models proposes that our ML-DL
SafetyNet model outperforms in perceiving various sorts of speech spoofing,
including TTS, replay attacks and cloning-based attacks. It is noteworthy, that
our model established excellent results on ASV spoof 2019. We can conclude that
model ML-DL SafetyNet is a strong deceiving indicator, authenticated through
the usefulness in cross justification over the ASVspoof 2019 evaluation set. In
future, our ambition is to encompass cross-validation to different speech-deceiving
datasets and additionally enhance the model’s performance.</p>
</sec>
</body>
<back>
<fn-group content-type="conflict-of-interest">
  <title>Conflict of Interest</title>
  <fn fn-type="conflict">
<p>The authors declare that they have no conflicts of interest.</p>
  </fn>
</fn-group>
<fn-group content-type="ethics-statement">
  <title>Ethics Statement</title>
  <fn fn-type="ethics">
<p>This study did not require formal ethics approval.</p>
  </fn>
</fn-group>
<fn-group content-type="data-availability">
  <title>Data Availability</title>
  <fn fn-type="data-availability-statement">
<p>Data sharing is not applicable to this article.</p>
  </fn>
</fn-group>
<app-group>
  <app id="app-suppl">
    <title>Supplementary Materials</title>
<supplementary-material id="suppl-pdf" content-type="pdf" xlink:href="https://gssrjournal.com/pdf/gssr/gGRM3eygEY.pdf">
  <label>PDF</label>
  <caption>
    <title>Full Text PDF</title>
  </caption>
</supplementary-material>
  </app>
</app-group>
<ref-list>
  <title>References</title>
<ref id="Almutairi">
  <label>1</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Almutairi, Z.,&amp; H. Elgibreen</person-group>
    <year>2023</year>
    <source>â€œDetecting Fake Audio of Arabic Speakers Using Self-Supervised Deep Learning,â€ IEEE Access</source>
    Almutairi, Z.,&amp; H. Elgibreen(2023). â€œDetecting Fake Audio of Arabic Speakers Using Self-Supervised Deep Learning,â€ IEEE Access, 1,
    <pub-id pub-id-type="doi">10.1109/ACCESS.2023.3286864</pub-id>
    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/ACCESS.2023.3286864">10.1109/ACCESS.2023.3286864</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?q=Almutairi,+Z.,%26+H.+Elgibreen(2023).+%E2%80%9CDetecting+Fake+Audio+of+Arabic+Speakers+Using+Self-Supervised+Deep+Learning,%E2%80%9D+IEEE+Access,&amp;hl=en&amp;as_sdt=0,5">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Alzantot">
  <label>2</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Alzantot, M., Wang, Z.,&amp;B. Srivastava</person-group>
    <year>2019</year>
    <source>â€œDeep residual neural networks for audio spoofing detection,â€ in Proceedings of the Annual Conference of the International Speech Communication Association, Interspeech, 1078â€“ 1082</source>
    <page-range>roceedings</page-range>
    Alzantot, M., Wang, Z.,&amp;B. Srivastava, (2019)â€œDeep residual neural networks for audio spoofing detection,â€ in Proceedings of the Annual Conference of the International Speech Communication Association, Interspeech, 1078â€“ 1082.
    <pub-id pub-id-type="doi">10.21437/Interspeech.2019-3174</pub-id>
    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.21437/Interspeech.2019-3174">10.21437/Interspeech.2019-3174</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&amp;as_sdt=0%2C5&amp;q=Alzantot%2C+M.%2C+Wang%2C+Z.%2C%26B.+Srivastava%2C+%282019%29%E2%80%9CDeep+residual+neural+networks+for+audio+spoofing+detection%2C%E2%80%9D+in+Proceedings+of+the+Annual+Conference+of+the+International+Speech+Communication+Association%2C+Interspeech%2C+1078%E2%80%93+1082.&amp;btnG=">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Balamurali">
  <label>3</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Balamurali, B. T. Lin, K. E. S. Lui, J. M. Chen, &amp; D</person-group>
    <year>2019</year>
    <source>Herremans, â€œToward robust audio spoofing detection: A detailed comparison of traditional and learned features,â€ IEEE Access</source>
    Balamurali, B. T. Lin, K. E. S. Lui, J. M. Chen, &amp; D. (2019). Herremans, â€œToward robust audio spoofing detection: A detailed comparison of traditional and learned features,â€ IEEE Access, 7, 84229â€“84241,
    <pub-id pub-id-type="doi">10.1109/ACCESS.2019.2923806</pub-id>
    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/ACCESS.2019.2923806">10.1109/ACCESS.2019.2923806</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&amp;as_sdt=0%2C5&amp;q=Balamurali%2C+B.+T.+Lin%2C+K.+E.+S.+Lui%2C+J.+M.+Chen%2C+%26+D.+%282019%29.+Herremans%2C+%E2%80%9CToward+robust+audio+spoofing+detection%3A+A+detailed+comparison+of+traditional+and+learned+features%2C%E2%80%9D+IEEE+Access%2C+7%2C+84229%E2%80%9384241%2C&amp;btnG=">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://ieeexplore.ieee.org/document/8740863">Fulltext</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&amp;as_sdt=0%2C5&amp;q=Balamurali%2C+B.+T.+Lin%2C+K.+E.+S.+Lui%2C+J.+M.+Chen%2C+%26+D.+%282019%29.+Herremans%2C+%E2%80%9CToward+robust+audio+spoofing+detection%3A+A+detailed+comparison+of+traditional+and+learned+features%2C%E2%80%9D+IEEE+Access%2C+7%2C+84229%E2%80%9384241%2C&amp;btnG=">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://ieeexplore.ieee.org/document/8740863">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Columbia">
  <label>4</label>
  <mixed-citation publication-type="journal">
    <article-title>Columbia,B.( 2021)</article-title>
    <source>â€œA Capsule Network Based Approach for Detection of Audio Spoofing Attacks 1. Key Lab of Information Security, School of Computer Science and Engineering, Sun Yat-Sen University, 2. Alibaba Group, Hangzhou, China,â€ 6359â€“6363</source>
    Columbia,B.( 2021). â€œA Capsule Network Based Approach for Detection of Audio Spoofing Attacks 1. Key Lab of Information Security, School of Computer Science and Engineering, Sun Yat-Sen University, 2. Alibaba Group, Hangzhou, China,â€ 6359â€“6363,
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&amp;as_sdt=0%2C5&amp;q=Columbia%2CB.%282021%29.+%E2%80%9CA+Capsule+Network+Based+Approach+for+Detection+of+Audio+Spoofing+Attacks+1.+Key+Lab+of+Information+Security%2C+School+of+Computer+Science+and+Engineering%2C+Sun+Yat-Sen+University%2C+2.+Alibaba+Group%2C+Hangzhou%2C+China%2C%E2%80%9D+6359%E2%80%936363%2C&amp;btnG=">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Delgado">
  <label>5</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Delgado,H. et al</person-group>
    <year>2021</year>
    <source>â€œASVspoof</source>
    <page-range>lan</page-range>
    Delgado,H. et al., (2021). â€œASVspoof 2021: Automatic Speaker Verification Spoofing and Countermeasures Challenge Evaluation Plan,â€
    <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/2109.00535">http://arxiv.org/abs/2109.00535</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?q=Delgado,H.+et+al.,+(2021).+%E2%80%9CASVspoof+2021:+Automatic+Speaker+Verification+Spoofing+and+Countermeasures+Challenge+Evaluation+Plan,&amp;hl=en&amp;as_sdt=0,5">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Dua">
  <label>6</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Dua, M. C. Jain, &amp; Kumar, S</person-group>
    <year>2022</year>
    <article-title>. â€œLSTM and CNN based ensemble approach for spoof detection task in automatic speaker verification systems,â€ J</article-title>
    <source>Ambient Intell. Humaniz.Comput</source>
    <volume>13</volume>
    <issue>4</issue>
    <fpage>1985</fpage>
    Dua, M. C. Jain, &amp; Kumar, S.(2022). â€œLSTM and CNN based ensemble approach for spoof detection task in automatic speaker verification systems,â€ J. Ambient Intell. Humaniz.Comput, 13(4), 1985â€“2000,
    <pub-id pub-id-type="doi">10.1007/s12652-021-02960-0</pub-id>
    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1007/s12652-021-02960-0">10.1007/s12652-021-02960-0</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?q=Dua,+M.+C.+Jain,+%26+Kumar,+S.(2022).+%E2%80%9CLSTM+and+CNN+based+ensemble+approach+for+spoof+detection+task+in+automatic+speaker+verification+systems,%E2%80%9D+J.+Ambient+Intell.+Humaniz.Comput,+13(4),+1985%E2%80%932000,&amp;hl=en&amp;as_sdt=0,5">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Gao">
  <label>7</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Gao, Y.Vuong, M. Elyasi, G. Bharaj, &amp; Singh, R</person-group>
    <year>2021</year>
    <source>â€œGeneralized Spoofing Detection Inspired from Audio Generation Artifacts,â€</source>
    Gao, Y.Vuong, M. Elyasi, G. Bharaj, &amp; Singh, R.(2021). â€œGeneralized Spoofing Detection Inspired from Audio Generation Artifacts,â€
    <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/2104.04111">http://arxiv.org/abs/2104.04111</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?q=Gao,+Y.Vuong,+M.+Elyasi,+G.+Bharaj,+%26+Singh,+R.(2021).+%E2%80%9CGeneralized+Spoofing+Detection+Inspired+from+Audio+Generation+Artifacts,&amp;hl=en&amp;as_sdt=0,5">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="GomezalanisA">
  <label>8</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Gomez-alanisA. et al</person-group>
    <year>2017</year>
    <source>â€œOn Joint Optimization of Automatic Speaker Verification and Anti-spoofing in the Embedding Space,â€ i, 1â€“15</source>
    Gomez-alanisA. et al. (2017). â€œOn Joint Optimization of Automatic Speaker Verification and Anti-spoofing in the Embedding Space,â€ i, 1â€“15
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?q=Gomez-alanisA.+et+al.+(2017).+%E2%80%9COn+Joint+Optimization+of+Automatic+Speaker+Verification+and+Anti-spoofing+in+the+Embedding+Space,%E2%80%9D+i,+1%E2%80%9315.&amp;hl=en&amp;as_sdt=0,5">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Hamza">
  <label>9</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Hamza, A. et al</person-group>
    <year>2022</year>
    <source>â€œDeepfake Audio Detection via MFCC features using Machine Learning,â€ IEEE Access</source>
    Hamza, A. et al. (2022). â€œDeepfake Audio Detection via MFCC features using Machine Learning,â€ IEEE Access, 10, 134018â€“134028,
    <pub-id pub-id-type="doi">10.1109/ACCESS.2022.3231480</pub-id>
    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/ACCESS.2022.3231480">10.1109/ACCESS.2022.3231480</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?q=Hamza,+A.+et+al.+(2022).+%E2%80%9CDeepfake+Audio+Detection+via+MFCC+features+using+Machine+Learning,%E2%80%9D+IEEE+Access,+10,+134018%E2%80%93134028,&amp;hl=en&amp;as_sdt=0,5">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Ismail">
  <label>10</label>
  <mixed-citation publication-type="journal">
    <article-title>Ismail, A</article-title>
    <source>M. Elpeltagy, M. S. Zaki., &amp; K. Eldahshan, â€œA New Deep Learning-Based Methodology for Video Deepfake,â€ 1â€“1</source>
    Ismail, A. M. Elpeltagy, M. S. Zaki., &amp; K. Eldahshan, â€œA New Deep Learning-Based Methodology for Video Deepfake,â€ 1â€“1
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&amp;as_sdt=0%2C5&amp;q=Ismail%2C+A.+M.+Elpeltagy%2C+M.+S.+Zaki.%2C+%26+K.+Eldahshan%2C+%E2%80%9CA+New+Deep+Learning-Based+Methodology+for+Video+Deepfake%2C%E2%80%9D+1%E2%80%9315.&amp;btnG=">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Kinnunen">
  <label>11</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Kinnunen, T. et al</person-group>
    <year>2017</year>
    <source>â€œThe ASVspoof 2017 challenge: Assessing the limits of replay spoofing attack detection,â€ in Proceedings of the Annual Conference of the International Speech Communication Association, Interspeech, 2â€“6</source>
    <page-range>roceedings</page-range>
    Kinnunen, T. et al. (2017). â€œThe ASVspoof 2017 challenge: Assessing the limits of replay spoofing attack detection,â€ in Proceedings of the Annual Conference of the International Speech Communication Association, Interspeech, 2â€“6.
    <pub-id pub-id-type="doi">10.21437/Interspeech.2017-1111</pub-id>
    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.21437/Interspeech.2017-1111">10.21437/Interspeech.2017-1111</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&amp;as_sdt=0%2C5&amp;q=Kinnunen%2C+T.+et+al.+%282017%29.+%E2%80%9CThe+ASVspoof+2017+challenge%3A+Assessing+the+limits+of+replay+spoofing+attack+detection%2C%E2%80%9D+in+Proceedings+of+the+Annual+Conference+of+the+International+Speech+Communication+Association%2C+Interspeech%2C+2%E2%80%936.&amp;btnG=">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Lavrentyeva">
  <label>13</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Lavrentyeva, G. S. Novoselov, E. Malykh, A. Kozlov, O. Kudashev., &amp;Shchemelinin, V</person-group>
    <year>2017</year>
    <source>â€œAudio replay attack detection with deep learning frameworks,â€ in Proceedings of the Annual Conference of the International Speech Communication Association, Interspeech, 82â€“ 86</source>
    <page-range>roceedings</page-range>
    Lavrentyeva, G. S. Novoselov, E. Malykh, A. Kozlov, O. Kudashev., &amp;Shchemelinin, V. (2017). â€œAudio replay attack detection with deep learning frameworks,â€ in Proceedings of the Annual Conference of the International Speech Communication Association, Interspeech, 82â€“ 86.
    <pub-id pub-id-type="doi">10.21437/Interspeech.2017-360</pub-id>
    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.21437/Interspeech.2017-360">10.21437/Interspeech.2017-360</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?q=Lavrentyeva,+G.+S.+Novoselov,+E.+Malykh,+A.+Kozlov,+O.+Kudashev.,+%26Shchemelinin,+V.+(2017).+%E2%80%9CAudio+replay+attack+detection+with+deep+learning+frameworks,%E2%80%9D+in+Proceedings+of+the+Annual+Conference+of+the+International+Speech+Communication+Association,+Interspeech,+82%E2%80%93+86.&amp;hl=en&amp;as_sdt=0,5">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="LorenzoTrueba">
  <label>14</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Lorenzo-Trueba,J. et al</person-group>
    <year>2018</year>
    <source>â€œThe Voice Conversion Challenge</source>
    <page-range>romoting</page-range>
    Lorenzo-Trueba,J. et al.(2018). â€œThe Voice Conversion Challenge 2018: Promoting Development of Parallel and Nonparallel Methods,â€
    <ext-link ext-link-type="uri" xlink:href="http://arxiv.org/abs/1804.04262">http://arxiv.org/abs/1804.04262</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?q=Lorenzo-Trueba,J.+et+al.(2018).+%E2%80%9CThe+Voice+Conversion+Challenge+2018:+Promoting+Development+of+Parallel+and+Nonparallel+Methods,&amp;hl=en&amp;as_sdt=0,5">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Mcuba">
  <label>15</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Mcuba, M. A. Singh, R. A. Ikuesan., &amp; H. Venter</person-group>
    <year>2022</year>
    <article-title>. â€œThe Effect of Deep Learning Methods on Deepfake Audio Detection for Digital Investigation,â€ Procedia Comput</article-title>
    <source>Sci</source>
    <page-range>rocedia</page-range>
    Mcuba, M. A. Singh, R. A. Ikuesan., &amp; H. Venter, (2022). â€œThe Effect of Deep Learning Methods on Deepfake Audio Detection for Digital Investigation,â€ Procedia Comput. Sci., 219, 211â€“ 219.
    <pub-id pub-id-type="doi">10.1016/j.procs.2023.01.283</pub-id>
    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1016/j.procs.2023.01.283">10.1016/j.procs.2023.01.283</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&amp;as_sdt=0%2C5&amp;q=Mcuba%2C+M.+A.+Singh%2C+R.+A.+Ikuesan.%2C+%26+H.+Venter%2C+%282022%29.+%E2%80%9CThe+Effect+of+Deep+Learning+Methods+on+Deepfake+Audio+Detection+for+Digital+Investigation%2C%E2%80%9D+Procedia+Comput.+Sci.%2C+219%2C+211%E2%80%93+219.&amp;btnG=">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Todisco">
  <label>16</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Todisco, M. et al</person-group>
    <year>2019</year>
    <article-title>. â€œASVSpoof Future horizons in spoofed and fake audio detection,â€ Proc</article-title>
    <source>Annu. Conf. Int. Speech Commun. Assoc. Interspeech, 1008â€“1012</source>
    <page-range>roc</page-range>
    Todisco, M. et al.(2019). â€œASVSpoof Future horizons in spoofed and fake audio detection,â€ Proc. Annu. Conf. Int. Speech Commun. Assoc. Interspeech, 1008â€“1012,
    <pub-id pub-id-type="doi">10.21437/Interspeech.2019-2249</pub-id>
    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.21437/Interspeech.2019-2249">10.21437/Interspeech.2019-2249</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&amp;as_sdt=0%2C5&amp;q=Todisco%2C+M.+et+al.%282019%29.+%E2%80%9CASVSpoof+Future+horizons+in+spoofed+and+fake+audio+detection%2C%E2%80%9D+Proc.+Annu.+Conf.+Int.+Speech+Commun.+Assoc.+Interspeech%2C+1008%E2%80%931012%2C&amp;btnG=">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Wang">
  <label>17</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Wang, Z. S. Cui, X. Kang, W. Sun., &amp; Z. Li</person-group>
    <year>2020</year>
    <source>â€œDensely Connected Convolutional Network for Audio Spoofing Detection,â€ 1352â€“1360</source>
    Wang, Z. S. Cui, X. Kang, W. Sun., &amp; Z. Li, (2020). â€œDensely Connected Convolutional Network for Audio Spoofing Detection,â€ 1352â€“1360
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&amp;as_sdt=0%2C5&amp;q=Wang%2C+Z.+S.+Cui%2C+X.+Kang%2C+W.+Sun.%2C+%26+Z.+Li%2C+%282020%29.+%E2%80%9CDensely+Connected+Convolutional+Network+for+Audio+Spoofing+Detection%2C%E2%80%9D+1352%E2%80%931360.&amp;btnG=">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://www.apsipa.org/proceedings/2020/pdfs/0001352.pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Wenger">
  <label>18</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Wenger, E. M. Bronckers, C. Cianfarani, J. Cryan, A. Sha., &amp; B. Y. Zhao</person-group>
    <year>2021</year>
    <source>â€œHello, Itâ€™s Meâ€: Deep Learning-based Speech Synthesis A acks in the Real World,â€ 235â€“251</source>
    Wenger, E. M. Bronckers, C. Cianfarani, J. Cryan, A. Sha., &amp; B. Y. Zhao, (2021). â€œHello, Itâ€™s Meâ€: Deep Learning-based Speech Synthesis A acks in the Real World,â€ 235â€“251.
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?hl=en&amp;as_sdt=0%2C5&amp;q=Wenger%2C+E.+M.+Bronckers%2C+C.+Cianfarani%2C+J.+Cryan%2C+A.+Sha.%2C+%26+B.+Y.+Zhao%2C+%282021%29.+%E2%80%9CHello%2C+It%E2%80%99s+Me%E2%80%9D%3A+Deep+Learning-based+Speech+Synthesis+A+acks+in+the+Real+World%2C%E2%80%9D+235%E2%80%93251&amp;btnG=">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Yang">
  <label>19</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Yang, Y. et al</person-group>
    <year>2019</year>
    <source>â€œThe SJTU Robust Anti- spoofing System for the ASVspoof 2019 Challenge,â€ 1038â€“1042</source>
    Yang, Y. et al.(2019). â€œThe SJTU Robust Anti- spoofing System for the ASVspoof 2019 Challenge,â€ 1038â€“1042,
    <pub-id pub-id-type="doi">10.21437/Interspeech.2019-2170</pub-id>
    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.21437/Interspeech.2019-2170">10.21437/Interspeech.2019-2170</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?q=Yang,+Y.+et+al.(2019).+%E2%80%9CThe+SJTU+Robust+Antispoofing+System+for+the+ASVspoof+2019&amp;hl=en&amp;as_sdt=0,5">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Yi">
  <label>20</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Yi, J. et al</person-group>
    <year>2021</year>
    <article-title>. â€œHalf-truth: A partially fake audio detection dataset,â€ Proc</article-title>
    <source>Annu. Conf. Int. Speech Commun. Assoc. INTERSPEECH</source>
    <page-range>artially</page-range>
    Yi, J. et al.(2021). â€œHalf-truth: A partially fake audio detection dataset,â€ Proc. Annu. Conf. Int. Speech Commun. Assoc. INTERSPEECH, 4, 2683â€“2687,
    <pub-id pub-id-type="doi">10.21437/Interspeech.2021-930</pub-id>
    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.21437/Interspeech.2021-930">10.21437/Interspeech.2021-930</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?q=Yi,+J.+et+al.(2021).+%E2%80%9CHalf-truth:+A+partially+fake+audio+detection+dataset,%E2%80%9D+Proc.+Annu.+Conf.+Int.+Speech+Commun.+Assoc.+INTERSPEECH,+4,+2683%E2%80%932687,&amp;hl=en&amp;as_sdt=0,5">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://ieeexplore.ieee.org/document/9996362">Fulltext</ext-link>
  </mixed-citation>
</ref>
<ref id="Yu">
  <label>22</label>
  <mixed-citation publication-type="journal">
    <person-group person-group-type="author">Yu, Y. et al</person-group>
    <year>2020</year>
    <source>â€œRMAF : Relu-Memristor-Like Activation Function for Deep Learning,â€ IEEE Access</source>
    Yu, Y. et al.(2020). â€œRMAF : Relu-Memristor-Like Activation Function for Deep Learning,â€ IEEE Access, 8, 72727â€“72741,
    <pub-id pub-id-type="doi">10.1109/ACCESS.2020.2987829</pub-id>
    <ext-link ext-link-type="doi" xlink:href="https://doi.org/10.1109/ACCESS.2020.2987829">10.1109/ACCESS.2020.2987829</ext-link>
    <ext-link ext-link-type="uri" xlink:href="https://scholar.google.com/scholar?q=Yu,+Y.+et+al.(2020).+%E2%80%9CRMAF+:+Relu-Memristor-Like+Activation+Function+for+Deep+Learning,%E2%80%9D+IEEE+Access,+8,+72727%E2%80%9372741,&amp;hl=en&amp;as_sdt=0,5">Google Scholar</ext-link>
    <ext-link ext-link-type="uri" xlink:href="http://citeseerx.ist.psu.edu/viewdoc/download?doi=10.1.1.490.6927&amp;rep=rep1&amp;type=pdf">Fulltext</ext-link>
  </mixed-citation>
</ref>
</ref-list>
</back>
</article>