<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CSSE</journal-id>
<journal-id journal-id-type="nlm-ta">CSSE</journal-id>
<journal-id journal-id-type="publisher-id">CSSE</journal-id>
<journal-title-group>
<journal-title>Computer Systems Science &#x0026; Engineering</journal-title>
</journal-title-group>
<issn pub-type="ppub">0267-6192</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">16189</article-id>
<article-id pub-id-type="doi">10.32604/csse.2021.016189</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Residential Electricity Classification Method Based On Cloud Computing Platform and Random Forest</article-title><alt-title alt-title-type="left-running-head">Residential Electricity Classification Method Based On Cloud Computing Platform and Random Forest</alt-title><alt-title alt-title-type="right-running-head">Residential Electricity Classification Method Based On Cloud Computing Platform and Random Forest</alt-title>
</title-group>
<contrib-group content-type="authors">
<contrib id="author-1" contrib-type="author">
<name name-style="western">
<surname>Li</surname>
<given-names>Ming</given-names>
</name>
<xref ref-type="aff" rid="aff-1">1</xref>
</contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western">
<surname>Fang</surname>
<given-names>Zhong</given-names>
</name>
<xref ref-type="aff" rid="aff-2">2</xref>
</contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western">
<surname>Cao</surname>
<given-names>Wanwan</given-names>
</name>
<xref ref-type="aff" rid="aff-1">1</xref>
</contrib>
<contrib id="author-4" contrib-type="author" corresp="yes">
<name name-style="western">
<surname>Ma</surname>
<given-names>Yong</given-names>
</name>
<xref ref-type="aff" rid="aff-1">1</xref><email>mayongah@163.com</email>
</contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western">
<surname>Wu</surname>
<given-names>Shang</given-names>
</name>
<xref ref-type="aff" rid="aff-1">1</xref>
</contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western">
<surname>Guo</surname>
<given-names>Yang</given-names>
</name>
<xref ref-type="aff" rid="aff-1">1</xref>
</contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western">
<surname>Xue</surname>
<given-names>Yu</given-names>
</name>
<xref ref-type="aff" rid="aff-3">3</xref>
</contrib>
<contrib id="author-8" contrib-type="author">
<name name-style="western">
<surname>Mansour</surname>
<given-names>Romany F.</given-names>
</name>
<xref ref-type="aff" rid="aff-4">4</xref>
</contrib>
<aff id="aff-1">
<label>1</label><institution>Information and Communication Branch of State Grid Anhui Electric Power Co., Ltd.</institution>, <addr-line>Hefei, 230009</addr-line>, <country>China</country></aff>
<aff id="aff-2">
<label>2</label><institution>State Grid Anhui Electric Power Co., Ltd., Chuzhou Power Supply Company</institution>, <addr-line>Chuzhou, 239000</addr-line>, <country>China</country></aff>
<aff id="aff-3">
<label>3</label><institution>School of Computer and Software, Nanjing University of Information Science and Technology</institution>, <addr-line>Nanjing, 210044</addr-line>, <country>China</country></aff>
<aff id="aff-4">
<label>4</label><institution>Department of Mathematics, Faculty of Science, New Valley University</institution>, <addr-line>El-Kharga, 72511</addr-line>, <country>Egypt</country></aff>
</contrib-group><author-notes><corresp id="cor1">&#x002A;Corresponding Author: Yong Ma. Email: <email>mayongah@163.com</email></corresp></author-notes>
<pub-date pub-type="epub" date-type="pub" iso-8601-date="2021-01-01">
<day>1</day>
<month>1</month>
<year iso-8601-date="2021">2021</year>
</pub-date>
<volume>38</volume>
<issue>1</issue>
<fpage>39</fpage>
<lpage>46</lpage>
<history>
<date date-type="received">
<day>20</day>
<month>12</month>
<year iso-8601-date="2020">2020</year>
</date>
<date date-type="accepted">
<day>26</day>
<month>1</month>
<year iso-8601-date="2021">2021</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2021 Li et al.</copyright-statement>
<copyright-year>2021</copyright-year>
<copyright-holder>Li et al.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CSSE_16189.pdf"></self-uri>
<abstract>
<p>With the rapid development and popularization of new-generation technologies such as cloud computing, big data, and artificial intelligence, the construction of smart grids has become more diversified. Accurate quick reading and classification of the electricity consumption of residential users can provide a more in-depth perception of the actual power consumption of residents, which is essential to ensure the normal operation of the power system, energy management and planning. Based on the distributed architecture of cloud computing, this paper designs an improved random forest residential electricity classification method. It uses the unique out-of-bag error of random forest and combines the Drosophila algorithm to optimize the internal parameters of the random forest, thereby improving the performance of the random forest algorithm. This method uses MapReduce to train an improved random forest model on the cloud computing platform, and then uses the trained model to analyze the residential electricity consumption data set, divides all residents into 5 categories, and verifies the effectiveness of the model through experiments and feasibility.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Cloud computing</kwd>
<kwd>Hadoop</kwd>
<kwd>random forest</kwd>
<kwd>user classification</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>With the construction and development of urbanization in China, the number of residential quarters of cities is increasing. The total amount of electricity consumed by urban residents is also increasing, which brings new challenges to the stable operation of the power grid. Categorizing the electricity consumption of residential users can help the power supply side perform better resource management and distribution, and reduce power loss [<xref ref-type="bibr" rid="ref-1">1</xref>,<xref ref-type="bibr" rid="ref-2">2</xref>].</p>
<p>There have been many analyses and studies on residential electricity consumption. Song et al. [<xref ref-type="bibr" rid="ref-3">3</xref>] used factors such as holidays and weekends as the parameters of the ANN (Artificial Neural Network) model to predict short-term power consumption. Loktionov et al. [<xref ref-type="bibr" rid="ref-4">4</xref>] used the seasonal decomposition method to study the relationship between urban electricity consumption and climate change. They analyze electricity consumption in terms of time. Liu et al. [<xref ref-type="bibr" rid="ref-5">5</xref>] proposed a method to identify users in the same transformer area based on an improved k-means clustering algorithm. However, these do not involve the classification of specific users. Zhang et al. [<xref ref-type="bibr" rid="ref-6">6</xref>] used the two-step clustering method to classify residential users by extracting the characteristics of the electricity load curve. Song et al. [<xref ref-type="bibr" rid="ref-7">7</xref>] analyzed the relationship between household income and user electricity consumption. Song et al. [<xref ref-type="bibr" rid="ref-8">8</xref>] analyzed the user&#x2019;s power consumption pattern of pattern recognition power supplies. The above methods have classified users from different perspectives, but due to the increasing amount of data in the grid system now. Some current methods focus more on optimizing classification accuracy, and ignore the timeliness of response, which can no longer meet the demand.</p>
<p>Based on this, this paper designs a random forest residential electricity classification method based on the distributed architecture of cloud computing. This method firstly uses the fruit fly algorithm to improve the random forest model, and then uses MapReduce to train the improved random forest model on the cloud computing platform. After that, the trained model is used to analyze the residential electricity consumption data set, and all residents are divided into 5 categories. The effect of the model is verified through experiments.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Preparation</title>
<sec id="s2_1">
<label>2.1</label>
<title>Cloud Computing Architecture</title>
<p>Cloud computing is a concept first proposed by Google [<xref ref-type="bibr" rid="ref-9">9</xref>]. It is the development of distributed computing, grid computing and parallel computing. It distributes computing tasks to a large number of computers or virtual machines through the network, and realizes the decomposition of huge computing tasks, thereby obtaining the calculation results of the tasks faster. Therefore, it is very suitable for the analysis and processing of massive concurrent data. Currently the main popular cloud computing technologies are Hadoop and Spark. This paper uses Hadoop architecture to build the model. The main task deployment of Hadoop is divided into three parts: client machine, master node and slave node [<xref ref-type="bibr" rid="ref-10">10</xref>]. The master node is responsible for the supervision of HDFS (Hadoop Distributed File System) and Map Reduce, and the slave node is responsible for computing instructions and storing data [<xref ref-type="bibr" rid="ref-11">11</xref>].</p>
<p>The overall framework of Hadoop is shown in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>. Among them, HDFS is a distributed file system, MapReduce is a programming model, and HDFS and Map Reduce are two key functional modules of Hadoop. In addition, HBase is a distributed database based on the column storage model, and Pig is a data analysis platform that provides operations and programming interfaces for massive data parallel operations. Hive is a tool that provides SQL queries, Sqoop is a tool for data transfer between Hadoop and traditional databases, ZooKeeper is a collaborative working system, and Avro is a data serialization system [<xref ref-type="bibr" rid="ref-12">12</xref>].</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>The overall framework of Hadoop</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-1.png"/>
</fig>
</sec>
<sec id="s2_2">
<label>2.2</label>
<title>Random Forest</title>
<p>Random forest is an ensemble classification algorithm composed of several decision trees based on statistical learning theory [<xref ref-type="bibr" rid="ref-13">13</xref>,<xref ref-type="bibr" rid="ref-14">14</xref>]. The principle is to extract some samples of the original sample set by using the re-sampling method that can be replaced to train multiple decision trees. The final prediction result of the random forest model is obtained by comprehensively training the prediction results of multiple decision trees.</p>
<p>Suppose a random forest <inline-formula id="ieqn-1">
<!--<alternatives><inline-graphic xlink:href="ieqn-1.png"/><tex-math id="tex-ieqn-1"><![CDATA[RF = \left\{ {\Re (X,{T_i}),i = 1,2,3,&#x2026;,N} \right\}]]></tex-math>--><mml:math id="mml-ieqn-1"><mml:mi>R</mml:mi><mml:mi>F</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi mathvariant="normal">&#x211C;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mi>N</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math>
<!--</alternatives>--></inline-formula> contains <italic>N</italic> decision trees, <inline-formula id="ieqn-2">
<!--<alternatives><inline-graphic xlink:href="ieqn-2.png"/><tex-math id="tex-ieqn-2"><![CDATA[{T_i}]]></tex-math>--><mml:math id="mml-ieqn-2"><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math>
<!--</alternatives>--></inline-formula> is an independently distributed random vector, and <italic>X</italic> is a sample set. As shown in the <xref ref-type="fig" rid="fig-2">Fig. 2</xref>: the basic idea of random forest classification is: randomly extract a sampling sample set <italic>A</italic> from the sample set, train a decision tree respectively, and use the majority voting method to vote according to the output results of these decision trees to obtain the final random forest classification result <inline-formula id="ieqn-3">
<!--<alternatives><inline-graphic xlink:href="ieqn-3.png"/><tex-math id="tex-ieqn-3"><![CDATA[RF\left( x \right) = \arg \max \sum\limits_{i = 1}^N {F\left( {{T_i},x} \right)}]]></tex-math>--><mml:math id="mml-ieqn-3"><mml:mi>R</mml:mi><mml:mi>F</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mi>arg</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mo form="prefix" movablelimits="true">max</mml:mo><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mi>N</mml:mi></mml:munderover><mml:mrow><mml:mi>F</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mi>x</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:math>
<!--</alternatives>--></inline-formula>, where represents the output result of the random forest on the data <italic>x</italic>, and <italic>A</italic> represents the indicative function of the classification result of the <italic>i</italic>-th decision tree on the data <italic>x</italic>.</p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Random forest diagram</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-2.png"/>
</fig>
<p>When constructing a random forest, an out-of-bag data set will be generated. The out-of-bag error of the out-of-bag data set can be used to measure the performance of the model. The random forest out-of-bag error formula is as follows:</p>
<p><disp-formula id="eqn-1">
<label>(1)</label>
<!--<alternatives><graphic mimetype="image" mime-subtype="png" xlink:href="eqn-1.png"/><tex-math id="tex-eqn-1"><![CDATA[Oob = \displaystyle{{Err\left( {X - A} \right)} \over {X - A}}]]></tex-math>--><mml:math id="mml-eqn-1" display="block"><mml:mi>O</mml:mi><mml:mi>o</mml:mi><mml:mi>b</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mstyle scriptlevel="0" displaystyle="true"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>E</mml:mi><mml:mi>r</mml:mi><mml:mi>r</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>X</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>A</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>X</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>A</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math>
<!--</alternatives>--></disp-formula></p>
<p>The function <italic>Err</italic> is the count of the number of misclassifications.</p>
<p>According to random forest <inline-formula id="ieqn-4">
<!--<alternatives><inline-graphic xlink:href="ieqn-4.png"/><tex-math id="tex-ieqn-4"><![CDATA[RF = \left\{ {\Re (X,{T_i}),i = 1,2,3,&#x2026;,N} \right\}]]></tex-math>--><mml:math id="mml-ieqn-4"><mml:mi>R</mml:mi><mml:mi>F</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi mathvariant="normal">&#x211C;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mi>N</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math>
<!--</alternatives>--></inline-formula>, the remaining quantity function can be obtained as:</p>
<p><disp-formula id="eqn-2">
<label>(2)</label>
<!--<alternatives><graphic mimetype="image" mime-subtype="png" xlink:href="eqn-2.png"/><tex-math id="tex-eqn-2"><![CDATA[MG\left( X \right) = a{v_k}\left( {F\left( {{T_i},X} \right) = Y} \right) - MAXa{v_k}\left( {F\left( {{T_i},X} \right) \ne Y} \right)]]></tex-math>--><mml:math id="mml-eqn-2" display="block"><mml:mi>M</mml:mi><mml:mi>G</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>X</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mi>a</mml:mi><mml:mrow><mml:msub><mml:mi>v</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>F</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mi>X</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mi>Y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>M</mml:mi><mml:mi>A</mml:mi><mml:mi>X</mml:mi><mml:mi>a</mml:mi><mml:mrow><mml:msub><mml:mi>v</mml:mi><mml:mi>k</mml:mi></mml:msub></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>F</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mi>X</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>&#x2260;</mml:mo><mml:mi>Y</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
<!--</alternatives>--></disp-formula></p>
<p>The larger the margin value, the more reliable the classification prediction.</p>
</sec>
</sec>
<sec id="s3">
<label>3</label>
<title>Model Building</title>
<sec id="s3_1">
<label>3.1</label>
<title>In-line Style</title>
<p>Since the collected electricity consumption data will have problems such as missing and irregularities, the data needs to be preprocessed. Preprocessing is divided into two steps: missing data completion and data standardization.</p>
<p>Use interpolation method to complete the data, set the original data set as <inline-formula id="ieqn-5">
<!--<alternatives><inline-graphic xlink:href="ieqn-5.png"/><tex-math id="tex-ieqn-5"><![CDATA[\bar X = \{ {x_{1,1}},&#x2026;,{\bar x_{1,n}},&#x2026;,{\bar x_{m,1}},&#x2026;,{\bar x_{m,n}}\}]]></tex-math>--><mml:math id="mml-ieqn-5"><mml:mrow><mml:mover><mml:mi>X</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mo stretchy="false" fence="false">{</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false" fence="false">}</mml:mo></mml:math>
<!--</alternatives>--></inline-formula>, if the data is missing, then</p>
<p><disp-formula id="eqn-3">
<label>(3)</label>
<!--<alternatives><graphic mimetype="image" mime-subtype="png" xlink:href="eqn-3.png"/><tex-math id="tex-eqn-3"><![CDATA[{\bar x_{i,j}} = \displaystyle{{{{\bar x}_{i,j - 1}} + {{\bar x}_{i,j + 1}}} \over 2}]]></tex-math>--><mml:math id="mml-eqn-3" display="block"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mstyle scriptlevel="0" displaystyle="true"><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x002B;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:mfrac></mml:mrow></mml:mstyle></mml:math>
<!--</alternatives>--></disp-formula></p>
<p>If continuous data is missing, for example, <inline-formula id="ieqn-6">
<!--<alternatives><inline-graphic xlink:href="ieqn-6.png"/><tex-math id="tex-ieqn-6"><![CDATA[{\bar x_{i,j}},&#x2026;,{\bar x_{i,j + t}}]]></tex-math>--><mml:math id="mml-ieqn-6"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:math>
<!--</alternatives>--></inline-formula> are all missing, then</p>
<p><disp-formula id="eqn-4">
<label>(4)</label>
<!--<alternatives><graphic mimetype="image" mime-subtype="png" xlink:href="eqn-4.png"/><tex-math id="tex-eqn-4"><![CDATA[{\bar x_{i,j}} = \displaystyle{{{{\bar x}_{i,j - 1}} + {{\bar x}_{i,j + t + 1}}} \over 2}]]></tex-math>--><mml:math id="mml-eqn-4" display="block"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mstyle scriptlevel="0" displaystyle="true"><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x002B;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:mfrac></mml:mrow></mml:mstyle></mml:math>
<!--</alternatives>--></disp-formula></p>
<p><disp-formula id="eqn-6">
<label>(5)</label>
<!--<alternatives><graphic mimetype="image" mime-subtype="png" xlink:href="eqn-6.png"/><tex-math id="tex-eqn-6"><![CDATA[{\bar x_{i,j + i}} = \displaystyle{{{{\bar x}_{i,j + i - 1}} + {{\bar x}_{i,j + t + 1}}} \over 2}, \ i = 1,&#x2026;,t]]></tex-math>--><mml:math id="mml-eqn-6" display="block"><mml:mrow><mml:msub><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x002B;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:mrow><mml:mover><mml:mi>x</mml:mi><mml:mo stretchy="false">&#x00AF;</mml:mo></mml:mover></mml:mrow></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mi>t</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mn>2</mml:mn></mml:mfrac></mml:mrow><mml:mo>,</mml:mo><mml:mtext> </mml:mtext><mml:mi>i</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mi>t</mml:mi></mml:mstyle></mml:math>
<!--</alternatives>--></disp-formula></p>
<p>Suppose the data set to complete the missing value completion is <inline-formula id="ieqn-7">
<!--<alternatives><inline-graphic xlink:href="ieqn-7.png"/><tex-math id="tex-ieqn-7"><![CDATA[{X}^{\prime} = \{ {{x}^{\prime}_{1,1}},&#x2026;,{{x}^{\prime}_{1,n}},&#x2026;,{{x}^{\prime}_{m,1}},&#x2026;,{{x}^{\prime}_{m,n}}\}]]></tex-math>--><mml:math id="mml-ieqn-7"><mml:msup><mml:mrow><mml:mi>X</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>&#x003D;</mml:mo><mml:mo stretchy="false" fence="false">{</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:msubsup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msubsup></mml:mrow><mml:mo stretchy="false" fence="false">}</mml:mo></mml:math>
<!--</alternatives>--></inline-formula>, and then use the <inline-formula id="ieqn-8">
<!--<alternatives><inline-graphic xlink:href="ieqn-8.png"/><tex-math id="tex-ieqn-8"><![CDATA[\min {-} \max]]></tex-math>--><mml:math id="mml-ieqn-8"><mml:mo movablelimits="true" form="prefix">min</mml:mo><mml:mrow><mml:mo>&#x2212;</mml:mo></mml:mrow><mml:mo movablelimits="true" form="prefix">max</mml:mo></mml:math>
<!--</alternatives>--></inline-formula> function to standardize it so that the values of all data are in the interval <inline-formula id="ieqn-9">
<!--<alternatives><inline-graphic xlink:href="ieqn-9.png"/><tex-math id="tex-ieqn-9"><![CDATA[[0,1]]]></tex-math>--><mml:math id="mml-ieqn-9"><mml:mo stretchy="false">[</mml:mo><mml:mn>0</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn><mml:mo stretchy="false">]</mml:mo></mml:math>
<!--</alternatives>--></inline-formula>.The <inline-formula id="ieqn-10">
<!--<alternatives><inline-graphic xlink:href="ieqn-10.png"/><tex-math id="tex-ieqn-10"><![CDATA[\min - \max]]></tex-math>--><mml:math id="mml-ieqn-10"><mml:mo form="prefix" movablelimits="true">min</mml:mo><mml:mo>&#x2212;</mml:mo><mml:mo form="prefix" movablelimits="true">max</mml:mo></mml:math>
<!--</alternatives>--></inline-formula> function is</p>
<p><disp-formula id="eqn-7">
<label>(6)</label>
<!--<alternatives><graphic mimetype="image" mime-subtype="png" xlink:href="eqn-7.png"/><tex-math id="tex-eqn-7"><![CDATA[{x_{i,j}} = \displaystyle{{{{{x}^{\prime}}_{i,j}} - {{{x}^{\prime}}_{\min }}} \over {{{{x}^{\prime}}_{\max }} - {{{x}^{\prime}}_{\min }}}}]]></tex-math>--><mml:math id="mml-eqn-7" display="block"><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mstyle scriptlevel="0" displaystyle="true"><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>,</mml:mo><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mo form="prefix" movablelimits="true">min</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mrow><mml:mrow><mml:msub><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mo form="prefix" movablelimits="true">max</mml:mo></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:msub><mml:mrow><mml:msup><mml:mrow><mml:mi>x</mml:mi></mml:mrow><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:mrow><mml:mrow><mml:mo form="prefix" movablelimits="true">min</mml:mo></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math>
<!--</alternatives>--></disp-formula></p>
<p>The data set after preprocessing is <inline-formula id="ieqn-11">
<!--<alternatives><inline-graphic xlink:href="ieqn-11.png"/><tex-math id="tex-ieqn-11"><![CDATA[X = \{ {x_{1,1}},&#x2026;,{x_{1,n}},&#x2026;,{x_{m,1}},&#x2026;,{x_{m,n}}\}]]></tex-math>--><mml:math id="mml-ieqn-11"><mml:mi>X</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mo stretchy="false" fence="false">{</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>m</mml:mi><mml:mo>,</mml:mo><mml:mi>n</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo stretchy="false" fence="false">}</mml:mo></mml:math>
<!--</alternatives>--></inline-formula>.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Random Forest Model</title>
<p>This paper uses the unique out-of-bag error of random forest to optimize the internal parameters of random forest and improve random forest to improve the performance of random forest model algorithm. Assume that the random forest model <inline-formula id="ieqn-12">
<!--<alternatives><inline-graphic xlink:href="ieqn-12.png"/><tex-math id="tex-ieqn-12"><![CDATA[RF = \left\{ {\Re (X,{T_i}),i = 1,2,3,&#x2026;,N} \right\}]]></tex-math>--><mml:math id="mml-ieqn-12"><mml:mi>R</mml:mi><mml:mi>F</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mi mathvariant="normal">&#x211C;</mml:mi><mml:mo stretchy="false">(</mml:mo><mml:mi>X</mml:mi><mml:mo>,</mml:mo><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false">)</mml:mo><mml:mo>,</mml:mo><mml:mi>i</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mn>1</mml:mn><mml:mo>,</mml:mo><mml:mn>2</mml:mn><mml:mo>,</mml:mo><mml:mn>3</mml:mn><mml:mo>,</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mi>N</mml:mi></mml:mrow><mml:mo>}</mml:mo></mml:mrow></mml:math>
<!--</alternatives>--></inline-formula> trained by the sample set <italic>X</italic> contains <italic>N</italic> decision trees, and the number of features of the sample set <italic>X</italic> are <italic>M</italic>.</p>
<p>We combine the drosophila algorithm to optimize the internal parameters of the random forest model [<xref ref-type="bibr" rid="ref-15">15</xref>,<xref ref-type="bibr" rid="ref-16">16</xref>]. In the process of random forest classification and recognition, the parameters that determine the accuracy are generally the number of decision trees <italic>TN</italic> and the size of the attribute feature subset <italic>FT</italic>. Where the number of decision trees <italic>TN</italic> is too large, the model training time may be too long, and too small will cause the accuracy to decrease, and the appropriate attribute features subset size <italic>FT</italic> can also greatly improve the model performance. Therefore, it is very important to <italic>TN</italic> and <italic>FT</italic> optimization, and the fruit fly algorithm can meet this requirement.</p>
<p>Suppose the size of the fruit fly population is <inline-formula id="ieqn-13">
<!--<alternatives><inline-graphic xlink:href="ieqn-13.png"/><tex-math id="tex-ieqn-13"><![CDATA[ps]]></tex-math>--><mml:math id="mml-ieqn-13"><mml:mi>p</mml:mi><mml:mi>s</mml:mi></mml:math>
<!--</alternatives>--></inline-formula>, the maximum number of iterations of the algorithm is <inline-formula id="ieqn-14">
<!--<alternatives><inline-graphic xlink:href="ieqn-14.png"/><tex-math id="tex-ieqn-14"><![CDATA[\alpha]]></tex-math>--><mml:math id="mml-ieqn-14"><mml:mi>&#x03B1;</mml:mi></mml:math>
<!--</alternatives>--></inline-formula>, and the initial fruit fly population coordinates are <inline-formula id="ieqn-15">
<!--<alternatives><inline-graphic xlink:href="ieqn-15.png"/><tex-math id="tex-ieqn-15"><![CDATA[\left( {OO{B_{best}},T{N_{ori}},F{T_{ori}}} \right)]]></tex-math>--><mml:math id="mml-ieqn-15"><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>O</mml:mi><mml:mi>O</mml:mi><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mi>T</mml:mi><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mi>F</mml:mi><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:math>
<!--</alternatives>--></inline-formula>. The fruit fly algorithm is divided into two stages, namely the olfactory search stage and the visual positioning stage. In the olfactory search stage, calculate the average odor concentration <inline-formula id="ieqn-16">
<!--<alternatives><inline-graphic xlink:href="ieqn-16.png"/><tex-math id="tex-ieqn-16"><![CDATA[AO{B_i}^j]]></tex-math>--><mml:math id="mml-ieqn-16"><mml:mi>A</mml:mi><mml:mi>O</mml:mi><mml:msup><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mi>j</mml:mi></mml:msup></mml:math>
<!--</alternatives>--></inline-formula> of the <italic>i-</italic>th fruit fly in the <italic>j-</italic>th generation</p>
<p><disp-formula id="eqn-8">
<label>(7)</label>
<!--<alternatives><graphic mimetype="image" mime-subtype="png" xlink:href="eqn-8.png"/><tex-math id="tex-eqn-8"><![CDATA[AOB_i^j = \displaystyle{{\left( {\sum\limits_{i = 1}^{ps} {OOB_i^j} } \right)} \over {ps}}]]></tex-math>--><mml:math id="mml-eqn-8" display="block"><mml:mi>A</mml:mi><mml:mi>O</mml:mi><mml:msubsup><mml:mi>B</mml:mi><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>&#x003D;</mml:mo><mml:mstyle scriptlevel="0" displaystyle="true"><mml:mrow><mml:mfrac><mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:munderover><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:munderover><mml:mrow><mml:mi>O</mml:mi><mml:mi>O</mml:mi><mml:msubsup><mml:mi>B</mml:mi><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:msubsup></mml:mrow></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow><mml:mrow><mml:mi>p</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math>
<!--</alternatives>--></disp-formula></p>
<p>where <inline-formula id="ieqn-17">
<!--<alternatives><inline-graphic xlink:href="ieqn-17.png"/><tex-math id="tex-ieqn-17"><![CDATA[OOB_i^j]]></tex-math>--><mml:math id="mml-ieqn-17"><mml:mi>O</mml:mi><mml:mi>O</mml:mi><mml:msubsup><mml:mi>B</mml:mi><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:msubsup></mml:math>
<!--</alternatives>--></inline-formula> represents the odor concentration of the <italic>i</italic>-th fruit fly at the <italic>j</italic>-th generation, that is, the error outside the random forest bag, and <inline-formula id="ieqn-18">
<!--<alternatives><inline-graphic xlink:href="ieqn-18.png"/><tex-math id="tex-ieqn-18"><![CDATA[A{d_i}]]></tex-math>--><mml:math id="mml-ieqn-18"><mml:mi>A</mml:mi><mml:mrow><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow></mml:math>
<!--</alternatives>--></inline-formula> represent the odor fitness value of the <italic>i</italic>-th fruit fly. The formula is as follows:</p>
<p><disp-formula id="eqn-9">
<label>(8)</label>
<!--<alternatives><graphic mimetype="image" mime-subtype="png" xlink:href="eqn-9.png"/><tex-math id="tex-eqn-9"><![CDATA[A{d_i} = \displaystyle{1 \over {\sqrt {TN_i^2 + FT_i^2} }}]]></tex-math>--><mml:math id="mml-eqn-9" display="block"><mml:mi>A</mml:mi><mml:mrow><mml:msub><mml:mi>d</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mstyle scriptlevel="0" displaystyle="true"><mml:mrow><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:msqrt><mml:mi>T</mml:mi><mml:msubsup><mml:mi>N</mml:mi><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:msubsup><mml:mo>&#x002B;</mml:mo><mml:mi>F</mml:mi><mml:msubsup><mml:mi>T</mml:mi><mml:mi>i</mml:mi><mml:mn>2</mml:mn></mml:msubsup></mml:msqrt></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math>
<!--</alternatives>--></disp-formula></p>
<p>Then calculate the rate of change of <inline-formula id="ieqn-19">
<!--<alternatives><inline-graphic xlink:href="ieqn-19.png"/><tex-math id="tex-ieqn-19"><![CDATA[AO{B_i}^j]]></tex-math>--><mml:math id="mml-ieqn-19"><mml:mi>A</mml:mi><mml:mi>O</mml:mi><mml:msup><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mi>j</mml:mi></mml:msup></mml:math>
<!--</alternatives>--></inline-formula>:</p>
<p><disp-formula id="eqn-10">
<label>(9)</label>
<!--<alternatives><graphic mimetype="image" mime-subtype="png" xlink:href="eqn-10.png"/><tex-math id="tex-eqn-10"><![CDATA[V = \displaystyle{{AO{B_j} - AO{B_{j - 1}}} \over {AO{B_{j - 1}}}}]]></tex-math>--><mml:math id="mml-eqn-10" display="block"><mml:mi>V</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mstyle scriptlevel="0" displaystyle="true"><mml:mrow><mml:mfrac><mml:mrow><mml:mi>A</mml:mi><mml:mi>O</mml:mi><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mi>j</mml:mi></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>A</mml:mi><mml:mi>O</mml:mi><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mrow><mml:mi>A</mml:mi><mml:mi>O</mml:mi><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>j</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mn>1</mml:mn></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:mfrac></mml:mrow></mml:mstyle></mml:math>
<!--</alternatives>--></disp-formula></p>
<p>According to the change rate, the optimal step weight <inline-formula id="ieqn-20">
<!--<alternatives><inline-graphic xlink:href="ieqn-20.png"/><tex-math id="tex-ieqn-20"><![CDATA[W = V + \epsilon]]></tex-math>--><mml:math id="mml-ieqn-20"><mml:mi>W</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mi>V</mml:mi><mml:mo>&#x002B;</mml:mo><mml:mi>&#x03B5;</mml:mi></mml:math>
<!--</alternatives>--></inline-formula> is obtained, where <inline-formula id="ieqn-21">
<!--<alternatives><inline-graphic xlink:href="ieqn-21.png"/><tex-math id="tex-ieqn-21"><![CDATA[\epsilon]]></tex-math>--><mml:math id="mml-ieqn-21"><mml:mi>&#x03B5;</mml:mi></mml:math>
<!--</alternatives>--></inline-formula> is the adaptive threshold.</p>
<p>Calculate the adaptive optimization route of drosophila:</p>
<p><disp-formula id="eqn-11">
<label>(10)</label>
<!--<alternatives><graphic mimetype="image" mime-subtype="png" xlink:href="eqn-11.png"/><tex-math id="tex-eqn-11"><![CDATA[T{N_i} = T{N_{ori}} + WV,F{T_i} = F{T_{ori}} + WV]]></tex-math>--><mml:math id="mml-eqn-11" display="block"><mml:mi>T</mml:mi><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mi>T</mml:mi><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x002B;</mml:mo><mml:mi>W</mml:mi><mml:mi>V</mml:mi><mml:mo>,</mml:mo><mml:mi>F</mml:mi><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mi>F</mml:mi><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x002B;</mml:mo><mml:mi>W</mml:mi><mml:mi>V</mml:mi></mml:math>
<!--</alternatives>--></disp-formula></p>
<p>then:</p>
<p><disp-formula id="eqn-12">
<label>(11)</label>
<!--<alternatives><graphic mimetype="image" mime-subtype="png" xlink:href="eqn-12.png"/><tex-math id="tex-eqn-12"><![CDATA[OO{B_{best}} &#x003D; \left\{ {\matrix{ {OO{B_{best}},\left( {OO{B_{best}} - OOB_i^j < \theta } \right)} \cr {OO{B_i}^j,\left( {OO{B_{best}} - OOB_i^j \ge \theta } \right)} \cr } } \right.]]></tex-math>--><mml:math id="mml-eqn-12" display="block"><mml:mi>O</mml:mi><mml:mi>O</mml:mi><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x003D;</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mrow><mml:mtable columnspacing="1em" rowspacing="4pt"><mml:mtr><mml:mtd><mml:mrow><mml:mi>O</mml:mi><mml:mi>O</mml:mi><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>O</mml:mi><mml:mi>O</mml:mi><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>O</mml:mi><mml:mi>O</mml:mi><mml:msubsup><mml:mi>B</mml:mi><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>&#x003C;</mml:mo><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mrow><mml:mi>O</mml:mi><mml:mi>O</mml:mi><mml:msup><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mi>i</mml:mi></mml:msub></mml:mrow><mml:mi>j</mml:mi></mml:msup><mml:mo>,</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mi>O</mml:mi><mml:mi>O</mml:mi><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mi>O</mml:mi><mml:mi>O</mml:mi><mml:msubsup><mml:mi>B</mml:mi><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:msubsup><mml:mo>&#x2265;</mml:mo><mml:mi>&#x03B8;</mml:mi></mml:mrow><mml:mo>)</mml:mo></mml:mrow></mml:mrow></mml:mtd></mml:mtr></mml:mtable></mml:mrow><mml:mo stretchy="true" fence="true" symmetric="true"></mml:mo></mml:mrow></mml:math>
<!--</alternatives>--></disp-formula></p>
<p>Repeat several times until the maximum number of iterations <inline-formula id="ieqn-22">
<!--<alternatives><inline-graphic xlink:href="ieqn-22.png"/><tex-math id="tex-ieqn-22"><![CDATA[\alpha]]></tex-math>--><mml:math id="mml-ieqn-22"><mml:mi>&#x03B1;</mml:mi></mml:math>
<!--</alternatives>--></inline-formula> are reached,</p>
<p>Get the drosophila <inline-formula id="ieqn-23">
<!--<alternatives><inline-graphic xlink:href="ieqn-23.png"/><tex-math id="tex-ieqn-23"><![CDATA[BEST = \left[ {OO{B_{best}},T{N_{best}},F{T_{best}}} \right]]]></tex-math>--><mml:math id="mml-ieqn-23"><mml:mi>B</mml:mi><mml:mi>E</mml:mi><mml:mi>S</mml:mi><mml:mi>T</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mrow><mml:mo>[</mml:mo><mml:mrow><mml:mi>O</mml:mi><mml:mi>O</mml:mi><mml:mrow><mml:msub><mml:mi>B</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mi>T</mml:mi><mml:mrow><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mi>F</mml:mi><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>b</mml:mi><mml:mi>e</mml:mi><mml:mi>s</mml:mi><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow><mml:mo>]</mml:mo></mml:mrow></mml:math>
<!--</alternatives>--></inline-formula> with the highest odor concentration on the drosophila cluster. Use BEST to construct an optimal random forest.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Random Forest Model Based On Cloud Computing</title>
<p>On the built Hapood platform, use the MapReduce module to build a random forest model, the steps are as follows:<list list-type="alpha-lower"><list-item>
<p>The random subspace sampling method is used for sampling, multiple sets of characteristic attribute sets <inline-formula id="ieqn-24">
<!--<alternatives><inline-graphic xlink:href="ieqn-24.png"/><tex-math id="tex-ieqn-24"><![CDATA[A = \{ F{T_1},F{T_2}&#x2026;,F{T_N}\}]]></tex-math>--><mml:math id="mml-ieqn-24"><mml:mi>A</mml:mi><mml:mo>&#x003D;</mml:mo><mml:mo stretchy="false" fence="false">{</mml:mo><mml:mi>F</mml:mi><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mn>1</mml:mn></mml:msub></mml:mrow><mml:mo>,</mml:mo><mml:mi>F</mml:mi><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mn>2</mml:mn></mml:msub></mml:mrow><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>.</mml:mo><mml:mo>,</mml:mo><mml:mi>F</mml:mi><mml:mrow><mml:msub><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:msub></mml:mrow><mml:mo stretchy="false" fence="false">}</mml:mo></mml:math>
<!--</alternatives>--></inline-formula> are extracted, and the number of each group is fixed.</p></list-item><list-item>
<p>The data is distributed among each distributed computing node, and the information <italic>A</italic> of the characteristic attribute set is sent to each node, and each Map node gets a key/value pair.</p></list-item><list-item>
<p>Each Map node executes the program, reads and maps the information on the characteristic attribute set <italic>S</italic> of the mapper, and separately counts the branch information on different decision tree nodes, and obtains one or more key/value pairs after processing.</p></list-item><list-item>
<p>After all the Map work is completed, each Map node passes the information to the Reduce node, and the Reduce node merges the information, performs iterative calculations on the data with the same key, and calculates the nodes of the first layer of different decision trees.</p></list-item><list-item>
<p>Iteratively execute steps 2 to 4 until all decision trees are constructed.</p></list-item></list></p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Experiment Analysis</title>
<p>In order to verify the performance of the algorithm, this article compares and verifies with the random forest model under a single server. Collect the daily electricity consumption data of 200 households in a community in Shanghai from August 2019 to August 2020, and collect the daily electricity consumption of each user every hour. Smart electricity collection equipment is installed in the residential houses of the community, and the high-power consumption equipment in the home, such as washing machines, refrigerators, air conditioners, etc., is transmitted to the home smart gateway wirelessly (433 MHz) to complete the data collection task. According to the daily power consumption trend, users are divided into eight categories, category 1 is high power consumption users throughout the day, and category 2 is peak power consumption users from 6 am to 9 am. Category 3 is the peak power user from 9 am to 12 am, Category 3 is the peak power user from 12 noon to 3 pm, and Category 4 is the peak power user from 3 pm to 5 pm. Category 5 is peak electricity consumption users from 5 to 7 pm, category 6 is peak electricity consumption users from 8 pm to 10 pm, category 7 is peak electricity consumption users from 10 pm to 1 am, and category 8 is low electricity consumption throughout the day power users. Use 70% of the collected electricity consumption data as the training set, and the remaining 30% as the test set.</p>
<p>To verify the performance of the model, the classification accuracy of the model under different data sets was tested <xref ref-type="fig" rid="fig-3">Fig. 3</xref>. Data set <italic>A</italic> contains 3 types of electricity users, data set <italic>B</italic> contains 4 types of electricity users, data set <italic>C</italic> contains 5 types of electricity users, data set <italic>D</italic> contains 6 types of electricity users, and data set <italic>E</italic> contains 7 types of electricity users, Data set <italic>F</italic> contains 8 power users. It can be seen that as the data set contains more types, although the model classification accuracy rate has decreased, the overall classification remains above 90%.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Model classification accuracy on different data sets</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-3.png"/>
</fig>
<p>In order to verify the superiority of the model, the <italic>TN</italic> and <italic>FT</italic> of the traditional random forest are compared with the improved random forest proposed in this paper. As shown in the <xref ref-type="fig" rid="fig-4">Figs. 4</xref> and <xref ref-type="fig" rid="fig-5">5</xref>, it can be seen that the improved random forest model proposed in this paper is generally better than the traditional random forest in classification performance.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Change the classification accuracy of the decision tree number model</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-4.png"/>
</fig>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Change the classification accuracy of the attribute feature subset size model</title>
</caption><graphic mimetype="image" mime-subtype="png" xlink:href="fig-5.png"/>
</fig>
</sec>
<sec id="s5">
<label>5</label>
<title>Conclusions</title>
<p>In order to help the power supply side perform better resource management and distribution, it is necessary to classify the electricity consumption of residential users. Based on the distributed architecture of cloud computing, this paper uses an improved random forest model to design a residential electricity classification method. This method firstly uses the fruit fly algorithm to improve the random forest model, and then uses MapReduce to train the improved random forest model on the cloud computing platform. After that, the trained model is used to analyze the residential electricity consumption data set, and all residents are divided into 5 categories. And the effect of the model is verified through experiments.</p>
</sec>
</body>
<back>
<ack>
<p>This work was supported by the I6000 migration to the cloud micro-application pilot construction project of the Information and Communication Branch of State Grid Anhui Electric Power Co., Ltd., Technical project (contract number: SGAHXT00XYXX2000121).</p>
</ack><fn-group>
<fn fn-type="other">
<p><bold>Funding Statement:</bold> This work was partially supported by the National Natural Science Foundation of China (61876089).</p>
</fn>
<fn fn-type="conflict">
<p><bold>Conflicts of Interest:</bold> The authors declare that they have no conflicts of interest to report regarding the present study.</p>
</fn>
</fn-group>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1">
<label>[1]</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><string-name>
<given-names>C.</given-names> 
<surname>Song</surname></string-name>, <string-name>
<given-names>W.</given-names> 
<surname>Xu</surname></string-name>, <string-name>
<given-names>G.</given-names> 
<surname>Han</surname></string-name>, <string-name>
<given-names>P.</given-names> 
<surname>Zeng</surname></string-name>, <string-name>
<given-names>Z.</given-names> 
<surname>Wang</surname></string-name> <etal>et al.</etal>
</person-group><italic>,</italic> &#x201C;
<article-title>A cloud edge collaborative intelligence method of insulator string defect detection for power IIoT</article-title>,&#x201D; 
<source>IEEE Internet of Things Journal</source>, vol. 
<volume>2020</volume>, pp. 
<fpage>1</fpage>&#x2013;
<lpage>11</lpage>, 
<year iso-8601-date="2020">2020</year>.</mixed-citation>
</ref>
<ref id="ref-2">
<label>[2]</label><mixed-citation publication-type="conf-proc">
<person-group person-group-type="author"><string-name>
<given-names>B.</given-names> 
<surname>Xu</surname></string-name>, <string-name>
<given-names>Y.</given-names> 
<surname>Sun</surname></string-name>, <string-name>
<given-names>H.</given-names> 
<surname>Wang</surname></string-name> and <string-name>
<given-names>S.</given-names> 
<surname>Yi</surname></string-name>
</person-group>, &#x201C;
<article-title>Short-term electricity consumption forecasting method for residential users based on cluster classification and backpropagation neural network</article-title>,&#x201D; in <conf-name>2019 11th Int. Conf. on Intelligent Human-Machine Systems and Cybernetics (IHMSC)</conf-name>, 
<publisher-loc>Hangzhou, China</publisher-loc>, pp. 
<fpage>55</fpage>&#x2013;
<lpage>59</lpage>, 
<year iso-8601-date="2019">2019</year>. </mixed-citation>
</ref>
<ref id="ref-3">
<label>[3]</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><string-name>
<given-names>C.</given-names> 
<surname>Song</surname></string-name>, <string-name>
<given-names>W.</given-names> 
<surname>Xu</surname></string-name>, <string-name>
<given-names>Z.</given-names> 
<surname>Wang</surname></string-name>, <string-name>
<given-names>S.</given-names> 
<surname>Yu</surname></string-name>, <string-name>
<given-names>P.</given-names> 
<surname>Zeng</surname></string-name> <etal>et al.</etal>
</person-group><italic>,</italic> &#x201C;
<article-title>Analysis on the impact of data augmentation on target recognition for UAV-based transmission line inspection</article-title>,&#x201D; 
<source>Complexity</source>, vol. 
<volume>2020</volume>, pp. 
<fpage>1</fpage>&#x2013;
<lpage>11</lpage>, 
<year iso-8601-date="2020">2020</year>.</mixed-citation>
</ref>
<ref id="ref-4">
<label>[4]</label><mixed-citation publication-type="conf-proc">
<person-group person-group-type="author"><string-name>
<given-names>O. A.</given-names> 
<surname>Loktionov</surname></string-name>, <string-name>
<given-names>O. E.</given-names> 
<surname>Kondrateva</surname></string-name>, <string-name>
<given-names>N. V.</given-names> 
<surname>Zvonkova</surname></string-name> and <string-name>
<given-names>D. A.</given-names> 
<surname>Burdyukov</surname></string-name>
</person-group>, &#x201C;
<article-title>Seasonal decomposition application for the energy consumption analysis of cities</article-title>,&#x201D; in <conf-name>2019 Int. Youth Conf. on Radio Electronics, Electrical and Power Engineering (REEPE)</conf-name>, 
<publisher-loc>Moscow, Russia</publisher-loc>, pp. 
<fpage>1</fpage>&#x2013;
<lpage>4</lpage>, 
<year iso-8601-date="2019">2019</year>. </mixed-citation>
</ref>
<ref id="ref-5">
<label>[5]</label><mixed-citation publication-type="book">
<person-group person-group-type="author"><string-name>
<given-names>C.</given-names> 
<surname>Liu</surname></string-name>, <string-name>
<given-names>J.</given-names> 
<surname>Wang</surname></string-name>, <string-name>
<given-names>M.</given-names> 
<surname>Wu</surname></string-name>, <string-name>
<given-names>J.</given-names> 
<surname>Bai</surname></string-name>, <string-name>
<given-names>X.</given-names> 
<surname>Wang</surname></string-name> <etal>et al.</etal>
</person-group><italic>,</italic> &#x201C;<chapter-title>Research on the transformer area recognition method based on improved K-means clustering algorithm</chapter-title>,&#x201D; in 
<source>2019 IEEE Innovative Smart Grid Technologies&#x2014;Asia (ISGT Asia)</source>. 
<publisher-loc>Chengdu, China</publisher-loc>, pp. 
<fpage>4137</fpage>&#x2013;
<lpage>4141</lpage>, 
<year iso-8601-date="2019">2019</year>.</mixed-citation>
</ref>
<ref id="ref-6">
<label>[6]</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><string-name>
<given-names>J.</given-names> 
<surname>Zhang</surname></string-name>
</person-group>, &#x201C;
<article-title>Classification method of resident users based on load analysis</article-title>,&#x201D; 
<source>Industrial Control Computer</source>, vol. 
<volume>33</volume>, no. 
<issue>5</issue>, pp. 
<fpage>142</fpage>&#x2013;
<lpage>144</lpage>, 
<year iso-8601-date="2020">2020</year>.</mixed-citation>
</ref>
<ref id="ref-7">
<label>[7]</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><string-name>
<given-names>C.</given-names> 
<surname>Song</surname></string-name>, <string-name>
<given-names>W.</given-names> 
<surname>Jing</surname></string-name>, <string-name>
<given-names>P.</given-names> 
<surname>Zeng</surname></string-name> and <string-name>
<given-names>C.</given-names> 
<surname>Rosenberg</surname></string-name>
</person-group>, &#x201C;
<article-title>An analysis on the energy consumption of circulating pumps of residential swimming pools for peak load management</article-title>,&#x201D; 
<source>Applied Energy</source>, vol. 
<volume>195</volume>, no. 
<issue>3</issue>, pp. 
<fpage>1</fpage>&#x2013;
<lpage>12</lpage>, 
<year iso-8601-date="2017">2017</year>.</mixed-citation>
</ref>
<ref id="ref-8">
<label>[8]</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><string-name>
<given-names>C.</given-names> 
<surname>Song</surname></string-name>, <string-name>
<given-names>W.</given-names> 
<surname>Jing</surname></string-name>, <string-name>
<given-names>P.</given-names> 
<surname>Zeng</surname></string-name>, <string-name>
<given-names>H.</given-names> 
<surname>Yu</surname></string-name> and <string-name>
<given-names>C.</given-names> 
<surname>Rosenberg</surname></string-name>
</person-group>, &#x201C;
<article-title>Energy consumption analysis of residential swimming pools for peak load shaving</article-title>,&#x201D; 
<source>Applied Energy</source>, vol. 
<volume>220</volume>, no. 
<issue>Part 3</issue>, pp. 
<fpage>176</fpage>&#x2013;
<lpage>191</lpage>, 
<year iso-8601-date="2018">2018</year>.</mixed-citation>
</ref>
<ref id="ref-9">
<label>[9]</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><string-name>
<given-names>H.</given-names> 
<surname>Zhang</surname></string-name>, <string-name>
<given-names>G.</given-names> 
<surname>Chen</surname></string-name> and <string-name>
<given-names>X.</given-names> 
<surname>Li</surname></string-name>
</person-group>, &#x201C;
<article-title>Resource management in cloud computing with optimal pricing policies</article-title>,&#x201D; 
<source>Computer Systems Science and Engineering</source>, vol. 
<volume>34</volume>, no. 
<issue>4</issue>, pp. 
<fpage>249</fpage>&#x2013;
<lpage>254</lpage>, 
<year iso-8601-date="2019">2019</year>.</mixed-citation>
</ref>
<ref id="ref-10">
<label>[10]</label><mixed-citation publication-type="conf-proc">
<person-group person-group-type="author"><string-name>
<given-names>G. S.</given-names>
<surname> Bhathal</surname> </string-name> and <string-name>
<given-names>A. S.</given-names> 
<surname>Dhiman</surname></string-name>
</person-group>, &#x201C;
<article-title>Big data solution: improvised distributions framework of Hadoop</article-title>,&#x201D; in <conf-name>2018 Second Int. Conf. on Intelligent Computing and Control Systems (ICICCS)</conf-name>, 
<publisher-loc>Madurai, India</publisher-loc>, pp. 
<fpage>35</fpage>&#x2013;
<lpage>38</lpage>, 
<year iso-8601-date="2018">2018</year>. </mixed-citation>
</ref>
<ref id="ref-11">
<label>[11]</label><mixed-citation publication-type="conf-proc">
<person-group person-group-type="author"><string-name>
<given-names>V.</given-names> 
<surname>Sontakke</surname></string-name> and <string-name>
<given-names>R. B.</given-names> 
<surname>Dayanand</surname></string-name>
</person-group>, &#x201C;
<article-title>Optimization of Hadoop MapReduce model in cloud computing environment</article-title>,&#x201D; in <conf-name>2019 Int. Conf. on Smart Systems and Inventive Technology (ICSSIT)</conf-name>, 
<publisher-loc>Tirunelveli, India</publisher-loc>, pp. 
<fpage>510</fpage>&#x2013;
<lpage>515</lpage>, 
<year iso-8601-date="2019">2019</year>. </mixed-citation>
</ref>
<ref id="ref-12">
<label>[12]</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><string-name>
<given-names>T.</given-names> 
<surname>Wilcox</surname></string-name> and <string-name>
<given-names>J.</given-names> 
<surname>Nanlin</surname></string-name>
</person-group>, &#x201C;
<article-title>A Big Data platform for smart meter data analytics</article-title>,&#x201D; 
<source>Computers in Industry</source>, vol. 
<volume>105</volume>, pp. 
<fpage>250</fpage>&#x2013;
<lpage>259</lpage>, 
<year iso-8601-date="2019">2019</year>.</mixed-citation>
</ref>
<ref id="ref-13">
<label>[13]</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><string-name>
<given-names>Y.</given-names> 
<surname>Yang</surname></string-name>, <string-name>
<given-names>P.</given-names> 
<surname>Fu</surname></string-name>, <string-name>
<given-names>X.</given-names> 
<surname>Yang</surname></string-name>, <string-name>
<given-names>H.</given-names> 
<surname>Hong</surname></string-name> and <string-name>
<given-names>D.</given-names> 
<surname>Zhou</surname></string-name>
</person-group>, &#x201C;
<article-title>MOOC learner&#x2019;s final grade prediction based on an improved random forests method</article-title>,&#x201D; 
<source>Computers Materials &#x0026; Continua</source>, vol. 
<volume>65</volume>, no. 
<issue>3</issue>, pp. 
<fpage>2413</fpage>&#x2013;
<lpage>2423</lpage>, 
<year iso-8601-date="2020">2020</year>.</mixed-citation>
</ref>
<ref id="ref-14">
<label>[14]</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><string-name>
<given-names>X. L.</given-names> 
<surname>Wei</surname></string-name>, <string-name>
<given-names>J. W.</given-names> 
<surname>Liu</surname></string-name>, <string-name>
<given-names>Y. A.</given-names> 
<surname>Wang</surname></string-name>, <string-name>
<given-names>C. G.</given-names> 
<surname>Tang</surname></string-name> and <string-name>
<given-names>Y. Y.</given-names> 
<surname>Hu</surname></string-name>
</person-group>, &#x201C;
<article-title>Wireless edge caching based on content similarity in dynamic environments</article-title>,&#x201D; 
<source>Journal of Systems Architecture</source>, vol. 
<volume>115</volume>, no. 
<issue>2</issue>, pp. 
<fpage>102000</fpage>, 
<year iso-8601-date="2021">2021</year>.</mixed-citation>
</ref>
<ref id="ref-15">
<label>[15]</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><string-name>
<given-names>F.</given-names> 
<surname>Bi</surname></string-name>, <string-name>
<given-names>X.</given-names> 
<surname>Fu1</surname></string-name>, <string-name>
<given-names>W.</given-names> 
<surname>Chen</surname></string-name>, <string-name>
<given-names>W.</given-names> 
<surname>Fang</surname></string-name>, <string-name>
<given-names>X.</given-names> 
<surname>Miao</surname></string-name> <etal>et al.</etal>
</person-group><italic>,</italic> &#x201C;
<article-title>Fire detection method based on improved fruit fly optimization-based SVM</article-title>,&#x201D; 
<source>Computers Materials &#x0026; Continua</source>, vol. 
<volume>62</volume>, no. 
<issue>1</issue>, pp. 
<fpage>199</fpage>&#x2013;
<lpage>216</lpage>, 
<year iso-8601-date="2020">2020</year>.</mixed-citation>
</ref>
<ref id="ref-16">
<label>[16]</label><mixed-citation publication-type="journal">
<person-group person-group-type="author"><string-name>
<given-names>A.</given-names> 
<surname>Feng</surname></string-name>, <string-name>
<given-names>Z.</given-names> 
<surname>Gao</surname></string-name>, <string-name>
<given-names>X.</given-names> 
<surname>Song</surname></string-name>, <string-name>
<given-names>K.</given-names> 
<surname>Ke</surname></string-name>, <string-name>
<given-names>T.</given-names> 
<surname>Xu</surname></string-name> <etal>et al.</etal>
</person-group><italic>,</italic> &#x201C;
<article-title>Modeling multi-targets sentiment classification via graph convolutional networks and auxiliary relation</article-title>,&#x201D; 
<source>Computers Materials &#x0026; Continua</source>, vol. 
<volume>64</volume>, no. 
<issue>2</issue>, pp. 
<fpage>909</fpage>&#x2013;
<lpage>923</lpage>, 
<year iso-8601-date="2020">2020</year>.</mixed-citation>
</ref>
</ref-list>
</back>
</article>