<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">32864</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2023.032864</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>Automated File Labeling for Heterogeneous Files Organization Using Machine Learning</article-title>
<alt-title alt-title-type="left-running-head">Automated File Labeling for Heterogeneous Files Organization Using Machine Learning</alt-title>
<alt-title alt-title-type="right-running-head">Automated File Labeling for Heterogeneous Files Organization Using Machine Learning</alt-title>
</title-group>
<contrib-group content-type="authors">
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Abbas</surname><given-names>Sagheer</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author">
<name name-style="western"><surname>Raza</surname><given-names>Syed Ali</given-names></name><xref ref-type="aff" rid="aff-1">1</xref>
<xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Khan</surname><given-names>M. A.</given-names></name><xref ref-type="aff" rid="aff-3">3</xref></contrib>
<contrib id="author-4" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Khan</surname><given-names>Muhammad Adnan</given-names></name><xref ref-type="aff" rid="aff-4">4</xref><email>adnan@gachon.ac.kr</email></contrib>
<contrib id="author-5" contrib-type="author"><name name-style="western"><surname>Atta-ur-Rahman</surname></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>Sultan</surname><given-names>Kiran</given-names></name><xref ref-type="aff" rid="aff-6">6</xref></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Mosavi</surname><given-names>Amir</given-names></name><xref ref-type="aff" rid="aff-7">7</xref>
<xref ref-type="aff" rid="aff-8">8</xref>
<xref ref-type="aff" rid="aff-9">9</xref></contrib>
<aff id="aff-1"><label>1</label><institution>School of Computer Science, National College of Business Administration &#x0026; Economics</institution>, <addr-line>Lahore, 54000</addr-line>, <country>Pakistan</country></aff>
<aff id="aff-2"><label>2</label><institution>Department of Computer Science, GC University Lahore</institution>, <country>Pakistan</country></aff>
<aff id="aff-3"><label>3</label><institution>Riphah School of Computing &#x0026; Innovation, Faculty of Computing, Riphah International University, Lahore Campus</institution>, <addr-line>Lahore, 54000</addr-line>, <country>Pakistan</country></aff>
<aff id="aff-4"><label>4</label><institution>Department of Software, Pattern Recognition and Machine Learning Lab, Gachon University</institution>, <addr-line>Seongnam, 13120</addr-line>, <country>Korea</country></aff>
<aff id="aff-5"><label>5</label><institution>Department of Computer Science, College of Computer Science and Information Technology (CCSIT), Imam Abdulrahman Bin Faisal University (IAU)</institution>, <addr-line>P.O. Box 1982, Dammam, 31441</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-6"><label>6</label><institution>Department of CIT, The Applied College, King Abdulaziz University</institution>, <addr-line>Jeddah, 31261</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-7"><label>7</label><institution>John von Neumann Faculty of Informatics, Obuda University</institution>, <addr-line>Budapest, 1034</addr-line>, <country>Hungary</country></aff>
<aff id="aff-8"><label>8</label><institution>Institute of Information Engineering, Automation and Mathematics, Slovak University of Technology in Bratislava</institution>, <addr-line>Bratislava, 81107</addr-line>, <country>Slovakia</country></aff>
<aff id="aff-9"><label>9</label><institution>Faculty of Civil Engineering, TU-Dresden</institution>, <addr-line>Dresden, 01062</addr-line>, <country>Germany</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Muhammad Adnan Khan. Email: <email>adnan@gachon.ac.kr</email></corresp>
</author-notes>
<pub-date pub-type="epub" date-type="pub" iso-8601-date="2022-10-28"><day>28</day>
<month>10</month>
<year>2022</year></pub-date>
<volume>74</volume>
<issue>2</issue>
<fpage>3263</fpage>
<lpage>3278</lpage>
<history>
<date date-type="received"><day>31</day><month>5</month><year>2022</year></date>
<date date-type="accepted"><day>05</day><month>7</month><year>2022</year></date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2023 Abbas et al.</copyright-statement>
<copyright-year>2023</copyright-year>
<copyright-holder>Abbas et al.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_32864.pdf"></self-uri>
<abstract>
<p>File labeling techniques have a long history in analyzing the anthological trends in computational linguistics. The situation becomes worse in the case of files downloaded into systems from the Internet. Currently, most users either have to change file names manually or leave a meaningless name of the files, which increases the time to search required files and results in redundancy and duplications of user files. Currently, no significant work is done on automated file labeling during the organization of heterogeneous user files. A few attempts have been made in topic modeling. However, one major drawback of current topic modeling approaches is better results. They rely on specific language types and domain similarity of the data. In this research, machine learning approaches have been employed to analyze and extract the information from heterogeneous corpus. A different file labeling technique has also been used to get the meaningful and &#x0060;cohesive topic of the files. The results show that the proposed methodology can generate relevant and context-sensitive names for heterogeneous data files and provide additional insight into automated file labeling in operating systems.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Automated file labeling</kwd>
<kwd>file organization</kwd>
<kwd>machine learning</kwd>
<kwd>topic modeling</kwd>
</kwd-group>
</article-meta>
</front>
<body>
<sec id="s1"><label>1</label><title>Introduction</title>
<p>Whether physical or digital, data organization is a critical part of our routine matters. The growing amount of data in modern working environments demands an effective system to manage files and folders. The file organization and management have several aspects, including labeling and naming digital files. Names and labels play a significant part in organizing the digital files stored in thousands inside computer systems. The number of digital files stored in any computer system is aggregating rapidly with the progress in the development of mass secondary storage devices [<xref ref-type="bibr" rid="ref-1">1</xref>]. With this increasing number of files, most users find it difficult to search and access any specific file. Computer users usually spend much time every day interacting with digital files and folders stored in their systems. These interactions consist of several actions, including creating, downloading, labeling, reviewing, navigating, searching for, moving, saving, copying, sharing, and deleting digital files. In addition, files that contain heterogeneous data, such as numbers, images, or sounds, are becoming popular. Users usually tend to demonstrate substantial creativity in file labeling [<xref ref-type="bibr" rid="ref-2">2</xref>]. However, the file labeling patterns are recognizable such as files being named to display the file they represent, their purpose, or a relevant creation date or deadline [<xref ref-type="bibr" rid="ref-3">3</xref>], but may also contain characters to expedite the sorting of the files to reduce clutter. Commonly, it is believed that a balanced and expressive file name may oblige multiple tenacities, including labeling, information organization, decreased searching time, and increased readability [<xref ref-type="bibr" rid="ref-4">4</xref>].</p>
<p>Filename generation is typically an unsupervised machine learning technique through which the hidden semantic information can be extracted from the corpus [<xref ref-type="bibr" rid="ref-5">5</xref>]. The corpus used in file labeling can typically be made up of thousands of files in the form of a data set, CSV file, and text file. Another essential aspect of this is that sometimes the user files require more than one topic or include many words. No hard and fast rule can be applied to get highly accurate and meaningful data in file labeling [<xref ref-type="bibr" rid="ref-6">6</xref>]. Therefore, file labeling relies on linguistic analysis and preprocessing techniques to filter the data to get reasonably meaningful results [<xref ref-type="bibr" rid="ref-1">1</xref>].</p>
<p>One of the significant issues in file labeling is analyzing the type of content and language used in the file. The data provided for file labeling first needs to go through a rigorous preprocessing stage to remove insignificant content that does not play any role in file labeling. Once effective content is extracted, the hidden feature is analyzed to compute the preprocessed file&#x2019;s dimensionality, structure, and size. The semantic information obtained can be further converted into one hot vector for the unique representation of each word in files because files may show many Labels, and these Labels can also overlap with each other. File labeling is also used to break down the file into different Labels based on the probability distribution. In this research, different techniques have been used, such as Latent Dirichlet Allocation (LDA) [<xref ref-type="bibr" rid="ref-7">7</xref>], Latent Semantic Analysis (LSA) [<xref ref-type="bibr" rid="ref-8">8</xref>], and Non-negative Matrix Factorization (NMF) [<xref ref-type="bibr" rid="ref-9">9</xref>] for automated file labeling. This research aims to incorporate computational linguistic analysis based on machine learning techniques. It is hypothesized that an automated system can be developed to generate contextually meaningful labels for user files to assist users in the overall management and organization of digital files inside any computer system.</p>
</sec>
<sec id="s2"><label>2</label><title>Related Work</title>
<p>Previously, several attempts have been made in topic modeling and labeling short text and articles. Existing topic modeling tasks are mainly done using specific datasets such as fake news, ABC news dataset, New York times dataset, Twitter dataset, etc. Considering the monotony in these datasets, existing labeling and topic modeling approaches perform very well [<xref ref-type="bibr" rid="ref-10">10</xref>]. Daniel&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-11">11</xref>] studied biological science-related data and analyzed that standard preprocessing approaches are not good enough to apprehend such diversified documents. They [<xref ref-type="bibr" rid="ref-11">11</xref>] developed a standalone toolbox and extracted common daily life words through graphical representation and heat-map and graphical representation based on topic similarity. Rubayyi&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-12">12</xref>] used different techniques such as LDA, LSA, and Probabilistic Latent Semantic Analysis (PLSA). They identified probability-based topics over a dataset of hundred documents based on frequent keywords.</p>
<p>Pantel&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-13">13</xref>] proposed using a lexical-semantic pattern for labeling semantic classes, encompassing all class members to learn a different label. A significant limitation of this task was that it only worked well over the semantically homogeneous, fine-grained clusters. Alokaili&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-14">14</xref>] proposed a neural approach using the sequence-to-sequence technique for document labeling that does not suffer from this limitation. The model was trained using distant supervision over a sizeable synthetic dataset. Human experts evaluated the model by comparing labels to ones generated by the proposed technique. Initially, Seung&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-9">9</xref>] conducted [<xref ref-type="bibr" rid="ref-15">15</xref>] human scoring of topics. Humans evaluated topics used in a novel and directly scored different topics learned by a topic model based upon pointwise mutual information. Qiaozhu&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-16">16</xref>] used unsupervised machine learning approaches for automatically labeling topics. They generated expected labels through bigrams and noun chunks and afterward ranked those expected labels based on divergence with the preselected topic. Another approach that many researchers have used is to match topic words to concepts based on knowledge base [<xref ref-type="bibr" rid="ref-17">17</xref>,<xref ref-type="bibr" rid="ref-18">18</xref>]. Jey&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-19">19</xref>] extracted top N terms to select topics, while some others have used summarization approaches to create labels for topics [<xref ref-type="bibr" rid="ref-20">20</xref>,<xref ref-type="bibr" rid="ref-21">21</xref>].</p>
<p>One common drawback of existing topic modeling approaches is the absence of interpretable topic space [<xref ref-type="bibr" rid="ref-20">20</xref>]. LDA [<xref ref-type="bibr" rid="ref-7">7</xref>,<xref ref-type="bibr" rid="ref-22">22</xref>] and PLSA [<xref ref-type="bibr" rid="ref-8">8</xref>] are traditional topic modeling techniques and envision different documents as a concoction of some discretized topics on some fixed parameter. Angelov [<xref ref-type="bibr" rid="ref-23">23</xref>] presented the Top2Vec approach based on Word2Vec and Doc2Vec models to build a document and topic vector and an interpretable word space. Another similar approach is BERTopic [<xref ref-type="bibr" rid="ref-24">24</xref>] which separates the embedding stage from the topic creation stage in contrast to the Top2Vec approach. Top2Vec considers words adjacent to the centroid of a cluster and creates coherent and interpretable topic representations very nicely.</p>
<p>On the other hand, BERTopic [<xref ref-type="bibr" rid="ref-24">24</xref>] focuses on the cluster and attempts to model the topic representation from the entire cluster. This allows the topic representations to be a bit more diverse and disregards the notion of centroids. One major limitation of document embedding techniques based on training is that they do not ensure quality on heterogeneous datasets. The model can only learn semantic associations like notions and thoughts between words and sentences but cannot comprehend the central idea of the document [<xref ref-type="bibr" rid="ref-25">25</xref>].</p>
</sec>
<sec id="s3"><label>3</label><title>Materials and Methods</title>
<p>The proposed model consists of the following modules: preprocessing the data, Document matrix (DM) and Term analysis module (TAM), Topic modeling module (TMM), and File name generation module (FNGM) as shown in <xref ref-type="fig" rid="fig-1">Fig. 1</xref>. The dataset used in this research is an amalgam of ABC news data set, different research articles, and books. The reason for collecting various datasets is to increase heterogeneity in the data used for training the system.</p>
<fig id="fig-1"><label>Figure 1</label><caption><title>Proposed model for automated labeling of heterogeneous files</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_32864-fig-1.png"/></fig>
<sec id="s3_1"><label>3.1</label><title>Preprocessing Module</title>
<p>Humans have the senses to understand rich knowledge about the language, the hidden meanings, and the facts behind words used in languages. On the contrary, operating systems do not possess any such information. Most of the files in the computer system are usually classified as unstructured documents; therefore, almost all of the files must be preprocessed before file labeling begins. The preprocessing of the files includes multiple steps such as Filtering, Stop Word Removal, Tokenization, and Stemming of document words as shown in <xref ref-type="fig" rid="fig-2">Fig. 2</xref>. This module&#x2019;s main idea is to convert digital files into strings. Therefore a pipeline process is required that can make sentences like &#x201C;The man stands in front of the dog&#x201D; and turn them into a string of words: &#x201C;the,&#x201D; &#x201C;man,&#x201D; &#x201C;stand,&#x201D; &#x201C;in,&#x201D; and &#x201C;front,&#x201D; &#x201C;of,&#x201D;&#x00A0;&#x201C;dog.&#x0022;Afterward, these words are translated into a numeric representation to make them machine-readable. Once the stop words like in, of, the, etc. are removed, the next step is to tokenize these sentences into chunks of individual words. After tokenization, the chunks of words are transformed into their stem to reduce each word to its basic form, such as energies, and energy is changed to energy. The last preprocessing step is to recognize named entities in the text. Named entity recognition categorizes the text with tags of relevant named entities such as people, location, association, phone numbers, email, etc. This module helps identify important names used as possible file labels.</p>
<fig id="fig-2"><label>Figure 2</label><caption><title>Preprocessing pipeline for proposed automated file labelling of heterogeneous files</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_32864-fig-2.png"/></fig>
</sec>
<sec id="s3_2"><label>3.2</label><title>Document Matrix and Term Analysis (DM-TA) Module</title>
<p>The DM-TA module consists of different submodules. The purpose of these sub-modules is to represent each document better. For this purpose, Term frequency-inverse document frequency (TF-IDF), Word Embeddings, and Word2Doc Models have been used.</p>
<sec id="s3_2_1"><label>3.2.1</label><title>TF-IDF</title>
<p>Term Frequency over inverse document frequency [<xref ref-type="bibr" rid="ref-26">26</xref>] is a typical statistical method used by Natural language processing experts. In standard systems to convert text documents into the matrix representations of the feature vectors. TF-IDF scores represent the value or the relevance of the respective terms in the given set of documents. TF-IDF is computationally dependent on the vocabulary set and thus fails in the environment where there is a constant and frequent change in the text corpora.</p>
</sec>
<sec id="s3_2_2"><label>3.2.2</label><title>Word Embedding</title>
<p>Word Embeddings represent a word in a high dimensional context. It may appear in a vocabulary employing a real value vector. Since word embedding preserves the real contextual purpose of the word it is used for, it yields better results in several tasks, including but not limited to similarity analysis and label extraction.</p>
</sec>
<sec id="s3_2_3"><label>3.2.3</label><title>Word2Vec</title>
<p>Word2Vec [<xref ref-type="bibr" rid="ref-27">27</xref>] uses an external neural network trained on an extensive data set and represents a word as a vector in vector space where its locations represent the meaning it may be perceived. Word closers and clustering tend to have similar syntactic and semantic meanings. Word2Vec works excellently to predict the similarity of two words in their syntactical and semantic capacity but still cannot predict words in their contextual space.</p>
</sec>
<sec id="s3_2_4"><label>3.2.4</label><title>Term Analysis</title>
<p>Once significant tokens are extracted through TF-IDF, word embedding, and the Word2Vec approach, these tokens are represented through Doc2Vec for term analysis. Doc2Vec is an interactive and user-friendly package to learn the vector representation of words described in [<xref ref-type="bibr" rid="ref-11">11</xref>] in Word2Vec. Each of these vectors is stored as columns of a matrix W. The sum of the ordered words in vocabulary is used as a feature to predict the next word as a classifier. The goal is to maximize average log probability. The prediction is made using softmax for a multiclass classifier where each of y(i) is a non-normalized log probability for each output word i computed using <xref ref-type="disp-formula" rid="eqn-1">Eq. (1)</xref>
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mn>1</mml:mn><mml:mi>T</mml:mi></mml:mfrac></mml:mstyle><mml:msubsup><mml:mrow><mml:mo>&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>t</mml:mi><mml:mo>=</mml:mo><mml:mi>k</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mi>log</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo fence="false" stretchy="false">|</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>where the log probability p is calculated in <xref ref-type="disp-formula" rid="eqn-2">Eq. (2)</xref>
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:mi>p</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo fence="false" stretchy="false">|</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mrow><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:msup><mml:mrow><mml:munder><mml:mo>&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:munder><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mrow><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mrow></mml:msup></mml:mrow></mml:mfrac></mml:mstyle></mml:math></disp-formula></p>
<p>Finally, the y is computed as a sum of b and Uh where b &#x0026; U are softmax parameters and h is the resultant of the average or concatenation of the word vectors as mentioned in <xref ref-type="disp-formula" rid="eqn-3">Eq. (3)</xref>.
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mi>b</mml:mi><mml:mo>+</mml:mo><mml:mi>U</mml:mi><mml:mi>h</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mo>&#x2026;</mml:mo><mml:mo>,</mml:mo><mml:msub><mml:mi>w</mml:mi><mml:mrow><mml:mi>t</mml:mi><mml:mo>+</mml:mo><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>;</mml:mo><mml:mi>W</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>The neural language model is trained to predict word vectors. After the training, the words in the vector space having similar meanings are closer to each other than the contextually far apart words. These vectors can be used as input to an unsupervised clustering algorithm, among others, to assign them to individual groups based on the measure of their similarity as represented by various factors in their vector positioning in the Euclidean space.</p>
</sec>
</sec>
<sec id="s3_3"><label>3.3</label><title>Topic Modeling</title>
<p>Once significant terms are identified and represented through Doc2Vec [<xref ref-type="bibr" rid="ref-28">28</xref>], the next step is to summarize these variant terms into a simple, descriptive name for a given file. After analyzing the most common topic words, this process is conducted by most experts. However, recent attempts have been made to integrate supervised mechanisms so that labels can be determined in advance to match with learned labels [<xref ref-type="bibr" rid="ref-29">29</xref>]. In this research, supervision cannot be incorporated as the aim is to develop a fully functional automated approach that can assign names to user files independently. For this purpose, LDA [<xref ref-type="bibr" rid="ref-7">7</xref>], NMF [<xref ref-type="bibr" rid="ref-9">9</xref>], and LSA [<xref ref-type="bibr" rid="ref-30">30</xref>] techniques are applied to model labels for any user file.</p>
<sec id="s3_3_1"><label>3.3.1</label><title>LDA</title>
<p>The LDA approach can represent documents as random mixtures over hidden topics, where every topic can be categorized by a distribution of words, as shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>.</p>
<fig id="fig-3"><label>Figure 3</label><caption><title>LDA model for topic modeling</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_32864-fig-3.png"/></fig>
<p>The probability of a corpus (F) as mentioned in <xref ref-type="disp-formula" rid="eqn-4">Eq. (4)</xref> [<xref ref-type="bibr" rid="ref-27">27</xref>].
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>F</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>&#x03B1;</mml:mi><mml:mo>,</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>k</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>K</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x222B;</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>&#x03C9;</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>&#x03B1;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mo>(</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x220F;</mml:mo></mml:mrow><mml:mrow><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msubsup><mml:msub><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>k</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>k</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo fence="false" stretchy="false">|</mml:mo><mml:msub><mml:mi>&#x03C9;</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>W</mml:mi><mml:mrow><mml:mi>k</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo fence="false" stretchy="false">|</mml:mo><mml:msub><mml:mi>L</mml:mi><mml:mrow><mml:mi>k</mml:mi><mml:mi>l</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>&#x03B2;</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mi>k</mml:mi><mml:msub><mml:mi>&#x03C9;</mml:mi><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msub></mml:math></disp-formula>where &#x03B1; and &#x03B2; are the parameters of the Dirichlet before the per-file label and the per-label word distribution, respectively, the &#x03C9; represents each word, and L represents labels. At the same time, k and l are the range of words and labels. The merging distribution is acquired for a single file F by integrating &#x03C9; with the sum of labels L. According to the Bayes theorem, if the likelihood is a multinomial distribution of labels L and weight W (<italic>L<sub>kl</sub>, W<sub>kl</sub></italic>) and the prior probability is Dirichlet distributed over <bold>&#x03B1;</bold> and <bold><roman>&#x03B2;</roman></bold>, the posterior Dirichlet distribution can be easily obtained.</p>
</sec>
<sec id="s3_3_2"><label>3.3.2</label><title>NMF</title>
<p>One common problem with textual analysis is the curse of dimensionality. Considering many files with thousands of words having dozens of significant tokens can lead to a very complex situation and require unique mechanisms to deal with these many dimensions. NMF is one such approach that can reduce the dimensions of the data. NMF offers relatively more minor weightage to the tokens with a smaller coherence based on the factor analysis method. Suppose we have an input matrix F&#x00A0;of k&#x00A0;&#x00D7;&#x00A0;l&#x00A0;dimension, and we factorize this matrix F into two M &#x0026; N matrices with dimension of k&#x00A0;&#x00D7;&#x00A0;v and l&#x00A0;&#x00D7;&#x00A0;w respectively. Here each column of M characterizes the weightage of every word in a sentence, while each row of N represents the word embeddings. Nevertheless, it is considered that the entries of M and N are positive. The matrices M and N will be calculated and updated iteratively until convergence over the objective function as calculated in <xref ref-type="disp-formula" rid="eqn-5">Eq. (5)</xref> [<xref ref-type="bibr" rid="ref-9">9</xref>].
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac></mml:mstyle><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>F</mml:mi><mml:mo>&#x2212;</mml:mo><mml:mi>M</mml:mi><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mo fence="false" stretchy="false">|</mml:mo><mml:msubsup><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mrow><mml:mi>A</mml:mi></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msubsup><mml:mo>=</mml:mo><mml:msubsup><mml:mrow><mml:mo>&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>l</mml:mi></mml:mrow></mml:msubsup><mml:msubsup><mml:mrow><mml:mo>&#x2211;</mml:mo></mml:mrow><mml:mrow><mml:mi>j</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:msub><mml:mi>F</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>M</mml:mi><mml:mi>N</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:math></disp-formula></p>
<p>The rules for updating M and N can be derived using the objective functions calculated in <xref ref-type="disp-formula" rid="eqn-6">Eqs. (6)</xref> and <xref ref-type="disp-formula" rid="eqn-7">(7)</xref> [<xref ref-type="bibr" rid="ref-9">9</xref>].
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">&#x2190;</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>i</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>M</mml:mi><mml:mi>M</mml:mi><mml:mi>N</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>i</mml:mi><mml:mi>c</mml:mi></mml:mrow></mml:msub></mml:mfrac></mml:mstyle></mml:math></disp-formula>
<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">&#x2190;</mml:mo><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>M</mml:mi><mml:mi>F</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mrow><mml:mo>(</mml:mo><mml:mi>M</mml:mi><mml:mi>M</mml:mi><mml:mi>N</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mrow><mml:mi>c</mml:mi><mml:mi>j</mml:mi></mml:mrow></mml:msub></mml:mfrac></mml:mstyle></mml:math></disp-formula></p>
</sec>
<sec id="s3_3_3"><label>3.3.3</label><title>LSA</title>
<p>The basic theme behind LSA approach is to break down the matrix of files into two standalone matrices, namely File label matrix (FLM) and Label term matrix (LTM). Assumed that we have k number of files and l number of tokens, we created the k&#x00A0;&#x00D7;&#x00A0;l matrix where every row characterizes a file, and every column characterizes a term. To form LTM, LSA counts the number of times any word appeared in any file using TFIDF. One issue with the LTM is that this matrix is always scarce, noisy, and redundant across almost all dimensions. Therefore we need to apply dimensionality reduction on LTM to catch some latent labels that can characterize the connotation between terms and files. Truncated Singular Value Decomposition (TSVD) [<xref ref-type="bibr" rid="ref-31">31</xref>] is applied to reduce dimensionality. TSVD is a popular approach to factorize any matrix into the product of three different matrices such that A &#x003D; S &#x00D7; M &#x00D7; N where S contains the singular values of A and M, N are the factors of A. To reduce the dimensions using TSVD, we need the value of t, a hyperparameter that we can set independently. The probabilistic value of A can be estimated as mentioned in <xref ref-type="disp-formula" rid="eqn-8">Eq. (8)</xref> [<xref ref-type="bibr" rid="ref-30">30</xref>].
<disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:mi>A</mml:mi><mml:mo>&#x2248;</mml:mo><mml:msub><mml:mi>M</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>S</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:msubsup><mml:mrow><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>t</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi></mml:mrow></mml:msubsup></mml:math></disp-formula></p>
<p>A single label yields a single word, while different words must be incorporated to generate multiple labels. Therefore each file is reduced to a probability distribution over a set of labels. LDA defines the joint probability P of any file F with a word W as mentioned in <xref ref-type="disp-formula" rid="eqn-9">Eqs. (9)</xref> and <xref ref-type="disp-formula" rid="eqn-10">(10)</xref> [<xref ref-type="bibr" rid="ref-31">31</xref>].
<disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>F</mml:mi><mml:mo>,</mml:mo><mml:mi>W</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>D</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:msub><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>z</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>F</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>W</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>F</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>
<disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>F</mml:mi><mml:mo>,</mml:mo><mml:mi>W</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:msub><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>z</mml:mi></mml:mrow></mml:msub><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>z</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>F</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>z</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mi>P</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>W</mml:mi><mml:mo fence="false" stretchy="false">|</mml:mo><mml:mi>z</mml:mi><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>where P (F, W) defines the multinomial distributions that may be trained to deduce parameter estimates that are dependent on overlooked.</p>
</sec>
</sec>
<sec id="s3_4"><label>3.4</label><title>Label Generation</title>
<p>Once LDA, NFM, and LSA were applied, and files were transformed into n-dimension file embeddings, these embeddings were clustered into file labels. As these embeddings were large, BERT was applied on each label cluster to extract significant terms for the final label of each file. This step is crucial as topic modeling approaches are based on statistical methods, and there is a high probability that these approaches can extract non-significant words from the files. Each extracted label represents the actual file, but the problem with these labels is that most have different and distinct words. BERT is applied to rewrite these labels into more meaningful file names to generate a more meaningful label. BERT is a pre-trained deep neural network that uses multiple transformer layers and generates a language model effectively. Another significant aspect of BERT is that it can extract multiple word embeddings from a file based on its context. In this research, BERT is used as a sentence transformer. The results generated from LDA, NFM, and LSA had different, distinct words mentioned in <xref ref-type="fig" rid="fig-4">Fig. 4</xref> and required transformation into a helpful file label.</p>
<fig id="fig-4"><label>Figure 4</label><caption><title>Distinct labels generated based on topic modeling module of proposed system</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_32864-fig-4.png"/></fig>
</sec>
</sec>
<sec id="s4"><label>4</label><title>Results and Discussion</title>
<p>Automated file labeling is an exciting and challenging task simultaneously, as it requires dealing with heterogeneous data with varying amounts of words and tokens in different files. After the preprocessing step, the number of most frequent words was observed, as shown in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>.</p>
<fig id="fig-5"><label>Figure 5</label><caption><title>Word frequencies after preprocessing step of proposed system for automated file labeling</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_32864-fig-5.png"/></fig>
<p>Based on this frequency of words, the TF-IDF extracted terms from files are mentioned in <xref ref-type="fig" rid="fig-6">Fig. 6</xref>. Unfortunately, the TF-IDF and other document vector results were sparse, noisy, and redundant. Therefore, once the vector representation of the significant and frequent terms was completed, different unsupervised machine learning techniques were applied for label modeling. The main objective of the topic modeling module is to find a concealed theme that governs the semantics of a file, as abstract labels characterize these themes. However, realistically these results are not based on any probabilistic model, as mentioned in <xref ref-type="table" rid="table-1">Tab. 1</xref>.</p>
<fig id="fig-6"><label>Figure 6</label><caption><title>The most frequent terms extracted using TFIDF</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_32864-fig-6.png"/></fig><table-wrap id="table-1"><label>Table 1</label><caption><title>File labels extracted using LDA</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left"/>
<th align="left">Label 1</th>
<th align="left">Label 2</th>
<th align="left">Label 3</th>
<th align="left">Label 4</th>
<th align="left">Label 5</th>
<th align="left">Label 6</th>
<th align="left">Label 7</th>
<th align="left">Label 8</th>
<th align="left">Label 9</th>
<th align="left">Label 10</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">1</td>
<td align="left">process</td>
<td align="left">handle</td>
<td align="left">would</td>
<td align="left">train</td>
<td align="left">fact</td>
<td align="left">problem</td>
<td align="left">Area</td>
<td align="left">python</td>
<td align="left">perform</td>
<td align="left">represent</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">interest</td>
<td align="left">comfortable</td>
<td align="left">contact</td>
<td align="left">done</td>
<td align="left">would</td>
<td align="left">solution</td>
<td align="left">Whose</td>
<td align="left">Big</td>
<td align="left">logic</td>
<td align="left">present</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">inspired</td>
<td align="left">even</td>
<td align="left">Fail</td>
<td align="left">accuracy</td>
<td align="left">demand</td>
<td align="left">entire</td>
<td align="left">Supply</td>
<td align="left">Data</td>
<td align="left">selection</td>
<td align="left">Way</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">randomly</td>
<td align="left">way</td>
<td align="left">Door</td>
<td align="left">imagine</td>
<td align="left">intuitive</td>
<td align="left">notice</td>
<td align="left">Rough</td>
<td align="left">university</td>
<td align="left">formula</td>
<td align="left">physical</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">weird</td>
<td align="left">willing</td>
<td align="left">One</td>
<td align="left">guess</td>
<td align="left">evidence</td>
<td align="left">Ie</td>
<td align="left">institute</td>
<td align="left">center</td>
<td align="left">walking</td>
<td align="left">addition</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The same information is also displayed through a histogram, as shown in <xref ref-type="fig" rid="fig-7">Fig. 7</xref>. The labels obtained through NMF are mentioned in <xref ref-type="table" rid="table-2">Tab. 2</xref>. It is observable that NMF label results are not appropriate as, first of all, it repeats most words, and above all, the words extracted by NMF are not much meaningful, neither have they depicted the actual label of the file. For instance, if we consider labels 2 and 3 from <xref ref-type="table" rid="table-2">Tab. 2</xref>, both contain almost similar terms as extracted by NMF, while in <xref ref-type="table" rid="table-1">Tab. 1</xref>, LDA extracted utterly different words for these files.</p>
<fig id="fig-7"><label>Figure 7</label><caption><title>Histogram of extracted labels using LDA</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_32864-fig-7.png"/></fig><table-wrap id="table-2"><label>Table 2</label><caption><title>File labels extracted using NMF</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left"/>
<th align="left">Label 1</th>
<th align="left">Label 2</th>
<th align="left">Label 3</th>
<th align="left">Label 4</th>
<th align="left">Label 5</th>
<th align="left">Label 6</th>
<th align="left">Label 7</th>
<th align="left">Label 8</th>
<th align="left">Label 9</th>
<th align="left">Label 10</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">1</td>
<td align="left">learning</td>
<td align="left">data</td>
<td align="left">one</td>
<td align="left">network</td>
<td align="left">machine</td>
<td align="left">like</td>
<td align="left">neural</td>
<td align="left">Time</td>
<td align="left">use</td>
<td align="left">would</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">Zero</td>
<td align="left">zero</td>
<td align="left">zero</td>
<td align="left">usually</td>
<td align="left">natural</td>
<td align="left">price</td>
<td align="left">several</td>
<td align="left">attention</td>
<td align="left">zero</td>
<td align="left">zero</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">Fake</td>
<td align="left">fall</td>
<td align="left">fall</td>
<td align="left">cool</td>
<td align="left">attention</td>
<td align="left">zero</td>
<td align="left">advantage</td>
<td align="left">complete</td>
<td align="left">fairly</td>
<td align="left">fake</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">figure</td>
<td align="left">figure</td>
<td align="left">figure</td>
<td align="left">zero</td>
<td align="left">difference</td>
<td align="left">feature</td>
<td align="left">standard</td>
<td align="left">writing</td>
<td align="left">field</td>
<td align="left">figure</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">Field</td>
<td align="left">field</td>
<td align="left">field</td>
<td align="left">false</td>
<td align="left">recognize</td>
<td align="left">false</td>
<td align="left">zero</td>
<td align="left">feature</td>
<td align="left">felt</td>
<td align="left">field</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>LSA is also considered one of the most fundamental approaches in label modeling. After LDA and NMF, we applied LSA to extract label words from the given corpus mentioned in <xref ref-type="table" rid="table-3">Tab. 3</xref>. LSA results are much better than NMF, but LDA still got better results than LSA and NMF.</p>
<table-wrap id="table-3"><label>Table 3</label><caption><title>File labels extracted using LSA</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left"/>
<th align="left">Label 1</th>
<th align="left">Label 2</th>
<th align="left">Label 3</th>
<th align="left">Label 4</th>
<th align="left">Label 5</th>
<th align="left">Label 6</th>
<th align="left">Label 7</th>
<th align="left">Label 8</th>
<th align="left">Label 9</th>
<th align="left">Label 10</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">1</td>
<td align="left">Learning</td>
<td align="left">data</td>
<td align="left">fall</td>
<td align="left">network</td>
<td align="left">Machine</td>
<td align="left">one</td>
<td align="left">neural</td>
<td align="left">use</td>
<td align="left">like</td>
<td align="left">represent</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">One</td>
<td align="left">handle</td>
<td align="left">contact</td>
<td align="left">machine</td>
<td align="left">Neural</td>
<td align="left">big</td>
<td align="left">model</td>
<td align="left">would</td>
<td align="left">set</td>
<td align="left">present</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">Go</td>
<td align="left">field</td>
<td align="left">field</td>
<td align="left">layer</td>
<td align="left">recognize</td>
<td align="left">ovation</td>
<td align="left">show</td>
<td align="left">neural</td>
<td align="left">ai</td>
<td align="left">fake</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">problem</td>
<td align="left">flow</td>
<td align="left">door</td>
<td align="left">next</td>
<td align="left">Use</td>
<td align="left">level</td>
<td align="left">set</td>
<td align="left">complete</td>
<td align="left">problem</td>
<td align="left">field</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">process</td>
<td align="left">figure</td>
<td align="left">one</td>
<td align="left">problem</td>
<td align="left">Output</td>
<td align="left">case</td>
<td align="left">way</td>
<td align="left">feature</td>
<td align="left">process</td>
<td align="left">addition</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>It is clear from <xref ref-type="table" rid="table-1 table-2 table-3">Tabs. 1&#x2013;3</xref> that each LDA, NMF, and LSA extracted different terms for file labeling. Although NMF did not perform well for most of the files in some cases, such as labels 5 and 7, it extracted more meaningful terms. Furthermore, it was also observed that all of these three topic modeling approaches identified some non-significant words such as &#x201C;one,&#x201D; &#x201C;go,&#x201D; &#x201C;like,&#x201D; etc. <xref ref-type="fig" rid="fig-9">Fig. 9</xref> depicts the inter-topic distance map for labels of each file extracted using LDA, NMF, and LSA.</p>
<p>The dominant labels obtained through LDA and the coherence score of these labels are shown in <xref ref-type="fig" rid="fig-8">Fig. 8</xref>.</p>
<fig id="fig-8"><label>Figure 8</label><caption><title>Most significant labels extracted through lda from files with their probabilistic contribution</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_32864-fig-8.png"/></fig>
<fig id="fig-9"><label>Figure 9</label><caption><title>Intertopic distance map via multidimensional scaling of distinct labels</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_32864-fig-9.png"/></fig>
<p>One prime objective of this research was to generate an automatic system independent of any human interaction. Therefore coherence and relevance metrics are used to identify common and significant terms and better understand the semantics of extracted words for the final label. One particular problem with statistical methods in textual analysis is that the selection of the words is generally based on the probabilities extracted based on word frequency. These approaches cannot analyze whether any word contributes to specific semantics or is just a most frequent but semantically less significant word. Carson&#x00A0;et&#x00A0;al.&#x00A0;[<xref ref-type="bibr" rid="ref-32">32</xref>] established a relevance metric, which can rearrange the order of frequent words in a label by considering their relevancy to the file. This relevancy is measured through a weighting parameter <roman>&#x03B8;</roman> that can range from 0 to 1 and ascribes the corpus frequencies. If the value of <roman>&#x03B8;</roman> is close to 1, then the order of top words will be considered equal to the order of standard conditional probabilities, while <roman>&#x03B8;</roman> closer to 0 reorder the most specific word to the top of the list. All the label words extracted from different topic modeling approaches were used to calculate the relevance, as mentioned in <xref ref-type="fig" rid="fig-10">Fig. 10a</xref>.</p>
<fig id="fig-10"><label>Figure 10</label><caption><title>(a) Relevancy metric results of extracted labels (b) Coherence score for extracted labels of different files</title></caption><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_32864-fig-10a.png"/><graphic mimetype="image" mime-subtype="png" xlink:href="CMC_32864-fig-10b.png"/></fig>
<p>On the other hand, the coherence metric was developed by David et al. [<xref ref-type="bibr" rid="ref-33">33</xref>] and is being used for model selection and resolutions. However, it can help in guiding intuition when applied to single labels while identifying accurate labels in which a coherent concept might not be observed at first glance. <xref ref-type="fig" rid="fig-10">Fig. 10b</xref> shows the coherence among the most relevant labels extracted through LDA, NMF, and LSA techniques. Finally, the relevant and most coherent topic words were given to BERT for file label generation. BERT was implemented with a fixed set of 3 words per label, as shown in <xref ref-type="table" rid="table-4">Tab. 4</xref>.</p>
<table-wrap id="table-4"><label>Table 4</label><caption><title>Updated file labels extracted using BERT</title></caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th align="left">File No.</th>
<th align="left">Count</th>
<th align="left">Label</th>
</tr>
</thead>
<tbody>
<tr>
<td align="left">1</td>
<td align="left">94461</td>
<td align="left">using_data_image</td>
</tr>
<tr>
<td align="left">2</td>
<td align="left">30370</td>
<td align="left">model_classification_network</td>
</tr>
<tr>
<td align="left">3</td>
<td align="left">8035</td>
<td align="left">video_order_price</td>
</tr>
<tr>
<td align="left">4</td>
<td align="left">6802</td>
<td align="left">ai_architecture _design</td>
</tr>
<tr>
<td align="left">5</td>
<td align="left">6283</td>
<td align="left">clap_product_certainly</td>
</tr>
<tr>
<td align="left">6</td>
<td align="left">6185</td>
<td align="left">text_speech_translation</td>
</tr>
<tr>
<td align="left">7</td>
<td align="left">6048</td>
<td align="left">feed_grid _regression</td>
</tr>
<tr>
<td align="left">8</td>
<td align="left">5291</td>
<td align="left">multiple_team_development</td>
</tr>
<tr>
<td align="left">9</td>
<td align="left">5145</td>
<td align="left">political_history_usa</td>
</tr>
<tr>
<td align="left">10</td>
<td align="left">4465</td>
<td align="left">lower_path_map_</td>
</tr>
<tr>
<td align="left">11</td>
<td align="left">4270</td>
<td align="left">computational_code_activation</td>
</tr>
<tr>
<td align="left">12</td>
<td align="left">4072</td>
<td align="left">region_business _industry</td>
</tr>
<tr>
<td align="left">13</td>
<td align="left">4038</td>
<td align="left">pattern_loss_reduction</td>
</tr>
<tr>
<td align="left">14</td>
<td align="left">3958</td>
<td align="left">intelligent_reinforcement_reward</td>
</tr>
<tr>
<td align="left">15</td>
<td align="left">3743</td>
<td align="left">global_pollution_reduction</td>
</tr>
<tr>
<td align="left">16</td>
<td align="left">3630</td>
<td align="left">network_drive_engine</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>It is clear from these results that the final label is much more meaningful than the label words extracted by LDA, NMF, and LSA individually. Moreover, these labels also give the impression of traditional file names given by a human. For instance, file 9 was a book on the history of the USA and is labeled as &#x201C;Political history USA.&#x201D; The label for file 4, an article on intelligent agent architectures, is &#x201C;AI architecture design.&#x201D; On the contrary, file 5 was an introductory brusher of different company products, and file 10 was an article on AI&#x2019;s informed and uninformed search techniques. Both of these files did not get any meaningful results. One primary reason for these non-meaningful labels is that these files were not textually rich and had multiple redundant terms instead of a complete textual theme.</p>
</sec>
<sec id="s5"><label>5</label><title>Conclusion</title>
<p>The increasing size and capacity of secondary storage devices and fast operating systems assist in aggregating the files users can store in their systems. However, this increase in the number of digital files might confuse users and cause them to spend additional time and effort finding and managing required files efficiently. Conversely, the need to automatically analyze, manage and label digital files has become significantly relevant. The particular challenge with file labeling is that it contains relatively heterogeneous and noisy data that might infer an inaccurate label. The proposed methodology can reasonably overcome this problem and can be used to label any textual file, including academic papers, user files, PowerPoint presentations, and PDFs. Despite individual label modeling techniques, qualitative evaluation of the sentence level BERT embedding reveals that these embeddings effectively organize a diverse range of digital files.</p>
</sec>
</body>
<back>
<ack>
<p>Thanks to our families &#x0026; colleagues who supported us morally.</p>
</ack>
<fn-group>
<fn fn-type="other"><p><bold>Funding Statement:</bold> The authors received no specific funding for this study.</p></fn>
<fn fn-type="conflict"><p><bold>Conflicts of Interest:</bold> The authors declare that they have no conflicts of interest to report regarding the present study.</p></fn>
</fn-group>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J. D.</given-names> <surname>Dinneen</surname></string-name> and <string-name><given-names>J.</given-names> <surname>Charles-Antoine</surname></string-name></person-group>, &#x201C;<article-title>The ubiquitous digital file: A review of file management research</article-title>,&#x201D; <source>Journal of the Association for Information Science and Technology</source>, vol. <volume>71</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>32</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C.</given-names> <surname>John</surname></string-name></person-group>, &#x201C;<article-title>Creative names for personal files in an interactive computing environment</article-title>,&#x201D; <source>International Journal of Man Machine Studies</source>, vol. <volume>16</volume>, no. <issue>4</issue>, pp. <fpage>405</fpage>&#x2013;<lpage>438</lpage>, <year>1982</year>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Ben</surname></string-name>, <string-name><given-names>D.</given-names> <surname>Andy</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Palmer</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Hamish</surname></string-name></person-group>, &#x201C;<article-title>Organizing and managing personal electronic files: A mechanical engineer&#x2019;s perspective</article-title>,&#x201D; <source>ACM Transactions on Information Systems</source>, vol. <volume>26</volume>, no. <issue>4</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>40</lpage>, <year>2008</year>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Crowder</surname></string-name>, <string-name><given-names>S. M.</given-names> <surname>Jonathan</surname></string-name> and <string-name><given-names>R.</given-names> <surname>Michele</surname></string-name></person-group>, &#x201C;<article-title>File naming in digital media research: Examples from the humanities and social sciences</article-title>,&#x201D; <source>Journal of Librarianship and Scholarly Communication</source>, vol. <volume>3</volume>, no. <issue>3</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>22</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Harumasa</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Osamu</surname></string-name> and <string-name><given-names>H.</given-names> <surname>Masahiro</surname></string-name></person-group>, &#x201C;<article-title>A file naming scheme using hierarchical-keywords</article-title>,&#x201D; in <conf-name>26th Annual Int. Computer Software and Applications</conf-name>, <conf-loc>Oxford, England</conf-loc>, pp. <fpage>799</fpage>&#x2013;<lpage>804</lpage>, <year>2002</year>.</mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>L.</given-names> <surname>Alon</surname></string-name> and <string-name><given-names>N.</given-names> <surname>Rafi</surname></string-name></person-group>, &#x201C;<article-title>Gaps between actual and ideal personal information management behavior</article-title>,&#x201D; <source>Computers in Human Behavior</source>, vol. <volume>107</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>10</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>M. B.</given-names> <surname>David</surname></string-name>, <string-name><given-names>Y. N.</given-names> <surname>Andrew</surname></string-name> and <string-name><given-names>I. J.</given-names> <surname>Michael</surname></string-name></person-group>, &#x201C;<article-title>Latent dirichlet allocation</article-title>,&#x201D; <source>Journal of Machine Learning Research</source>, vol. <volume>3</volume>, no. <issue>1</issue>, pp. <fpage>993</fpage>&#x2013;<lpage>1022</lpage>, <year>2003</year>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Hofmann</surname></string-name></person-group>, &#x201C;<article-title>Unsupervised learning by probabilistic latent semantic analysis</article-title>,&#x201D; <source>Machine Learning</source>, vol. <volume>42</volume>, no. <issue>1</issue>, pp. <fpage>177</fpage>&#x2013;<lpage>196</lpage>, <year>2001</year>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Seung</surname></string-name> and <string-name><given-names>L. Lee</given-names></string-name></person-group>, &#x201C;<article-title>Algorithms for non-negative matrix factorization</article-title>,&#x201D; <source>Advances in Neural Information Processing Systems</source>, vol. <volume>13</volume>, no. <issue>1</issue>, pp. <fpage>556</fpage>&#x2013;<lpage>562</lpage>, <year>2001</year>.</mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>D. A.</given-names> <surname>Ostrowski</surname></string-name></person-group>, &#x201C;<article-title>Using latent dirichlet allocation for topic modelling in twitter</article-title>,&#x201D; in <conf-name>Proc. of the IEEE 9th Int. Conf. on Semantic Computing</conf-name>, <conf-loc>Anaheim, CA, USA</conf-loc>, pp. <fpage>493</fpage>&#x2013;<lpage>497</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><given-names>R.</given-names> <surname>Daniel</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Evan</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Jason</surname></string-name>, <string-name><given-names>D. M.</given-names> <surname>Christopher</surname></string-name> and <string-name><given-names>A. M.</given-names> <surname>Daniel</surname></string-name></person-group>, &#x201C;<chapter-title>Topic modeling for the social sciences</chapter-title>,&#x201D; in <source>NIPS 2009 Workshop on Applications for Topic Models: Text and Beyond</source>, <publisher-loc>Canada</publisher-loc>: <publisher-name>Whistler</publisher-name>, pp. <fpage>1</fpage>&#x2013;<lpage>4</lpage>, <year>2009</year>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Rubayyi</surname></string-name> and <string-name><given-names>A.</given-names> <surname>Khalid</surname></string-name></person-group>, &#x201C;<article-title>A survey of topic modeling in text mining</article-title>,&#x201D; <source>International Journal of Advanced Computer Science Application</source>, vol. <volume>6</volume>, no. <issue>1</issue>, pp. <fpage>147</fpage>&#x2013;<lpage>153</lpage>, <year>2015</year>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>P.</given-names> <surname>Pantel</surname></string-name> and <string-name><given-names>R.</given-names> <surname>Deepak</surname></string-name></person-group>, &#x201C;<article-title>Automatically labeling semantic classes</article-title>,&#x201D; in <conf-name>Proc. of the Human Language Technology Conf. of the North American Chapter of the Association for Computational Linguistics</conf-name>, <conf-loc>Rochester, New York, USA</conf-loc>, pp. <fpage>321</fpage>&#x2013;<lpage>328</lpage>, <year>2004</year>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Alokaili</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Aletras</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Stevenson</surname></string-name></person-group>, &#x201C;<article-title>Automatic generation of topic labels</article-title>,&#x201D; in <conf-name>Proc. of the 43rd Int. ACM Conf. on Research and Development in Information Retrieval</conf-name>, <conf-loc>Xi&#x2019;an, China</conf-loc>, pp. <fpage>1965</fpage>&#x2013;<lpage>1968</lpage>, <year>2020</year>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Chang</surname></string-name>, <string-name><given-names>G.</given-names> <surname>Sean</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Chong</surname></string-name>, <string-name><given-names>B. G.</given-names> <surname>Jordan</surname></string-name> and <string-name><given-names>B.</given-names> <surname>David</surname></string-name></person-group>, &#x201C;<article-title>Reading tea leaves: How humans interpret topic models</article-title>,&#x201D; in <conf-name>Twenty-Third Annual Conf. on Neural Information Processing Systems</conf-name>, <conf-loc>Vancouver, Canada</conf-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>9</lpage>, <year>2009</year>.</mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Qiaozhu</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Xuehua</surname></string-name> and <string-name><given-names>Z.</given-names> <surname>Chengxiang</surname></string-name></person-group>, &#x201C;<article-title>Automatic labeling of multinomial topic models</article-title>,&#x201D; in <conf-name>Proc. of Thirteenth ACM Int. Conf. on Knowledge Discovery and Data Mining</conf-name>, <conf-loc>San Jose, California</conf-loc>, pp. <fpage>490</fpage>&#x2013;<lpage>499</lpage>, <year>2007</year>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>H.</given-names> <surname>Ioana</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Conor</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Marcel</surname></string-name> and <string-name><given-names>G.</given-names> <surname>Derek</surname></string-name></person-group>, &#x201C;<article-title>Unsupervised graph-based topic labelling using dbpedia</article-title>,&#x201D; in <conf-name>Proc. of Int. Conf. on Web Search and Data Mining</conf-name>, <conf-loc>Rome, Italy</conf-loc>, pp. <fpage>465</fpage>&#x2013;<lpage>473</lpage>, <year>2013</year>.</mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Davide</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Silvia</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Davide</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Fabio</surname></string-name></person-group>, &#x201C;<article-title>Automatic labeling of topics</article-title>,&#x201D; in <conf-name>Ninth Int. Conf. on Intelligent Systems Design and Applications</conf-name>, <conf-loc>Pisa, Italy</conf-loc>, pp. <fpage>1227</fpage>&#x2013;<lpage>1232</lpage>, <year>2009</year>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>H. L.</given-names> <surname>Jey</surname></string-name>, <string-name><given-names>N.</given-names> <surname>David</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Sarvnaz</surname></string-name> and <string-name><given-names>B.</given-names> <surname>Timothy</surname></string-name></person-group>, &#x201C;<article-title>Best topic word selection for topic labelling</article-title>,&#x201D; in <conf-name>Proc. of the 23rd Int. Conf. on Computational Linguistics</conf-name>, <conf-loc>Beijing, China</conf-loc>, pp. <fpage>605</fpage>&#x2013;<lpage>613</lpage>, <year>2010</year>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>W.</given-names> <surname>Xiaojun</surname></string-name> and <string-name><given-names>W.</given-names> <surname>Tianming</surname></string-name></person-group>, &#x201C;<article-title>Automatic labeling of topic models using text summaries</article-title>,&#x201D; in <conf-name>Proc. of Association for Computational Linguistics</conf-name>, <conf-loc>Berlin, Germany</conf-loc>, pp. <fpage>2297</fpage>&#x2013;<lpage>2305</lpage>, <year>2016</year>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>E.</given-names> <surname>Amparo</surname></string-name>, <string-name><given-names>B.</given-names> <surname>Cano</surname></string-name>, <string-name><given-names>H.</given-names> <surname>Yulan</surname></string-name> and <string-name><given-names>X.</given-names> <surname>Ruifeng</surname></string-name></person-group>, &#x201C;<article-title>Automatic labelling of topic models learned from twitter by summarisation</article-title>,&#x201D; in <conf-name>Proc. of Association for Computational Linguistics</conf-name>, <conf-loc>Baltimore, USA</conf-loc>, pp. <fpage>618</fpage>&#x2013;<lpage>624</lpage>, <year>2014</year>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Hamed</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Yongli</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Chi</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Xia</surname></string-name>, <string-name><given-names>J.</given-names> <surname>Xiahui</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>Latent dirichlet allocation and topic modeling: Models, applications, a survey</article-title>,&#x201D; <source>Multimedia Tools and Applications</source>, vol. <volume>78</volume>, no. <issue>1</issue>, pp. <fpage>15169</fpage>&#x2013;<lpage>15211</lpage>, <year>2019</year>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Angelov</surname></string-name></person-group>, &#x201C;<article-title>Top2vec: Distributed representations of topics</article-title>,&#x201D; <italic>Arxiv</italic>, vol. <volume>22</volume>, pp. <fpage>1</fpage>&#x2013;<lpage>10</lpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>G.</given-names> <surname>Maarten</surname></string-name></person-group>, &#x201C;<article-title>Bertopic: Neural topic modeling with a class-based tf-idf procedure</article-title>,&#x201D; <italic>Arxiv</italic>, vol. <volume>22</volume>, pp. <fpage>1</fpage>&#x2013;<lpage>13</lpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>S. B.</given-names> <surname>Forrest</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Hebi</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Ge</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Cen</surname></string-name>, <string-name><given-names>Y.</given-names> <surname>Yinfei</surname></string-name> <etal>et al.,</etal></person-group> &#x201C;<article-title>End-to-end semantics-based summary quality assessment for single-document summarization</article-title>,&#x201D; in <conf-name>Annual Conf. of the North American Chapter of the Association for Computational Linguistics</conf-name>, <conf-loc>Seattle, Washington, USA</conf-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>9</lpage>, <year>2022</year>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. J.</given-names> <surname>Karen</surname></string-name></person-group>, &#x201C;<article-title>A statistical interpretation of term specificity and its application in retrieval</article-title>,&#x201D; <source>Journal of Documentation</source>, vol. <volume>28</volume>, no. <issue>1</issue>, pp. <fpage>339</fpage>&#x2013;<lpage>348</lpage>, <year>1972</year>.</mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>Tomas</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Kai</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Greg</surname></string-name> and <string-name><given-names>D.</given-names> <surname>Jeffrey</surname></string-name></person-group>, &#x201C;<article-title>Efficient estimation of word representations in vector space</article-title>,&#x201D; in <conf-name>1st Int. Conf. on Learning Representations</conf-name>, <conf-loc>Scottsdale, Arizona, USA</conf-loc>, pp. <fpage>1</fpage>&#x2013;<lpage>13</lpage>, <year>2013</year>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>Q.</given-names> <surname>Le and M</surname></string-name></person-group>. Tomas, &#x201C;<article-title>Distributed representations of sentences and documents</article-title>,&#x201D; in <conf-name>Int. Conf. on Machine Learning</conf-name>, <conf-loc>Beijing, China</conf-loc>, pp. <fpage>1188</fpage>&#x2013;<lpage>1196</lpage>, <year>2014</year>.</mixed-citation></ref>
<ref id="ref-29"><label>[29]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>G. N.</given-names> <surname>Gopal</surname></string-name>, <string-name><given-names>C. K.</given-names> <surname>Binsu</surname></string-name> and <string-name><given-names>U. Mini</given-names></string-name></person-group>, &#x201C;<article-title>Keyword template based semi-supervised topic modelling in tweets</article-title>,&#x201D; in <conf-name>Int. Conf. on Innovative Computing and Communications</conf-name>, <conf-loc>Singapore</conf-loc>, pp. <fpage>659</fpage>&#x2013;<lpage>666</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-30"><label>[30]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. T.</given-names> <surname>Dumais</surname></string-name></person-group>, &#x201C;<article-title>Latent semantic analysis</article-title>,&#x201D; <source>Annual Review of Information Science and Technology</source>, vol. <volume>38</volume>, no. <issue>1</issue>, pp. <fpage>189</fpage>&#x2013;<lpage>230</lpage>, <year>2004</year>.</mixed-citation></ref>
<ref id="ref-31"><label>[31]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>P. C.</given-names> <surname>Hansen</surname></string-name></person-group>, &#x201C;<article-title>Truncated singular value decomposition solutions to discrete ill-posed problems with ill-determined numerical rank</article-title>,&#x201D; <source>SIAM Journal on Scientific and Statistical Computing</source>, vol. <volume>11</volume>, no. <issue>3</issue>, pp. <fpage>503</fpage>&#x2013;<lpage>518</lpage>, <year>1990</year>.</mixed-citation></ref>
<ref id="ref-32"><label>[32]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>S.</given-names> <surname>Carson</surname></string-name> and <string-name><given-names>S.</given-names> <surname>Kenneth</surname></string-name></person-group>, &#x201C;<article-title>Ldavis: A method for visualizing and interpreting topics</article-title>,&#x201D; in <conf-name>Proc. of the Workshop on Interactive Language Learning, Visualization, and Interfaces</conf-name>, <conf-loc>Baltimore, Maryland, USA</conf-loc>, pp. <fpage>63</fpage>&#x2013;<lpage>70</lpage>, <year>2014</year>.</mixed-citation></ref>
<ref id="ref-33"><label>[33]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>M.</given-names> <surname>David</surname></string-name>, <string-name><given-names>M. W.</given-names> <surname>Hanna</surname></string-name>, <string-name><given-names>T.</given-names> <surname>Edmund</surname></string-name>, <string-name><given-names>L.</given-names> <surname>Miriam</surname></string-name> and <string-name><given-names>M.</given-names> <surname>Andrew</surname></string-name></person-group>, &#x201C;<article-title>Optimizing semantic coherence in topic models</article-title>,&#x201D; in <conf-name>Proc. of the Conf. on Empirical Methods in Natural Language Processing</conf-name>, <conf-loc>Edinburgh, Scotland, UK</conf-loc>, pp. <fpage>262</fpage>&#x2013;<lpage>272</lpage>, <year>2011</year>.</mixed-citation></ref>
</ref-list>
</back>
</article>













