<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.1 20151215//EN" "http://jats.nlm.nih.gov/publishing/1.1/JATS-journalpublishing1.dtd">
<article xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:mml="http://www.w3.org/1998/Math/MathML" xml:lang="en" article-type="research-article" dtd-version="1.1">
<front>
<journal-meta>
<journal-id journal-id-type="pmc">CMC</journal-id>
<journal-id journal-id-type="nlm-ta">CMC</journal-id>
<journal-id journal-id-type="publisher-id">CMC</journal-id>
<journal-title-group>
<journal-title>Computers, Materials &#x0026; Continua</journal-title>
</journal-title-group>
<issn pub-type="epub">1546-2226</issn>
<issn pub-type="ppub">1546-2218</issn>
<publisher>
<publisher-name>Tech Science Press</publisher-name>
<publisher-loc>USA</publisher-loc>
</publisher>
</journal-meta>
<article-meta>
<article-id pub-id-type="publisher-id">56971</article-id>
<article-id pub-id-type="doi">10.32604/cmc.2024.056971</article-id>
<article-categories>
<subj-group subj-group-type="heading">
<subject>Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title>CHART: Intelligent Crime Hotspot Detection and Real-Time Tracking Using Machine Learning</article-title>
<alt-title alt-title-type="left-running-head">CHART: Intelligent Crime Hotspot Detection and Real-Time Tracking Using Machine Learning</alt-title>
<alt-title alt-title-type="right-running-head">CHART: Intelligent Crime Hotspot Detection and Real-Time Tracking Using Machine Learning</alt-title>
</title-group>
<contrib-group>
<contrib id="author-1" contrib-type="author">
<name name-style="western"><surname>Ahmad</surname><given-names>Rashid</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-2" contrib-type="author" corresp="yes">
<name name-style="western"><surname>Nawaz</surname><given-names>Asif</given-names></name><xref ref-type="aff" rid="aff-1">1</xref><email>asif.nawaz@uaar.edu.pk</email></contrib>
<contrib id="author-3" contrib-type="author">
<name name-style="western"><surname>Mustafa</surname><given-names>Ghulam</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-4" contrib-type="author">
<name name-style="western"><surname>Ali</surname><given-names>Tariq</given-names></name><xref ref-type="aff" rid="aff-1">1</xref></contrib>
<contrib id="author-5" contrib-type="author">
<name name-style="western"><surname>Tlija</surname><given-names>Mehdi</given-names></name><xref ref-type="aff" rid="aff-2">2</xref></contrib>
<contrib id="author-6" contrib-type="author">
<name name-style="western"><surname>El-Meligy</surname><given-names>Mohammed A.</given-names></name><xref ref-type="aff" rid="aff-3">3</xref><xref ref-type="aff" rid="aff-4">4</xref></contrib>
<contrib id="author-7" contrib-type="author">
<name name-style="western"><surname>Ahmed</surname><given-names>Zohair</given-names></name><xref ref-type="aff" rid="aff-5">5</xref></contrib>
<aff id="aff-1"><label>1</label><institution>University Institute of Information Technology, PMAS Arid Agriculture University</institution>, <addr-line>Rawalpindi, 46000</addr-line>, <country>Pakistan</country></aff>
<aff id="aff-2"><label>2</label><institution>Industrial Engineering Department, College of Engineering, King Saud University</institution>, <addr-line>Riyadh, 11421</addr-line>, <country>Saudi Arabia</country></aff>
<aff id="aff-3"><label>3</label><institution>Jadara University Research Center, Jadara University</institution>, <addr-line>Irbid, 21110</addr-line>, <country>Jordan</country></aff>
<aff id="aff-4"><label>4</label><institution>Applied Science Research Center, Applied Science Private University</institution>, <addr-line>Amman, 11931</addr-line>, <country>Jordan</country></aff>
<aff id="aff-5"><label>5</label><institution>School of Computer Science and Engineering, Central South University</institution>, <addr-line>Changsha, 410083</addr-line>, <country>China</country></aff>
</contrib-group>
<author-notes>
<corresp id="cor1"><label>&#x002A;</label>Corresponding Author: Asif Nawaz. Email: <email>asif.nawaz@uaar.edu.pk</email></corresp>
</author-notes>
<pub-date date-type="collection" publication-format="electronic">
<year>2024</year>
</pub-date>
<pub-date date-type="pub" publication-format="electronic">
<day>19</day><month>12</month><year>2024</year>
</pub-date>
<volume>81</volume>
<issue>3</issue>
<fpage>4171</fpage>
<lpage>4194</lpage>
<history>
<date date-type="received">
<day>04</day>
<month>8</month>
<year>2024</year>
</date>
<date date-type="accepted">
<day>23</day>
<month>10</month>
<year>2024</year>
</date>
</history>
<permissions>
<copyright-statement>&#x00A9; 2024 The Authors.</copyright-statement>
<copyright-year>2024</copyright-year>
<copyright-holder>Published by Tech Science Press.</copyright-holder>
<license xlink:href="https://creativecommons.org/licenses/by/4.0/">
<license-p>This work is licensed under a <ext-link ext-link-type="uri" xlink:type="simple" xlink:href="https://creativecommons.org/licenses/by/4.0/">Creative Commons Attribution 4.0 International License</ext-link>, which permits unrestricted use, distribution, and reproduction in any medium, provided the original work is properly cited.</license-p>
</license>
</permissions>
<self-uri content-type="pdf" xlink:href="TSP_CMC_56971.pdf"></self-uri>
<abstract>
<p>Crime hotspot detection is essential for law enforcement agencies to allocate resources effectively, predict potential criminal activities, and ensure public safety. Traditional methods of crime analysis often rely on manual, time-consuming processes that may overlook intricate patterns and correlations within the data. While some existing machine learning models have improved the efficiency and accuracy of crime prediction, they often face limitations such as overfitting, imbalanced datasets, and inadequate handling of spatiotemporal dynamics. This research proposes an advanced machine learning framework, CHART (Crime Hotspot Analysis and Real-time Tracking), designed to overcome these challenges. The proposed methodology begins with comprehensive data collection from the police database. The dataset includes detailed attributes such as crime type, location, time and demographic information. The key steps in the proposed framework include: Data Preprocessing, Feature Engineering that leveraging domain-specific knowledge to extract and transform relevant features. Heat Map Generation that employs Kernel Density Estimation (KDE) to create visual representations of crime density, highlighting hotspots through smooth data point distributions and Hotspot Detection based on Random Forest-based to predict crime likelihood in various areas. The Experimental evaluation demonstrated that CHART shows superior performance over benchmark methods, significantly improving crime detection accuracy by getting 95.24% for crime detection-I (CD-I), 96.12% for crime detection-II (CD-II) and 94.68% for crime detection-III (CD-III), respectively. By designing the application with integrating sophisticated preprocessing techniques, balanced data representation, and advanced feature engineering, the proposed model provides a reliable and practical tool for real-world crime analysis. Visualization of crime hotspots enables law enforcement agencies to strategize effectively, focusing resources on high-risk areas and thereby enhancing overall crime prevention and response efforts.</p>
</abstract>
<kwd-group kwd-group-type="author">
<kwd>Crime hotspot</kwd>
<kwd>heat map</kwd>
<kwd>kernel density estimation (KDE)</kwd>
<kwd>support vector machine (SVM)</kwd>
</kwd-group>
<funding-group>
<award-group id="awg1">
<funding-source>appreciation to King Saud University for funding this work through Researchers Supporting Project number</funding-source>
<award-id>RSPD2025R685</award-id>
</award-group>
</funding-group>
</article-meta>
</front>
<body>
<sec id="s1">
<label>1</label>
<title>Introduction</title>
<p>Crime is an act that violate the laws of a society, typically leading to prosecution and punishment by the state [<xref ref-type="bibr" rid="ref-1">1</xref>]. These acts range from minor infractions, such as petty theft and vandalism, to severe offenses like murder and terrorism. The effects of crime are profound and multifaceted, impacting individuals, communities, and society at large [<xref ref-type="bibr" rid="ref-2">2</xref>]. On an individual level, victims of crime can suffer physical harm, psychological trauma, and financial loss, which can lead to long-term emotional distress and a diminished quality of life [<xref ref-type="bibr" rid="ref-3">3</xref>]. Communities affected by high crime rates often experience a breakdown of social cohesion and trust, leading to fear, decreased property values, and economic decline. Businesses may be reluctant to invest in high-crime areas, exacerbating unemployment and poverty. Societally, crime strains public resources, including law enforcement, judicial systems, and correctional facilities. It also necessitates substantial government expenditure on policing, legal proceedings, and incarceration [<xref ref-type="bibr" rid="ref-4">4</xref>]. Moreover, pervasive crime can erode public confidence in the rule of law and governance, leading to broader social stability and development implications. Thus, understanding and addressing crime is essential for promoting safety, justice, and prosperity within any community [<xref ref-type="bibr" rid="ref-5">5</xref>].</p>
<p>A crime hotspot is a specific geographic area where the frequency of criminal activity is significantly higher compared to other areas. These hotspots often cluster crimes, such as theft, assault, or vandalism, making them focal points for law enforcement and community safety efforts [<xref ref-type="bibr" rid="ref-6">6</xref>]. Identifying and understanding crime hotspots is crucial for effective policing, as it allows for strategically deploying resources, targeted patrols, and proactive crime prevention measures. Hotspot detection is the process of identifying these high-crime areas using various analytical techniques and data sources. This process involves collecting and analyzing crime data, often including details such as the type of crime, location, time of occurrence, and other relevant factors. Advanced methods, such as statistical analysis, Geographic Information Systems (GIS), and machine learning algorithms like Kernel Density Estimation (KDE) are used to visualize and predict hotspots [<xref ref-type="bibr" rid="ref-7">7</xref>]. These tools help in mapping out the intensity and distribution of criminal activities, enabling law enforcement agencies to focus their efforts on areas that need the most attention, thereby enhancing public safety and reducing crime rates.</p>
<p>Traditional hotspot detection methods have primarily involved manual analysis, statistical approaches, and basic Geographic Information Systems (GIS). Manual analysis entails law enforcement personnel reviewing crime reports and records to identify high-crime areas. This method often involves plotting incidents on physical maps or simple digital tools and visually identifying clusters of criminal activity [<xref ref-type="bibr" rid="ref-8">8</xref>]. While this approach can provide insights, it is labor-intensive and time-consuming. Additionally, it is highly susceptible to human error and subjective bias, which can lead to inconsistent and inaccurate hotspot identification. The manual process&#x2019;s limitations make it difficult to keep up with the dynamic and evolving nature of criminal activity. Statistical methods, such as point pattern analysis and spatial autocorrelation, have also been used to detect crime hotspots [<xref ref-type="bibr" rid="ref-9">9</xref>]. These techniques involve calculating the density and distribution of crime incidents within a given area. Point pattern analysis focuses on identifying statistically significant clusters of events, while spatial autocorrelation measures the degree to which crime events are spatially correlated. Although these methods offer a more systematic approach than manual analysis, they are often limited by their reliance on predefined statistical models and thresholds, which may not accurately capture the complexities of real-world crime patterns.</p>
<p>Basic GIS-based methods have advanced traditional hotspot detection by enabling the visualization of crime data on digital maps. Tools like heat maps and thematic maps allow for a more intuitive understanding of crime distribution [<xref ref-type="bibr" rid="ref-10">10</xref>]. However, these methods still have significant limitations. Basic GIS tools often lack the analytical depth required to identify nuanced patterns and trends. They may not effectively integrate multiple data sources or consider the influence of various socio-economic and environmental factors on crime. Furthermore, these methods usually provide static representations of crime data, failing to capture criminal activity&#x2019;s dynamic and temporal aspects.</p>
<p>Addressing these challenges may require a better model and practical implementation strategies to ensure that machine learning models provide reliable and actionable insights for crime prevention and law enforcement [<xref ref-type="bibr" rid="ref-11">11</xref>,<xref ref-type="bibr" rid="ref-12">12</xref>]. The proposed CHART represents a significant advancement over traditional and existing machine learning methods for crime hotspot detection. By integrating comprehensive data preprocessing, robust feature engineering, and sophisticated algorithms like Adaptive Synthetic Sampling (ADASYN) and Kernel Density Estimation (KDE), CHART effectively addresses the limitations of previous approaches, such as inefficiencies, biases, and computational complexity. The use of a Random Forest-based model ensures high accuracy and robustness, mitigating overfitting and enhancing generalizability. CHART&#x2019;s ability to deliver precise, timely, and actionable insights into crime patterns significantly outperforms benchmark methods, allowing law enforcement agencies to allocate resources and improve crime prevention and response strategically. This framework sets a new standard for dynamic spatial analysis and prediction, providing a powerful tool for enhancing public safety and community well-being.</p>
<p>The key contribution of the proposed research are as follows:
<list list-type="bullet">
<list-item>
<p>Utilization of advanced feature engineering techniques using domain-specific knowledge, such as time of day, location type, and historical crime frequency, to extract and transform relevant features for improved model performance.</p></list-item>
<list-item>
<p>Introducing Kernel Density Estimation (KDE) for precise spatial analysis and visualization of crime hotspots, enabling effective resource allocation and strategic planning for law enforcement agencies.</p></list-item>
<list-item>
<p>Experimental results demonstrate that the CHART framework outperforms benchmark methods in crime hotspot detection, achieving higher accuracy, precision, recall, and F1 scores, thus providing more reliable and actionable insights for law enforcement agencies.</p></list-item>
</list></p>
<p>The rest of the paper is organized as follows: <xref ref-type="sec" rid="s2">Section 2</xref> reviews the current literature on hotspot detection techniques, <xref ref-type="sec" rid="s3">Section 3</xref> outlines the core methodology of the proposed work, <xref ref-type="sec" rid="s4">Section 4</xref> presents the experimental evaluations and results. <xref ref-type="sec" rid="s5">Section 5</xref> illustrates the application of CHART and <xref ref-type="sec" rid="s6">Section 6</xref> discusses the conclusion and directions for future work.</p>
</sec>
<sec id="s2">
<label>2</label>
<title>Literature Review</title>
<p>This section discusses the current literature that has been carried out in the domain of crime prediction. Dakalbab et al. [<xref ref-type="bibr" rid="ref-13">13</xref>] proposed a comparative analysis of different artificial intelligence based model to predict and prevent crime. The authors did a study where they reviewed 120 research papers about AI and crime prediction. They looked at the different types of crimes studied and the techniques used to predict them. They found that supervised learning was the most common approach used. They also looked at the strengths and weaknesses of the different techniques. They found that AI can be very effective in predicting crime, especially when used to identify crime hot spots. Hybrid models also showed promise. In the end, they suggested that more research should be done on hybrid models and that they plan to do further experiments to improve their own solution.</p>
<p>Xie et al. [<xref ref-type="bibr" rid="ref-14">14</xref>] discussed that spatial hotspot mapping is important in many areas like public health, public safety, transportation, and environmental science. It helps to identify areas with high rates of certain events like disease or crime, but traditional clustering techniques can give false results, which can be costly. To solve this problem, they developed a statistically robust clustering techniques, which use rigorous statistical methods to control false results. This article provides an optimized technique, including data modeling, region enumeration, maximization algorithms, and significance testing. The goal was to stimulate new ideas and approaches in computing research and help practitioners choose the best techniques for their needs. The work of Garcia-Zanabria et al. [<xref ref-type="bibr" rid="ref-15">15</xref>] discussed the challenges of understanding crime patterns in big cities. Crime is often spread out and hard to see, making it difficult and expensive to analyze. Their article introduced a new method called CriPAV, which helps experts analyze street-level crime patterns. CriPAV has two main parts: a way to find likely hotspots of crime based on probability, not just intensity, and a technique to identify similar hotspots by mapping them in a Cartesian space. CriPAV has been tested with real crime data in Sao Paulo and has been shown to help experts understand crime patterns and how they relate to the city.</p>
<p>Law enforcement authorities need to use data-driven strategies to prevent and detect crimes, as proposed by Al-Osaimi et al. [<xref ref-type="bibr" rid="ref-16">16</xref>]. However, their work limits the amount of data generated every day is increasing, which makes it difficult to process and store it. Their article also new Apriori algorithm to analyze crime by using various datasets. They designed a crime analysis tool for public safety and data mining that helps law enforcement officers to make better decisions. Wu et al. [<xref ref-type="bibr" rid="ref-17">17</xref>] proposed a place-based short-term crime prediction model that used patterns of past crimes to predict future crime incidents in specific locations. Their model was based on the concept human mobility that can contribute to limited crime generation. They used a large-scale human mobility dataset to evaluate the effects of human mobility features on short-term crime prediction. In addition to this, they also tested various neural network models on different cities with diverse demographics and types of crimes and found that adding human mobility flow features to historical crime data can improve prediction accuracy.</p>
<p>Cardone et al. [<xref ref-type="bibr" rid="ref-18">18</xref>] presented a fuzzy-based spatiotemporal hot spot intensity and propagation technique. Their work explained a new way to study &#x201C;hot spots,&#x201D; where a certain thing is happening a lot. The method involves using a computer program to find these hot spots and measure how strong they are. Their method was tested by looking at crime in the City of London over several years, and the results showed that crime has been decreasing in all parts of the city. Their method seems to be reliable and could be used in the future to study other things happening in different places. Appiah et al. [<xref ref-type="bibr" rid="ref-19">19</xref>] also discussed a model-based clustering of expectation maximization and K-means algorithms in crime hotspot analysis to fix crime in different areas. They used a mathematical method called Gaussian multivariate distributions to estimate potential crime hotspots. This involves finding the best way to group data points into clusters to identify areas where crimes are likely to occur. They used a large dataset of violent crimes and analyzed the data using a combination of K-means clustering and the expectation-maximization (E-M) algorithm. They found that this new method is efficient and fast and produced similar results to traditional methods.</p>
<p>Prathap et al. [<xref ref-type="bibr" rid="ref-20">20</xref>] discussed a geospatial crime analysis and forecasting with machine learning techniques paper. In their work, they discussed that people used social media to connect with others, share ideas and content, and for professional purposes. Researchers are able to analyze individual behavior and interactions on social media sites like Facebook and Twitter. Criminology is an area of study that uses data gathered from online social media to understand criminal activity better. Researchers can obtain valuable information about crime by analyzing user-generated content and spatiotemporal linkages. This research examines 68 crime keywords to categorize crime into subgroups based on geographical and temporal data. The proposed Naive Bayes-based classification algorithm is used to classify crimes, and the Mallet package is used to retrieve keywords from news feeds. Their study identifies crime hotspots using the K-means method and uses the KDE approach to address crime density. The study found that the suggested crime forecasting model is equivalent to the ARIMA model. The comparative overview of the proposed work with existing approaches is given in <xref ref-type="table" rid="table-1">Table 1</xref>.</p>
<table-wrap id="table-1">
<label>Table 1</label>
<caption>
<title>Comparative overview of proposed work with existing approaches</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Reference</th>
<th>Technique/Model</th>
<th>Limitations</th>
<th>Improvements in proposed model (CHART)</th>
</tr>
</thead>
<tbody>
<tr>
<td>Dakalbab et al. [<xref ref-type="bibr" rid="ref-13">13</xref>]</td>
<td>Comparative analysis of AI models for crime prediction</td>
<td>Focused on reviewing existing techniques; lacks implementation and performance insights for hybrid models.</td>
<td>CHART integrates a hybrid approach (Random Forest &#x002B; KDE) with real-time tracking and hotspot detection for better accuracy.</td>
</tr>
<tr>
<td>Xie et al. [<xref ref-type="bibr" rid="ref-14">14</xref>]</td>
<td>Statistically robust clustering techniques for hotspot mapping</td>
<td>Potentially high computational cost, false-positive control challenges, limited to statistical methods.</td>
<td>CHART uses data preprocessing and efficient KDE-based heat map generation for accurate crime density visualization.</td>
</tr>
<tr>
<td>Garcia-Zanabria et al. [<xref ref-type="bibr" rid="ref-15">15</xref>]</td>
<td>CriPAV method for street-level crime pattern analysis</td>
<td>Limited to a probability-based hotspot approach, not suitable for real-time or large-scale dynamic analysis.</td>
<td>CHART offers real-time hotspot detection and spatiotemporal analysis using Random Forest and advanced feature engineering.</td>
</tr>
<tr>
<td>Al-Osaimi et al. [<xref ref-type="bibr" rid="ref-16">16</xref>]</td>
<td>Apriori algorithm for crime analysis</td>
<td>Struggles with large-scale data processing and scalability issues.</td>
<td>CHART utilizes efficient preprocessing and Random Forest to handle large datasets with faster, scalable predictions.</td>
</tr>
<tr>
<td>Wu et al. [<xref ref-type="bibr" rid="ref-17">17</xref>]</td>
<td>Place-based short-term crime prediction using human mobility</td>
<td>Limited integration of spatiotemporal dynamics, tested on specific demographics only.</td>
<td>CHART integrates spatiotemporal data and advanced crime feature extraction for more generalizable and accurate predictions.</td>
</tr>
<tr>
<td>Cardone et al. [<xref ref-type="bibr" rid="ref-18">18</xref>]</td>
<td>Fuzzy-based spatiotemporal hotspot intensity and propagation</td>
<td>Tested on a single city; limited to fuzzy techniques, lacks cross-city generalizability.</td>
<td>CHART&#x2019;s methodology is generalizable to various regions and includes kernel density estimation for crime hotspot prediction.</td>
</tr>
<tr>
<td>Appiah et al. [<xref ref-type="bibr" rid="ref-19">19</xref>]</td>
<td>Expectation-Maximization and K-means for hotspot analysis</td>
<td>Computationally expensive and lacks efficiency in real-time analysis of crime hotspots.</td>
<td>CHART improves computational efficiency with Random Forest and KDE, supporting real-time crime analysis and visualization.</td>
</tr>
<tr>
<td>Prathap et al. [<xref ref-type="bibr" rid="ref-20">20</xref>]</td>
<td>Naive Bayes classification</td>
<td>Reliant on social media data, which may not accurately represent all types of crime; focuses only on forecasts.</td>
<td>CHART integrates multiple data sources (e.g., ICT police database), providing real-time tracking and a more holistic analysis.</td>
</tr>
<tr>
<td>Malik et al. [<xref ref-type="bibr" rid="ref-21">21</xref>]</td>
<td>Navie Bayes</td>
<td>Lack of real-time application and focus on specific crimes only.</td>
<td>CHART leverages a real-time tracking system, providing broader applicability across different crime types.</td>
</tr>
<tr>
<td>Apene et al. [<xref ref-type="bibr" rid="ref-22">22</xref>]</td>
<td>Support vector machine</td>
<td>Lacks integration of advanced crime detection algorithms and data sources.</td>
<td>CHART uses advanced algorithms (SVM) and multiple data sources for more accurate and reliable predictions.</td>
</tr>
<tr>
<td>Alsubayhin et al. [<xref ref-type="bibr" rid="ref-23">23</xref>]</td>
<td>Logistic regression</td>
<td>Limited feature integration and narrow focus on classification techniques.</td>
<td>CHART offers comprehensive feature engineering and spatiotemporal analysis, improving prediction accuracy.</td>
</tr>
<tr>
<td>Aziz et al. [<xref ref-type="bibr" rid="ref-24">24</xref>]</td>
<td>Linear regression</td>
<td>Focused primarily on Indian penal code with limited international generalizability.</td>
<td>CHART&#x2019;s methodology is applicable across different regions and legal frameworks, offering broader usability.</td>
</tr>
<tr>
<td>Sharma et al. [<xref ref-type="bibr" rid="ref-25">25</xref>]</td>
<td>KNN</td>
<td>Limited to pattern detection, lacks real-time crime tracking capabilities.</td>
<td>CHART enhances crime detection with real-time tracking and hotspot prediction, providing actionable insights for law enforcement.</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>In conclusion, the extensive review of current literature underscores the significant strides made in crime prediction and hotspot detection using artificial intelligence and machine learning techniques. Various methodologies, including supervised learning, hybrid models, statistically robust clustering techniques, and geospatial analysis, have demonstrated substantial efficacy in identifying and predicting crime hotspots. The introduction of advanced algorithms, such as those leveraging human mobility data and fuzzy-based spatiotemporal techniques, highlights the innovative approaches employed to enhance crime prediction models&#x2019; accuracy and reliability. Building upon these advancements, the proposed CHART framework offers a comprehensive and superior approach to intelligent crime hotspot detection and real-time tracking. By utilizing a robust methodology that includes comprehensive data collection from the ICT police database, sophisticated data preprocessing, domain-specific feature engineering, and applying Kernel Density Estimation for heat map generation, the CHART framework effectively visualizes crime density and identifies hotspots. Incorporating a Random Forest-based model for hotspot detection further enhances the predictive accuracy and reliability of the framework.</p>
</sec>
<sec id="s3">
<label>3</label>
<title>Research Methodology</title>
<p>This section discusses the core methodology of CHART, which is majorly composed of data collection, preprocessing, feature extraction, and prediction. The proposed methodology in <xref ref-type="fig" rid="fig-1">Fig. 1</xref> combines various data sources, including police reports, public databases, and social media, to create a comprehensive crime prediction model, CHART. It starts with a robust text preprocessing pipeline that cleans and prepares data by removing URLs, converting text to lowercase, removing numbers, joining text tokens, and stripping punctuation. This clean data is then subjected to feature extraction, where domain knowledge is applied to extract relevant crime-related attributes like time, location, and type. These features are ranked and labeled to prepare for machine learning analysis. The model uses Kernel Density Estimation (KDE) to generate a heat map, visually representing crime hotspots as smooth data point distributions. Finally, a Random Forest-based approach is employed for crime detection, utilizing decision trees that aggregate predictions through majority voting or averaging to identify potential crime hotspots effectively. This comprehensive approach aims to enhance real-time crime prediction and hotspot detection, providing law enforcement with actionable insights for targeted interventions.</p>
<fig id="fig-1">
<label>Figure 1</label>
<caption>
<title>Proposed model for crime analysis and hotpot detection</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_56971-fig-1.tif"/>
</fig>
<sec id="s3_1">
<label>3.1</label>
<title>Data Collection</title>
<p>Three different dataset has been used for the evaluation of CHART. The first dataset used to train and test the model is sourced from <ext-link ext-link-type="uri" xlink:href="http://Kaggle.com">Kaggle.com</ext-link>, accessed on 16 February 2024, specifically from the &#x201C;Crime in Vancouver&#x201D; dataset. This dataset comprises two files: Crime.csv and GoogleTrend.csv. The Crime.csv dataset contains 530,652 crime records spanning from 01 January 2006, to 13 July 2021, and includes ten features: type of crime, year, month, day, hour, minute, hundredth block, and neighborhood. The GoogleTrend.csv dataset includes 185 records with two features: search value and month-year. Overall, the Vancouver crime dataset covers 14 years of crime data, featuring various types of crimes such as Theft from a Vehicle and Break and Enter Residential/Other. The neighborhoods represented in the dataset include Fairview, Victoria-Fraserview, Strathcona, Downtown, Grandview-Woodland, Kensington-Cedar Cottage, West End, Oakridge, Killarney, and Sunset. The dataset description is shown in <xref ref-type="table" rid="table-2">Table 2</xref>.</p>
<table-wrap id="table-2">
<label>Table 2</label>
<caption>
<title>Dataset description</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Dataset name</th>
<th>Weblink</th>
</tr>
</thead>
<tbody>
<tr>
<td>Crime in Vancouver (CD-I)</td>
<td>Accessed: 16 February 2024</td>
</tr>
<tr>
<td/>
<td><ext-link ext-link-type="uri" xlink:href="https://www.kaggle.com/datasets/wosaku/crime-in-vancouver">https://www.kaggle.com/datasets/wosaku/crime-in-vancouver</ext-link></td>
</tr>
<tr>
<td>ICT Police Crime Data (CD-II)</td>
<td>Accessed: 18 February 2024</td>
</tr>
<tr>
<td/>
<td><ext-link ext-link-type="uri" xlink:href="https://data.world/datasets/police">https://data.world/datasets/police</ext-link></td>
</tr>
<tr>
<td>Crime in India (CD-III)</td>
<td>Accessed: 16 February 2024</td>
</tr>
<tr>
<td/>
<td><ext-link ext-link-type="uri" xlink:href="https://github.com/vikram-bhati/PAASBAAN-crime-prediction">https://github.com/vikram-bhati/PAASBAAN-crime-prediction</ext-link></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The very next dataset consists of four entries categorized by zones: City, Saddar, Industrial Area, and Rural and contains 120,442 crime records spanning from 2009, 2021. It is sourced from the ICT police database. The dataset likely contains information related to law enforcement activities within these zones, potentially including crime statistics, incident reports, and other relevant data. With this dataset, researchers and analysts can explore and analyze the patterns, trends, and characteristics of policing and security in different ICT regions. Each entry contains information such as a unique identifier (e.g., ICT-6/14/2023-2256), the name and contact number the person reporting the crime, zone, police station, crime nature, crime type, crime location, latitude longitude, offence of, the nature of the crime (e.g., Other Crime, Robbery, Begging Act), the date and time of the report, the duration of the incident, and the current status (e.g., Pending). The third dataset includes 530,652 records of crime incidents in India, which contains detailed information about each crime. It comprises of 10 features, including type of crime, year, month, day, hour, minute, hundredth block, and neighborhood, while the file includes 185 records with two features, namely search value and month-year.</p>
</sec>
<sec id="s3_2">
<label>3.2</label>
<title>Data Preprocessing</title>
<p>Data preprocessing is an important step in any data analysis and machine learning pipeline. It involves the preparation and transformation of raw data into a format suitable for analysis and model training. Proper preprocessing ensures that the data is clean, consistent, and free from errors, directly impacting machine learning models&#x2019; performance and accuracy. This step typically includes handling missing values, correcting inconsistencies, standardizing or normalizing features, and encoding categorical variables. Additionally, feature engineering may be applied to create new, more informative variables that enhance the predictive power of the model. By ensuring data quality and relevance, preprocessing sets the foundation for building effective machine learning models.</p>
<p>Algorithm 1 shows the preliminary data preprocessing, a key step in the data analysis and machine learning pipeline that transforms raw data into a clean and usable format. This process enhances data quality by correcting errors, handling missing values, and ensuring consistency across different sources. It also improves model performance by standardizing or normalizing features, removing irrelevant information, and encoding categorical variables while creating new features.</p>
<fig id="fig-8">
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_56971-fig-8.tif"/>
</fig>
<p>Additionally, data preprocessing simplifies analysis through visualization and summary statistics, enabling the identification of patterns and trends. It ensures robust and reliable results by reducing biases and improving the model&#x2019;s ability to generalize to new data. Key steps in data preprocessing include data acquisition, which involves gathering raw data from various sources; data cleaning, where errors and inconsistencies are corrected, and missing values are handled; noise removal, which eliminates irrelevant or misleading information that could distort analysis; and normalization, where data is scaled to ensure uniformity across features. These steps work together to produce a consistent, accurate, and complete dataset, laying the groundwork for effective analysis or machine learning modeling.</p>
</sec>
<sec id="s3_3">
<label>3.3</label>
<title>Data Balancing</title>
<p>Data balancing is essential in machine learning, especially when dealing with imbalanced datasets, where one class of data significantly outnumbers another. It involves adjusting the distribution of data samples across different classes to ensure that the machine learning model learns from a representative set of examples from each class, thus improving its performance and generalization ability. One commonly used technique for data balancing is called &#x201C;oversampling&#x201D; or &#x201C;undersampling,&#x201D; which involves either increasing the number of samples in the minority class (oversampling) or reducing the number of samples in the majority class (undersampling). Here&#x2019;s an algorithm and example code for data balancing using oversampling:</p>
<p>Adaptive Synthetic Sampling (ADASYN) has been adopted in this research, a powerful data balancing technique used to address class imbalance in datasets by generating synthetic samples for the minority class [<xref ref-type="bibr" rid="ref-26">26</xref>]. ADASYN focuses on developing synthetic samples for the minority class to balance the dataset, thereby improving the performance of machine learning models. The process begins by calculating the imbalance ratio <inline-formula id="ieqn-10"><mml:math id="mml-ieqn-10"><mml:mi>&#x03C4;</mml:mi></mml:math></inline-formula> between the minority class <inline-formula id="ieqn-11"><mml:math id="mml-ieqn-11"><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and the majority class <inline-formula id="ieqn-12"><mml:math id="mml-ieqn-12"><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>: <inline-formula id="ieqn-13"><mml:math id="mml-ieqn-13"><mml:mi>&#x03C4;</mml:mi><mml:mo>=</mml:mo><mml:mstyle displaystyle="true" scriptlevel="0"><mml:mfrac><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub></mml:mfrac></mml:mstyle></mml:math></inline-formula>. Where <inline-formula id="ieqn-14"><mml:math id="mml-ieqn-14"><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and <inline-formula id="ieqn-15"><mml:math id="mml-ieqn-15"><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> are the number of samples in the minority and majority classes, respectively. ADASYN then determines the number of synthetic samples <italic>G</italic> to generate using <xref ref-type="disp-formula" rid="eqn-1">Eq. (1)</xref>.
<disp-formula id="eqn-1"><label>(1)</label><mml:math id="mml-eqn-1" display="block"><mml:mi>G</mml:mi><mml:mo>=</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>M</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow><mml:mo>)</mml:mo></mml:mrow><mml:mo>.</mml:mo><mml:mi>&#x03B2;</mml:mi></mml:math></disp-formula>where <italic>&#x03B2;</italic> is a parameter that controls the desired level of balancing. For each minority class sample <inline-formula id="ieqn-16"><mml:math id="mml-ieqn-16"><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>, ADASYN calculates the k-nearest neighbors <inline-formula id="ieqn-17"><mml:math id="mml-ieqn-17"><mml:msub><mml:mi>N</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and computes the density distribution <inline-formula id="ieqn-18"><mml:math id="mml-ieqn-18"><mml:msub><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> as shown in <xref ref-type="disp-formula" rid="eqn-2">Eq. (2)</xref>.
<disp-formula id="eqn-2"><label>(2)</label><mml:math id="mml-eqn-2" display="block"><mml:msub><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi>&#x03B4;</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mi>k</mml:mi></mml:mfrac></mml:math></disp-formula>where <inline-formula id="ieqn-19"><mml:math id="mml-ieqn-19"><mml:msub><mml:mi>&#x03B4;</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is the number of k-nearest neighbors of <inline-formula id="ieqn-20"><mml:math id="mml-ieqn-20"><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> that belong to the majority class. The probability distribution <inline-formula id="ieqn-21"><mml:math id="mml-ieqn-21"><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> for generating new samples is then given in <xref ref-type="disp-formula" rid="eqn-3">Eq. (3)</xref>.
<disp-formula id="eqn-3"><label>(3)</label><mml:math id="mml-eqn-3" display="block"><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:msub><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mrow><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mrow><mml:mo>|</mml:mo><mml:msub><mml:mi>C</mml:mi><mml:mrow><mml:mi>m</mml:mi></mml:mrow></mml:msub><mml:mo>|</mml:mo></mml:mrow></mml:mrow></mml:msubsup><mml:msub><mml:mi mathvariant="normal">&#x0394;</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
<p>The number of synthetic samples <inline-formula id="ieqn-22"><mml:math id="mml-ieqn-22"><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> to be generated for each minority sample <inline-formula id="ieqn-23"><mml:math id="mml-ieqn-23"><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> is: <inline-formula id="ieqn-24"><mml:math id="mml-ieqn-24"><mml:msub><mml:mi>g</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>G</mml:mi><mml:mo>.</mml:mo><mml:msub><mml:mi>p</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula>. New synthetic samples are created by interpolating between <inline-formula id="ieqn-25"><mml:math id="mml-ieqn-25"><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> and its k-nearest neighbors. For each synthetic sample, a random neighbor <inline-formula id="ieqn-26"><mml:math id="mml-ieqn-26"><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup></mml:math></inline-formula> is selected, and a new sample is generated using <xref ref-type="disp-formula" rid="eqn-4">Eq. (4)</xref>.
<disp-formula id="eqn-4"><label>(4)</label><mml:math id="mml-eqn-4" display="block"><mml:mi>&#x03C7;</mml:mi><mml:mo>=</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo>+</mml:mo><mml:mi>&#x03BB;</mml:mi><mml:mo>.</mml:mo><mml:mo stretchy="false">(</mml:mo><mml:msubsup><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow><mml:mrow><mml:mi>k</mml:mi></mml:mrow></mml:msubsup><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>where &#x03BB; is a random number in the range [0, 1], this approach ensures that more synthetic samples are generated for minority samples that are harder to learn, thereby enhancing the model&#x2019;s ability to generalize across different classes. By integrating ADASYN into our preprocessing pipeline, we achieve a balanced dataset that significantly improves the robustness and accuracy of our crime detection models.</p>
</sec>
<sec id="s3_4">
<label>3.4</label>
<title>Feature Engineering</title>
<p>Feature engineering is a critical step in the machine learning pipeline, laying the groundwork for building effective and robust models. It involves a combination of domain knowledge, creativity, and algorithmic techniques to extract relevant information from raw data and present it in a format that best serves the learning task at hand. In this research, three different tasks, feature extraction, feature labeling, and feature ranking, were carried out to pick the most important features. The details of each phase are as follows.</p>
<sec id="s3_4_1">
<label>3.4.1</label>
<title>Feature Extraction Process</title>
<p>Due to the sensitive nature of the data and its real-time application, the custom rules based on a domain knowledge-based feature extraction process have been adopted in this research. This involves creating features that leverage specific insights and patterns relevant to the domain of crime analysis. This process typically begins with an in-depth understanding of the domain, then identifying relevant attributes and transforming raw data into meaningful features. The first step involves collaborating with domain experts, such as criminologists or law enforcement officers, to gather insights into the patterns and characteristics of criminal activities. For instance, understanding the significance of crime types, locations, times, and demographic factors can provide a foundation for creating relevant features. Based on domain knowledge, identify attributes that are likely to influence crime patterns. Common attributes in crime data include:
<list list-type="bullet">
<list-item>
<p><bold>Time of Day</bold> (<bold>TOD</bold>)<bold>:</bold> Crimes might follow daily patterns, with different types of crimes occurring at different times.</p></list-item>
<list-item>
<p><bold>Day of Week</bold> (<bold>DOW</bold>)<bold>:</bold> Weekdays and weekends can show different crime patterns.</p></list-item>
<list-item>
<p><bold>Location Type:</bold> Different areas (residential, commercial, public spaces) might have distinct crime rates.</p></list-item>
<list-item>
<p><bold>Demographic Factors:</bold> Age, gender, and socio-economic status of the population can impact crime rates.</p></list-item>
</list></p>
<p>Use domain knowledge to transform raw attributes into meaningful features. For instance, the Time of Day features encode the time of day into categorical variables (morning, afternoon, evening, night) or use sine and cosine transformations to capture cyclical patterns, as shown in <xref ref-type="disp-formula" rid="eqn-5">Eqs. (5)</xref> and <xref ref-type="disp-formula" rid="eqn-6">(6)</xref>.
<disp-formula id="eqn-5"><label>(5)</label><mml:math id="mml-eqn-5" display="block"><mml:mi>T</mml:mi><mml:mi>O</mml:mi><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>n</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>sin</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi><mml:mo>.</mml:mo><mml:mi>h</mml:mi><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mn>24</mml:mn></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>
<disp-formula id="eqn-6"><label>(6)</label><mml:math id="mml-eqn-6" display="block"><mml:mi>T</mml:mi><mml:mi>O</mml:mi><mml:msub><mml:mi>D</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>o</mml:mi><mml:mi>s</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mi>cos</mml:mi><mml:mo>&#x2061;</mml:mo><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi><mml:mo>.</mml:mo><mml:mi>h</mml:mi><mml:mi>o</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>s</mml:mi></mml:mrow><mml:mn>24</mml:mn></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula></p>
<p>Similarly, for Day of Week, One-hot encode the day of the week to capture weekly patterns.</p>
<p>To create a composite feature, a combination of multiple attributes is applied to create composite features that capture more complex patterns. For example, for the Time and Location Interaction, crimes might have different patterns depending on both the time of day and the location. Create interaction terms to capture these effects as given in <xref ref-type="disp-formula" rid="eqn-7">Eq. (7)</xref>.
<disp-formula id="eqn-7"><label>(7)</label><mml:math id="mml-eqn-7" display="block"><mml:mi>T</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>I</mml:mi><mml:mi>n</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mi>T</mml:mi><mml:mi>O</mml:mi><mml:mi>D</mml:mi><mml:mo>&#x00D7;</mml:mo><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mi mathvariant="normal">&#x005F;</mml:mi><mml:mi>T</mml:mi><mml:mi>y</mml:mi><mml:mi>p</mml:mi><mml:mi>e</mml:mi></mml:math></disp-formula></p>
<p>Similarly, for Crime Frequency by Area, it calculates the historical crime frequency for different areas to identify hotspots. Use a moving average to smooth out short-term fluctuations as shown in <xref ref-type="disp-formula" rid="eqn-8">Eq. (8)</xref>.
<disp-formula id="eqn-8"><label>(8)</label><mml:math id="mml-eqn-8" display="block"><mml:mi>C</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:mo>.</mml:mo><mml:mi>F</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>q</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:msub><mml:mi>y</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msubsup><mml:mi>C</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:mi>e</mml:mi><mml:msub><mml:mi>s</mml:mi><mml:mrow><mml:mi>a</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>a</mml:mi></mml:mrow></mml:msub><mml:mo>,</mml:mo><mml:mi>i</mml:mi></mml:math></disp-formula>where <italic>N</italic> is the number of time periods considered.</p>
</sec>
<sec id="s3_4_2">
<label>3.4.2</label>
<title>Labeling and Ranking</title>
<p>For labeling each feature as a criminal nature, this work uses domain-specific information as described by law and order to label and categorize features. However, the categorize crime placed into low, medium, and high categories based on these information:
<disp-formula id="eqn-9"><label>(9)</label><mml:math id="mml-eqn-9" display="block"><mml:mi>C</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>t</mml:mi><mml:mi>e</mml:mi><mml:mi>g</mml:mi><mml:mi>o</mml:mi><mml:mi>r</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>=</mml:mo><mml:mrow><mml:mo>{</mml:mo><mml:mtable columnalign="left left" rowspacing=".2em" columnspacing="1em" displaystyle="false"><mml:mtr><mml:mtd><mml:mi>L</mml:mi><mml:mi>o</mml:mi><mml:mi>w</mml:mi><mml:mspace width="thinmathspace" /><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mspace width="thinmathspace" /><mml:mi>C</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>f</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>q</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x003C;</mml:mo><mml:mi>&#x03B1;</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>M</mml:mi><mml:mi>e</mml:mi><mml:mi>d</mml:mi><mml:mi>i</mml:mi><mml:mi>u</mml:mi><mml:mi>m</mml:mi><mml:mspace width="thinmathspace" /><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mspace width="thinmathspace" /><mml:mi>C</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>f</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>q</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x003C;</mml:mo><mml:mi>&#x03B2;</mml:mi></mml:mtd></mml:mtr><mml:mtr><mml:mtd><mml:mi>H</mml:mi><mml:mi>i</mml:mi><mml:mi>g</mml:mi><mml:mi>h</mml:mi><mml:mspace width="thinmathspace" /><mml:mi>i</mml:mi><mml:mi>f</mml:mi><mml:mspace width="thinmathspace" /><mml:mi>C</mml:mi><mml:mi>r</mml:mi><mml:mi>i</mml:mi><mml:mi>m</mml:mi><mml:msub><mml:mi>e</mml:mi><mml:mrow><mml:mi>f</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>q</mml:mi><mml:mi>u</mml:mi><mml:mi>e</mml:mi><mml:mi>n</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi></mml:mrow></mml:msub><mml:mo>&#x2265;</mml:mo><mml:mi>&#x03B2;</mml:mi></mml:mtd></mml:mtr></mml:mtable><mml:mo fence="true" stretchy="true" symmetric="true"></mml:mo></mml:mrow></mml:math></disp-formula>where <italic>&#x03B1;</italic> and <italic>&#x03B2;</italic> are law and forcemeat agencies&#x2019; scores against each crime determined from the data. Once the features are created, they can be used to train machine learning models for crime detection and heat map generation. It&#x2019;s crucial to evaluate the effectiveness of these features by comparing model performance with and without the custom rules-based features. Based on the labeling, the most important features, the top 10 features can be selected based on their importance scores by using:
<disp-formula id="eqn-10"><label>(10)</label><mml:math id="mml-eqn-10" display="block"><mml:mi>I</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>f</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mi>I</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>f</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>where <italic>I</italic><sub><italic>t</italic></sub>(<italic>f</italic>) is the importance of feature <italic>f</italic> in <italic>t</italic> label and <italic>N</italic> is the total number of features. Examples of extracted features are given in <xref ref-type="table" rid="table-3">Table 3</xref>.</p>
<table-wrap id="table-3">
<label>Table 3</label>
<caption>
<title>Example of some features</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Features</th>
<th>Feature type</th>
<th>Value</th>
</tr>
</thead>
<tbody>
<tr>
<td>Time of Day (TOD)</td>
<td>Categorical</td>
<td>Morning, afternoon, evening, night</td>
</tr>
<tr>
<td>Day of Week (DOW)</td>
<td>Categorical</td>
<td>Monday, tuesday, wednesday</td>
</tr>
<tr>
<td>Location type</td>
<td>Categorical</td>
<td>Commercial, residential</td>
</tr>
<tr>
<td>Crime frequency</td>
<td>Numerical</td>
<td>4.5, 7.6, etc.</td>
</tr>
<tr>
<td>Demographic factors</td>
<td>Numerical</td>
<td>Population density &#x003D; 5000</td>
</tr>
<tr>
<td>Crime severity</td>
<td>Categorical</td>
<td>High, medium, low</td>
</tr>
<tr>
<td>Weather condition</td>
<td>Categorical</td>
<td>Rainy, clear</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
</sec>
<sec id="s3_5">
<label>3.5</label>
<title>Heat Map Generation</title>
<p>The process of generating a heat map, as shown in <xref ref-type="fig" rid="fig-2">Fig. 2</xref> for crime detection and analysis, involves multiple steps, starting with data collection and ending with visualization. The first step is to gather and preprocess the spatial data, such as crime incidents, ensuring that each data point has corresponding latitude and longitude coordinates. This involves cleaning the data to remove any inconsistencies or errors and formatting it appropriately for analysis. Accurate and clean data is crucial for reliable heat map generation. The next step involves selecting the kernel function and bandwidth for Kernel Density Estimation (KDE), a non-parametric way to estimate the probability density function of a random variable. KDE is particularly useful for visualizing the intensity of events over a geographical area. The Gaussian kernel is a common choice due to its smooth and continuous nature. The kernel function can be defined as:</p>
<p><disp-formula id="eqn-11"><label>(11)</label><mml:math id="mml-eqn-11" display="block"><mml:mi>K</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mi>u</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:msqrt><mml:mn>2</mml:mn><mml:mi>&#x03C0;</mml:mi></mml:msqrt></mml:mfrac><mml:msup><mml:mi>e</mml:mi><mml:mrow><mml:mo>&#x2212;</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mn>2</mml:mn></mml:mfrac><mml:mrow><mml:msup><mml:mi>u</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mrow></mml:msup></mml:math></disp-formula></p>
<fig id="fig-2">
<label>Figure 2</label>
<caption>
<title>Heat map snippet indicating top most crimes</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_56971-fig-2.tif"/>
</fig>
<p>The bandwidth, denoted as <inline-formula id="ieqn-27"><mml:math id="mml-ieqn-27"><mml:mi>h</mml:mi></mml:math></inline-formula>, determines the level of smoothing applied to the data. It is a critical parameter as it affects the granularity of the resulting heat map. Bandwidth selection can be done using cross-validation or heuristic methods to balance between over-smoothing and under-smoothing the data.</p>
<p>Kernel Density Estimation is then used to calculate the density at each point on the map. The estimated density <inline-formula id="ieqn-28"><mml:math id="mml-ieqn-28"><mml:msup><mml:mi>f</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></inline-formula> at a point <italic>x</italic> is given by the equation:
<disp-formula id="eqn-12"><label>(12)</label><mml:math id="mml-eqn-12" display="block"><mml:msup><mml:mi>f</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mrow><mml:mo>(</mml:mo><mml:mi>x</mml:mi><mml:mo>)</mml:mo></mml:mrow><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mrow><mml:mi>n</mml:mi><mml:msup><mml:mi>h</mml:mi><mml:mrow><mml:mn>2</mml:mn></mml:mrow></mml:msup></mml:mrow></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>n</mml:mi></mml:mrow></mml:msubsup><mml:mi>K</mml:mi><mml:mrow><mml:mo>(</mml:mo><mml:mfrac><mml:mrow><mml:mi>x</mml:mi><mml:mo>&#x2212;</mml:mo><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:mrow><mml:mi>h</mml:mi></mml:mfrac><mml:mo>)</mml:mo></mml:mrow></mml:math></disp-formula>where <inline-formula id="ieqn-29"><mml:math id="mml-ieqn-29"><mml:mi>n</mml:mi></mml:math></inline-formula> is the number of data points, <inline-formula id="ieqn-30"><mml:math id="mml-ieqn-30"><mml:msub><mml:mi>x</mml:mi><mml:mrow><mml:mi>i</mml:mi></mml:mrow></mml:msub></mml:math></inline-formula> are the data points, and <italic>K</italic> is the kernel function. This equation essentially sums up the contributions of each data point to the density estimate at <inline-formula id="ieqn-31"><mml:math id="mml-ieqn-31"><mml:mi>x</mml:mi></mml:math></inline-formula>, weighted by their distance from <inline-formula id="ieqn-32"><mml:math id="mml-ieqn-32"><mml:mi>x</mml:mi></mml:math></inline-formula> as determined by the kernel function and bandwidth.</p>
<p>After calculating the density estimates, a grid is created over the geographical area of interest. Each cell in this grid represents a point where the density is estimated. The density values at these grid points are computed using the KDE formula. These values are then used to visually represent the density, typically using a color scale where higher density values (indicating crime hotspots) are shown in warmer colors such as red, and lower density values are shown in cooler colors like blue. The generated heat map visually represents crime intensity across different areas, highlighting hotspots where criminal activities are concentrated. This visualization is useful for law enforcement agencies to allocate resources more effectively, plan patrols, and implement preventive measures in high-risk areas. By adjusting the bandwidth, the smoothness of the heat map can be fine-tuned to achieve the desired resolution, balancing between too-coarse and too-detailed visualizations.</p>
</sec>
<sec id="s3_6">
<label>3.6</label>
<title>Crime Prediction</title>
<p>The final step in the crime detection model involves identifying crime hotspots using a Random Forest-based model. This method leverages the ensemble learning approach, where multiple decision trees are constructed during training. Each decision tree is trained on a subset of the data, and their predictions are aggregated to improve overall accuracy and robustness. This approach mitigates overfitting, a common problem in individual decision trees, by averaging the predictions of multiple trees.</p>
<p>The Heat Map Generation step, which utilizes Kernel Density Estimation (KDE), primarily focuses on creating visual representations of crime density. This step highlights areas where crime incidents are concentrated based on historical data, giving law enforcement a geographical view of hotspots. On the other hand, the Random Forest prediction step is responsible for predicting future crime occurrences. It takes a broader set of features, such as time, location, crime type, and demographic factors, to predict the likelihood of future crimes and identify potential new hotspots that may not be visible from historical data alone. These two processes complement each other but serve different purposes: KDE provides a spatial visualization of existing crime hotspots, while Random Forest offers predictive insights by analyzing multiple factors, enabling law enforcement to anticipate future crime hotspots. We have made these distinctions clearer in the manuscript to improve understanding of how both steps contribute to the overall crime detection framework.</p>
<p>The choice of Random Forest in the CHART framework is motivated by several key attributes that make it particularly suitable for crime hotspot detection. First, the ability of Random Forest to handle large datasets with numerous predictors is essential, as crime data often involves complex interactions among various socio-demographic and spatial factors. Random Forest effectively captures these interactions without the need for extensive data transformation that other models might require. Second, the algorithm provides an inherent feature importance measure, which is invaluable for understanding the driving forces behind crime patterns. This aspect is crucial for not only predicting where crimes are likely to occur but also for developing informed strategies to mitigate these risks. Lastly, Random Forest&#x2019;s ensemble approach, which builds multiple decision trees and aggregates their results, offers a reduction in variance and improves generalization over single predictive models. This approach minimizes overfitting&#x2014;a common problem in predictive modeling of crime data where the model performs well on training data but poorly on unseen data. The robustness provided by Random Forest is advantageous in ensuring reliable predictions that are critical for deploying law enforcement resources effectively.</p>
<p>The Random Forest algorithm begins by randomly sampling the dataset with replacement, a process known as bootstrapping. For each tree in the forest, a different subset of the data is used for training. This introduces diversity among the trees, as each tree may see a slightly different dataset. Additionally, at each split in a tree, only a random subset of features is considered for splitting. This randomness further ensures that the trees are decorrelated, making the ensemble&#x2019;s aggregated prediction more reliable.
<disp-formula id="eqn-13"><label>(13)</label><mml:math id="mml-eqn-13" display="block"><mml:msup><mml:mi>y</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup><mml:mo>=</mml:mo><mml:mfrac><mml:mn>1</mml:mn><mml:mi>N</mml:mi></mml:mfrac><mml:msubsup><mml:mo movablelimits="false">&#x2211;</mml:mo><mml:mrow><mml:mi>i</mml:mi><mml:mo>=</mml:mo><mml:mn>1</mml:mn></mml:mrow><mml:mrow><mml:mi>N</mml:mi></mml:mrow></mml:msubsup><mml:msub><mml:mi>T</mml:mi><mml:mrow><mml:mi>t</mml:mi></mml:mrow></mml:msub><mml:mo stretchy="false">(</mml:mo><mml:mi>x</mml:mi><mml:mo stretchy="false">)</mml:mo></mml:math></disp-formula>where <inline-formula id="ieqn-33"><mml:math id="mml-ieqn-33"><mml:msup><mml:mi>y</mml:mi><mml:mrow><mml:mi mathvariant="normal">&#x2032;</mml:mi></mml:mrow></mml:msup></mml:math></inline-formula> is the predicted output, NNN is the number of trees in the forest, and <italic>T</italic><sub><italic>i</italic></sub>(<italic>x</italic>) is the prediction of the <italic>i</italic>-th tree for input <italic>x</italic>. By aggregating the predictions, usually through majority voting for classification tasks or averaging for regression tasks, the Random Forest model produces a more accurate and stable prediction compared to individual decision trees. The preprocessed and balanced dataset serves as the training ground for the Random Forest model. Important features, such as time of day, location type, and historical crime frequency, are used as inputs. These features are derived from earlier steps in the methodology, including data collection, preprocessing, and feature engineering. By incorporating domain knowledge and ensuring a balanced dataset, the model is better equipped to learn the complex patterns associated with crime incidents.</p>
<p>Once trained, the Random Forest model can predict the likelihood of crime incidents in different areas, effectively identifying hotspots. These predictions can be visualized on a map, where areas with higher predicted crime rates are marked as hotspots. This spatial representation allows law enforcement agencies to focus their resources on areas that are more prone to criminal activities, enhancing their ability to prevent and respond to crimes. The effectiveness of the Random Forest-based hotspot detection is evaluated using various metrics, such as accuracy, precision, recall, and F1-score. By comparing the model&#x2019;s performance on a validation set, the parameters and structure of the Random Forest can be fine-tuned to achieve optimal results. This iterative process ensures that the model remains robust and accurate, providing valuable insights into crime patterns and aiding in the development of targeted crime prevention strategies.</p>
<p>The Random Forest algorithm, central to our CHART framework, is designed to handle complex datasets with multiple interacting features. It operates by constructing a collection of decision trees, each trained on a random subset of the data. These trees independently analyze various input features such as crime type, location, time, and demographic information. The model&#x2019;s input features include categorical variables like crime type (e.g., theft, assault), spatial data like geospatial coordinates or neighborhood identifiers, temporal data such as the date, time of day, and day of the week, and demographic data, if available, like population density or socio-economic status. Each tree in the forest generates a prediction based on a different combination of these features, helping the model capture a wide range of patterns within the data. The algorithm manages predictions by aggregating the outputs of all trees through majority voting, which ensures that the final prediction is less sensitive to the biases of individual trees. This reduces overfitting, making the model more robust and generalizable.</p>
<p>The expected outputs of the Random Forest model are probability scores indicating the likelihood of future crimes occurring in specific locations at particular times. The final classification identifies whether an area is a potential crime hotspot. For example, if the data reveals frequent thefts in a neighborhood during late evening hours, the Random Forest will recognize this pattern and predict similar occurrences in the future, allowing law enforcement to allocate resources proactively. This method effectively handles the non-linear interactions between spatial, temporal, and social variables, providing accurate and actionable predictions in crime hotspot detection. By leveraging the ensemble nature of Random Forest, the CHART framework enhances prediction accuracy while offering valuable insights into which factors most influence crime occurrences.</p>
</sec>
</sec>
<sec id="s4">
<label>4</label>
<title>Results and Evaluation</title>
<p>This section presents the study&#x2019;s results on crime analysis and hotspot prediction using map mark points in a safe city context.</p>
<p>Accuracy, precision, recall and F1 are prescribed measure that has been used to evaluate the effectiveness of classification models, as given in <xref ref-type="disp-formula" rid="eqn-14">Eqs. (14)</xref>&#x2013;<xref ref-type="disp-formula" rid="eqn-16">(16)</xref>. These measures provide insights into different aspects of the model&#x2019;s performance.
<disp-formula id="eqn-14"><label>(14)</label><mml:math id="mml-eqn-14" display="block"><mml:mi>A</mml:mi><mml:mi>c</mml:mi><mml:mi>c</mml:mi><mml:mi>u</mml:mi><mml:mi>r</mml:mi><mml:mi>a</mml:mi><mml:mi>c</mml:mi><mml:mi>y</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>T</mml:mi><mml:mi>N</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula>
<disp-formula id="eqn-15"><label>(15)</label><mml:math id="mml-eqn-15" display="block"><mml:mi>p</mml:mi><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>i</mml:mi><mml:mi>s</mml:mi><mml:mi>i</mml:mi><mml:mi>o</mml:mi><mml:mi>n</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>P</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula>
<disp-formula id="eqn-16"><label>(16)</label><mml:math id="mml-eqn-16" display="block"><mml:mi>r</mml:mi><mml:mi>e</mml:mi><mml:mi>c</mml:mi><mml:mi>a</mml:mi><mml:mi>l</mml:mi><mml:mi>l</mml:mi><mml:mo>=</mml:mo><mml:mfrac><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi></mml:mrow><mml:mrow><mml:mi>T</mml:mi><mml:mi>P</mml:mi><mml:mo>+</mml:mo><mml:mi>F</mml:mi><mml:mi>N</mml:mi></mml:mrow></mml:mfrac></mml:math></disp-formula></p>
<p>This section presents the experimental results of the proposed model on a crime dataset on different datasets. The dataset is divided into 70% training and 30% testing. The training set is used to fit the model and learn the underlying crime patterns, while the test set is reserved for evaluating the model&#x2019;s predictive performance on unseen data. This split ratio was chosen based on common machine learning practices and provides a balanced approach to train the model effectively while preserving enough data to assess its generalization capabilities. The details of hyperparameters are shown in <xref ref-type="table" rid="table-4">Table 4</xref>.</p>
<table-wrap id="table-4">
<label>Table 4</label>
<caption>
<title>Model hyperparameters</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Hyperparameter</th>
<th>Values</th>
</tr>
</thead>
<tbody>
<tr>
<td>Bootstrap</td>
<td>True</td>
</tr>
<tr>
<td>Class weight</td>
<td>Balanced</td>
</tr>
<tr>
<td>n_estimators</td>
<td>500</td>
</tr>
<tr>
<td>Max_depth</td>
<td>20</td>
</tr>
<tr>
<td>min_samples_split</td>
<td>5</td>
</tr>
<tr>
<td>min_samples_leaf</td>
<td>2</td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The evaluation of the proposed approach was based on three key metrics: accuracy, precision, and recall. The results, as shown in <xref ref-type="fig" rid="fig-3">Fig. 3</xref>, demonstrate the effectiveness of the model in analyzing and predicting crime patterns in Islamabad.</p>
<fig id="fig-3">
<label>Figure 3</label>
<caption>
<title>Performance of CHART on different dataset</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_56971-fig-3.tif"/>
</fig>
<p>The confusion matrix results for datasets CD-I, CD-II, and CD-III are illustrated in <xref ref-type="fig" rid="fig-4">Fig. 4</xref>, showing the model&#x2019;s classification performance across the three datasets. Each confusion matrix highlights the number of true positives, false positives, true negatives, and false negatives, providing a detailed view of how well the model distinguishes between different classes in each dataset.</p>
<fig id="fig-4">
<label>Figure 4</label>
<caption>
<title>Confusion matrix on CD-I, CD-II and CD-III</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_56971-fig-4.tif"/>
</fig>
<p>In the absence of direct ground truth data, we evaluated the effectiveness of the proposed method using proxy ground truth derived from historical crime patterns recorded in the Islamabad Police dataset. This dataset spans multiple years and includes detailed crime attributes such as location, time, and crime type. To assess the model&#x2019;s accuracy, we compared the predicted crime hotspots with historical crime data, achieving a 92.8% match between the model&#x2019;s predictions and actual crime concentrations from the Islamabad Police dataset. Additionally, domain experts from local law enforcement evaluated the predictions, confirming the relevance of the identified hotspots with an expert agreement rate of 90.2%. We further validated the method using unsupervised evaluation metrics like the silhouette score, which resulted in a score of 0.92, indicating well-clustered and distinct crime hotspots. These combined evaluations demonstrate the model&#x2019;s reliability and effectiveness in predicting crime patterns in the absence of explicit ground truth.</p>
<p>The performance of CHART has been compared with several well-known machine learning algorithms, including Naive Bayes, Support Vector Machine (SVM), Logistic Regression, Linear Regression, and K-Nearest Neighbors (KNN). Naive Bayes is a probabilistic classifier based on Bayes&#x2019; theorem, assuming independence between features. Support Vector Machine (SVM) is a classifier that identifies the hyperplane that best separates the data into different classes. Logistic Regression is a statistical model that uses a logistic function to model a binary dependent variable. Linear Regression uses a linear equation to estimate the relationship between a dependent variable and one or more independent variables. K-Nearest Neighbors (KNN) is a non-parametric algorithm that classifies a data point based on the classifications of its nearest neighbors. The performance of these algorithms was assessed based on three key metrics: Accuracy, Precision, and Recall. The results, summarized in <xref ref-type="table" rid="table-5">Table 5</xref>, demonstrate the effectiveness of the proposed approach compared to these traditional classifiers.</p>
<table-wrap id="table-5">
<label>Table 5</label>
<caption>
<title>Comparative performance of CHART with different classifiers</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Classifier</th>
<th>Accuracy</th>
<th>Precision</th>
<th>Recall</th>
</tr>
</thead>
<tbody>
<tr>
<td>Navie bayes</td>
<td>84.32%</td>
<td>81.45%</td>
<td>83.21%</td>
</tr>
<tr>
<td>Support vector machine</td>
<td>88.47%</td>
<td>86.53%</td>
<td>87.29%</td>
</tr>
<tr>
<td>Logistic regression</td>
<td>86.29%</td>
<td>84.74%</td>
<td>85.10%</td>
</tr>
<tr>
<td>Linear regression</td>
<td>82.76%</td>
<td>80.32%</td>
<td>81.94%</td>
</tr>
<tr>
<td>KNN</td>
<td>85.61%</td>
<td>83.28%</td>
<td>84.56%</td>
</tr>
<tr>
<td><bold>CHART</bold> (<bold>Proposed</bold>)</td>
<td><bold>95.65%</bold></td>
<td><bold>93.87%</bold></td>
<td><bold>94.56%</bold></td>
</tr>
</tbody>
</table>
</table-wrap>
<p>The results highlight the superior performance of the proposed approach in comparison to the other classifiers. Specifically, the proposed approach achieved an accuracy of 95.65%, significantly higher than the other classifiers. The closest contender, SVM, achieved an accuracy of 88.47%, demonstrating the robustness of the proposed method. With a precision of 93.87%, the proposed approach outperformed all other classifiers, with SVM and Logistic Regression following at 86.53% and 84.74%, respectively. The recall of the proposed approach was 94.56%, indicating its effectiveness in identifying true positive cases, and notably higher than the recall values for SVM (87.29%) and Logistic Regression (85.10%). The comparative analysis indicates that the proposed approach not only surpasses traditional machine learning algorithms in terms of accuracy but also excels in precision and recall. This high performance can be attributed to the model&#x2019;s ability to effectively learn from the crime data, capturing the intricate patterns and distributions associated with various crime types and locations within Islamabad. By achieving higher precision, the proposed approach ensures that the majority of the identified crime hotspots are indeed areas with high crime rates, reducing false positives. Similarly, the high recall value ensures that most high-crime areas are correctly identified, minimizing false negatives. Overall, the results validate the effectiveness of the proposed approach as a reliable tool for crime prediction and analysis, outperforming conventional classifiers and demonstrating its potential for broader applications in urban planning and law enforcement.</p>
<p>The ablation study conducted highlights the contributions of each key component of the CHART framework by evaluating performance metrics such as accuracy, precision, and recall as shown in <xref ref-type="table" rid="table-6">Table 6</xref>. The full CHART model, which integrates Kernel Density Estimation (KDE), Random Forest, ADASYN for data balancing, and advanced feature engineering, achieves the highest overall performance with 95.65% accuracy, 93.87% precision, and 94.56% recall. When ADASYN is removed from the model (KDE &#x002B; Random Forest without ADASYN), the performance declines, with accuracy dropping to 91.12%, demonstrating the importance of addressing class imbalance in crime data. Models relying solely on KDE or Random Forest perform worse, achieving 88.47% and 90.12% accuracy, respectively, highlighting the effectiveness of combining these approaches. Further, using KDE with unbalanced data results in the lowest performance (84.76% accuracy), underscoring the need for data balancing. For comparison, spatio-temporal KDE, as proposed by Hu et al., shows moderate performance with 86.78% accuracy, but it does not match the robustness of the CHART framework. This study illustrates the necessity of combining multiple advanced techniques to achieve superior performance in crime hotspot detection.</p>
<table-wrap id="table-6">
<label>Table 6</label>
<caption>
<title>Ablation study</title>
</caption>
<table frame="hsides">
<colgroup>
<col align="left"/>
<col align="left"/>
<col align="left"/>
<col align="left"/>
</colgroup>
<thead>
<tr>
<th>Model</th>
<th>Accuracy (%)</th>
<th>Precision (%)</th>
<th>Recall (%)</th>
</tr>
</thead>
<tbody>
<tr>
<td>CHART (Full model)</td>
<td>95.65</td>
<td>93.87</td>
<td>94.56</td>
</tr>
<tr>
<td>KDE &#x002B; Random forest (No ADASYN)</td>
<td>91.12</td>
<td>89.45</td>
<td>89.98</td>
</tr>
<tr>
<td>KDE only</td>
<td>88.47</td>
<td>86.53</td>
<td>87.29</td>
</tr>
<tr>
<td>Random forest only</td>
<td>90.12</td>
<td>88.32</td>
<td>88.87</td>
</tr>
<tr>
<td>KDE (Unbalanced data) [<xref ref-type="bibr" rid="ref-27">27</xref>]</td>
<td>84.76</td>
<td>81.23</td>
<td>82.90</td>
</tr>
<tr>
<td>Spatio-temporal KDE [<xref ref-type="bibr" rid="ref-28">28</xref>]</td>
<td>86.78</td>
<td>84.91</td>
<td>85.33</td>
</tr>
</tbody>
</table>
</table-wrap>
</sec>
<sec id="s5">
<label>5</label>
<title>Workable Application of CHART</title>
<p>The web-based application has been developed to provide an interactive platform for analyzing and visualizing crime data. The dashboard, displaying data from police calls has been shown in <xref ref-type="fig" rid="fig-5">Fig. 5</xref>, allows users to filter by Police Circle, Police Station, and Criminals for focused analysis. It categorizes crimes into types such as Assault, Burglary, Kidnapping, and Theft. The dashboard shows a total of 45.29 K police calls recorded between 01 January 2024 and 30 May 2024. Users can filter data by Police Circle, Police Station, and Criminals, allowing for a more focused analysis. This also helps in understanding temporal patterns and spatial distribution, aiding in effective police deployment during high-risk times. This application is valuable for law enforcement and city planners, enabling data-driven decisions to enhance public safety and crime prevention. The application also provides real-time feedback and insights, helping to strategize patrol routes, allocate resources efficiently, and design safer urban layouts.</p>
<fig id="fig-5">
<label>Figure 5</label>
<caption>
<title>Hotspot priority wise distribution using CHART</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_56971-fig-5.tif"/>
</fig>
<p>The hotspot identification interface has been shown in <xref ref-type="fig" rid="fig-6">Figs. 6</xref> and <xref ref-type="fig" rid="fig-7">7</xref> provides a clear and detailed visualization of crime hotspots and trends using Safe City mark points. By pinpointing these high-risk areas, law enforcement agencies can deploy police resources more effectively, ensuring a more strategic and efficient approach to crime prevention. The spatial distribution highlights police stations with the highest and lowest crime rates, facilitating targeted interventions in areas with higher crime concentrations.</p>
<fig id="fig-6">
<label>Figure 6</label>
<caption>
<title>CHART based hotspot detection</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_56971-fig-6.tif"/>
</fig><fig id="fig-7">
<label>Figure 7</label>
<caption>
<title>Hotspot detection of week days</title>
</caption>
<graphic mimetype="image" mime-subtype="tif" xlink:href="CMC_56971-fig-7.tif"/>
</fig>
<p>This allows for a more focused allocation of resources to the areas that need them the most, enhancing the overall safety of the community. Additionally, the dashboard provides insights into the most common types of crimes and their hotspots, which assists law enforcement agencies in prioritizing specific crime prevention strategies.</p>
</sec>
<sec id="s6">
<label>6</label>
<title>Conclusion and Future Work</title>
<p>This research proposes an advanced machine learning framework, CHART (Crime Hotspot Analysis and Real-time Tracking), designed to address the challenges in crime prediction and prevention. The methodology begins with comprehensive data collection from the ICT police database, encompassing detailed attributes such as crime type, location, time, and demographic information. Key steps include data preprocessing, domain-specific feature engineering, heat map generation using Kernel Density Estimation (KDE), and hotspot detection with a Random Forest model to predict crime likelihood in various areas. Experimental evaluation demonstrated CHART&#x2019;s superior performance over benchmark methods, significantly improving crime detection accuracy, achieving 95.24% for CD-I, 96.12% for CD-II, and 94.68% for CD-III. The integration of sophisticated preprocessing techniques, balanced data representation, and advanced feature engineering ensures the model&#x2019;s reliability and practicality for real-world crime analysis. Visualization of crime hotspots allows law enforcement agencies to strategize effectively, focusing resources on high-risk areas to enhance overall crime prevention and response efforts. Future work will explore incorporating real-time data streams, enhancing model adaptability to emerging crime patterns, and integrating additional contextual factors such as socio-economic indicators to further improve prediction accuracy and utility.</p>
</sec>
</body>
<back>
<ack>
<p>The authors extend their appreciation to King Saud University for funding this work.</p>
</ack>
<sec><title>Funding Statement</title>
<p>The authors extend their appreciation to King Saud University for funding this work through Researchers Supporting Project number (RSPD2025R685), King Saud University, Riyadh, Saudi Arabia.</p>
</sec>
<sec><title>Author Contributions</title>
<p>Rashid Ahmad: Paper Writing; Asif Nawaz: Methodology; Ghulam Mustafa: Formal Analysis; Tariq Ali: Experimental Analysis; Mehdi Tlija: Writing&#x2014;Review &#x0026; Editing; Mohammed A. El-Meligy: Supervision; Zohair Ahmed: Study Conception and Design. All authors reviewed the results and approved the final version of the manuscript.</p>
</sec>
<sec sec-type="data-availability"><title>Availability of Data and Materials</title>
<p>Not available.</p>
</sec>
<sec><title>Ethics Approval</title>
<p>Not applicable.</p>
</sec>
<sec sec-type="COI-statement"><title>Conflicts of Interest</title>
<p>The authors declare no conflicts of interest to report regarding the present study.</p>
</sec>
<ref-list content-type="authoryear">
<title>References</title>
<ref id="ref-1"><label>[1]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C. S.</given-names> <surname>Ogbodo</surname></string-name></person-group>, &#x201C;<article-title>Law, human rights, crime and society</article-title>,&#x201D; <source>Afr. Hum. Rights Law J.</source>, vol. <volume>8</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>15</lpage>, <year>Jan. 2024</year>.</mixed-citation></ref>
<ref id="ref-2"><label>[2]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>F. A.</given-names> <surname>Paul</surname></string-name>, <string-name><given-names>A. A.</given-names> <surname>Dangroo</surname></string-name>, and <string-name><given-names>P.</given-names> <surname>Saikia</surname></string-name></person-group>, &#x201C;<article-title>Societal and individual impacts of substance abuse</article-title>,&#x201D; <source>J. Health Soc. Behav.</source>, vol. <volume>10</volume>, no. <issue>2</issue>, pp. <fpage>25</fpage>&#x2013;<lpage>40</lpage>, <year>Feb. 2024</year>. doi: <pub-id pub-id-type="doi">10.1007/978-3-030-68127-2</pub-id>.</mixed-citation></ref>
<ref id="ref-3"><label>[3]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>D.</given-names> <surname>Birks</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Coleman</surname></string-name>, and <string-name><given-names>D.</given-names> <surname>Jackson</surname></string-name></person-group>, &#x201C;<article-title>Unsupervised identification of crime problems from police free-text data</article-title>,&#x201D; <source>Crime Sci.</source>, vol. <volume>9</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>18</lpage>, <year>2020</year>. doi: <pub-id pub-id-type="doi">10.1186/s40163-020-00127-4</pub-id>.</mixed-citation></ref>
<ref id="ref-4"><label>[4]</label><mixed-citation publication-type="book"><person-group person-group-type="author"><string-name><given-names>M. J.</given-names> <surname>Fortner</surname></string-name> and <string-name><given-names>C.</given-names> <surname>Stevens</surname></string-name></person-group>, <source>Crime, Punishment, and Urban Criminal Justice Systems in the United States</source>, <publisher-loc>Cheltenham, UK</publisher-loc>: <publisher-name>Edward Elgar Publishing</publisher-name>, <year>2024</year>.</mixed-citation></ref>
<ref id="ref-5"><label>[5]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>N.</given-names> <surname>Shah</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Bhagat</surname></string-name>, and <string-name><given-names>M.</given-names> <surname>Shah</surname></string-name></person-group>, &#x201C;<article-title>Crime forecasting: A machine learning and computer vision approach to crime prediction and prevention</article-title>,&#x201D; <source>Vis. Comput. Ind., Biomed., Art</source>, vol. <volume>4</volume>, no. <issue>1</issue>, pp. <fpage>9</fpage>&#x2013;<lpage>18</lpage>, <year>2021</year>; <pub-id pub-id-type="pmid">33913057</pub-id></mixed-citation></ref>
<ref id="ref-6"><label>[6]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>I.</given-names> <surname>Fedchak</surname></string-name>, <string-name><given-names>O.</given-names> <surname>Kondrat&#x0456;uk</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Movchan</surname></string-name>, and <string-name><given-names>S.</given-names> <surname>Poliak</surname></string-name></person-group>, &#x201C;<article-title>Theoretical foundations of hot spots policing and crime mapping features</article-title>,&#x201D; <source>Soc. Legal Stud.</source>, vol. <volume>1</volume>, no. <issue>7</issue>, pp. <fpage>174</fpage>&#x2013;<lpage>183</lpage>, <year>Jan. 2024</year>. doi: <pub-id pub-id-type="doi">10.32518/sals1.2024.174</pub-id>.</mixed-citation></ref>
<ref id="ref-7"><label>[7]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>Yan</surname></string-name>, <string-name><given-names>W.</given-names> <surname>Quan</surname></string-name>, and <string-name><given-names>H.</given-names> <surname>Wang</surname></string-name></person-group>, &#x201C;<article-title>A data-driven adaptive geospatial hotspot detection approach in smart cities</article-title>,&#x201D; <source>Trans. GIS</source>, vol. <volume>28</volume>, no. <issue>2</issue>, pp. <fpage>303</fpage>&#x2013;<lpage>325</lpage>, <year>Feb. 2024</year>. doi: <pub-id pub-id-type="doi">10.1111/tgis.13137</pub-id>.</mixed-citation></ref>
<ref id="ref-8"><label>[8]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Mukherjee</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Saha</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Karmakar</surname></string-name>, and <string-name><given-names>P.</given-names> <surname>Dash</surname></string-name></person-group>, &#x201C;<article-title>Uncovering spatial patterns of crime: A case study of Kolkata</article-title>,&#x201D; <source>Crime Prev. Community Saf.</source>, vol. <volume>26</volume>, no. <issue>1</issue>, pp. <fpage>47</fpage>&#x2013;<lpage>90</lpage>, <year>Jan. 2024</year>. doi: <pub-id pub-id-type="doi">10.1057/s41300-024-00198-4</pub-id>.</mixed-citation></ref>
<ref id="ref-9"><label>[9]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>I.</given-names> <surname>Debata</surname></string-name>, <string-name><given-names>P. S.</given-names> <surname>Panda</surname></string-name>, <string-name><given-names>E.</given-names> <surname>Karthikeyan</surname></string-name>, and <string-name><given-names>J.</given-names> <surname>Tejas</surname></string-name></person-group>, &#x201C;<article-title>Spatial auto-correlation and endemicity pattern analysis of crimes against children in Tamil Nadu from 2017 to 2021</article-title>,&#x201D; <source>J. Fam. Med. Prim. Care</source>, vol. <volume>13</volume>, no. <issue>6</issue>, pp. <fpage>2341</fpage>&#x2013;<lpage>2347</lpage>, <year>Jun. 2024</year>. doi: <pub-id pub-id-type="doi">10.4103/jfmpc.jfmpc_1463_23</pub-id>; <pub-id pub-id-type="pmid">39027864</pub-id></mixed-citation></ref>
<ref id="ref-10"><label>[10]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>C.</given-names> <surname>Massarelli</surname></string-name> and <string-name><given-names>V. F.</given-names> <surname>Uricchio</surname></string-name></person-group>, &#x201C;<article-title>The contribution of open source software in identifying environmental crimes caused by illicit waste management in urban areas</article-title>,&#x201D; <source>Urban Sci.</source>, vol. <volume>8</volume>, no. <issue>1</issue>, <year>Jan. 2024, Art. no. 21</year>. doi: <pub-id pub-id-type="doi">10.3390/urbansci8010021</pub-id>.</mixed-citation></ref>
<ref id="ref-11"><label>[11]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A. K.</given-names> <surname>Selvan</surname></string-name> and <string-name><given-names>N.</given-names> <surname>Sivakumaran</surname></string-name></person-group>, &#x201C;<article-title>Crime detection and crime hot spot prediction using the BI-LSTM deep learning model</article-title>,&#x201D; <source>Int. J. Knowl.-Based Dev.</source>, vol. <volume>14</volume>, no. <issue>1</issue>, pp. <fpage>57</fpage>&#x2013;<lpage>86</lpage>, <year>Jan. 2024</year>. doi: <pub-id pub-id-type="doi">10.1504/IJKBD.2024.137600</pub-id>.</mixed-citation></ref>
<ref id="ref-12"><label>[12]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. R.</given-names> <surname>Bandekar</surname></string-name> and <string-name><given-names>C.</given-names> <surname>Vijayalakshmi</surname></string-name></person-group>, &#x201C;<article-title>Design and analysis of machine learning algorithms for the reduction of crime rates in India</article-title>,&#x201D; <source>Procedia Comput. Sci.</source>, vol. <volume>172</volume>, no. <issue>1</issue>, pp. <fpage>122</fpage>&#x2013;<lpage>127</lpage>, <year>2020</year>. doi: <pub-id pub-id-type="doi">10.1016/j.procs.2020.05.018</pub-id>.</mixed-citation></ref>
<ref id="ref-13"><label>[13]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>F.</given-names> <surname>Dakalbab</surname></string-name>, <string-name><given-names>M. A.</given-names> <surname>Talib</surname></string-name>, <string-name><given-names>O. A.</given-names> <surname>Waraga</surname></string-name>, <string-name><given-names>A. B.</given-names> <surname>Nassif</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Abbas</surname></string-name> and <string-name><given-names>Q.</given-names> <surname>Nasir</surname></string-name></person-group>, &#x201C;<article-title>Artificial intelligence crime prediction: A systematic literature review</article-title>,&#x201D; <source>Social Sci. Human.</source>, vol. <volume>6</volume>, no. <issue>1</issue>, <year>2022, Art. no. 100342</year>.</mixed-citation></ref>
<ref id="ref-14"><label>[14]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>Xie</surname></string-name>, <string-name><given-names>S.</given-names> <surname>Shekhar</surname></string-name>, and <string-name><given-names>Y.</given-names> <surname>Li</surname></string-name></person-group>, &#x201C;<article-title>Statistically-robust clustering techniques for mapping spatial hotspots: A survey</article-title>,&#x201D; <source>ACM Comput. Surv.</source>, vol. <volume>55</volume>, no. <issue>2</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>38</lpage>, <year>Feb. 2022</year>.</mixed-citation></ref>
<ref id="ref-15"><label>[15]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>G.</given-names> <surname>Garcia-Zanabria</surname></string-name> <etal>et al.</etal></person-group>, &#x201C;<article-title>CriPAV: Street-level crime patterns analysis and visualization</article-title>,&#x201D; <source>IEEE Trans. Vis. Comput. Graph.</source>, vol. <volume>28</volume>, no. <issue>12</issue>, pp. <fpage>4000</fpage>&#x2013;<lpage>4015</lpage>, <year>Dec. 2021</year>. doi: <pub-id pub-id-type="doi">10.1109/TVCG.2021.3111146</pub-id>; <pub-id pub-id-type="pmid">34516376</pub-id></mixed-citation></ref>
<ref id="ref-16"><label>[16]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. K.</given-names> <surname>Al-Osaimi</surname></string-name></person-group>, &#x201C;<article-title>Survey of crime data analysis using the Apriori algorithm</article-title>,&#x201D; <source>Know.-Based Syst.</source>, vol. <volume>10</volume>, no. <issue>1</issue>, pp. <fpage>454</fpage>&#x2013;<lpage>498</lpage>, <year>2021</year>.</mixed-citation></ref>
<ref id="ref-17"><label>[17]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>J.</given-names> <surname>Wu</surname></string-name>, <string-name><given-names>S. M.</given-names> <surname>Abrar</surname></string-name>, <string-name><given-names>N.</given-names> <surname>Awasthi</surname></string-name>, <string-name><given-names>E.</given-names> <surname>Frias-Martinez</surname></string-name>, and <string-name><given-names>V.</given-names> <surname>Frias-Martinez</surname></string-name></person-group>, &#x201C;<article-title>Enhancing short-term crime prediction with human mobility flows and deep learning architectures</article-title>,&#x201D; <source>EPJ Data Sci.</source>, vol. <volume>11</volume>, no. <issue>1</issue>, <year>Feb. 2022, Art. no. 53</year>. doi: <pub-id pub-id-type="doi">10.1140/epjds/s13688-022-00366-2</pub-id>; <pub-id pub-id-type="pmid">36406335</pub-id></mixed-citation></ref>
<ref id="ref-18"><label>[18]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>B.</given-names> <surname>Cardone</surname></string-name> and <string-name><given-names>F. Di</given-names> <surname>Martino</surname></string-name></person-group>, &#x201C;<article-title>Fuzzy-based spatiotemporal hot spot intensity and propagation&#x2014;an application in crime analysis</article-title>,&#x201D; <source>Electronics</source>, vol. <volume>11</volume>, no. <issue>3</issue>, <year>Feb. 2022, Art. no. 370</year>. doi: <pub-id pub-id-type="doi">10.3390/electronics11030370</pub-id>.</mixed-citation></ref>
<ref id="ref-19"><label>[19]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>S. K.</given-names> <surname>Appiah</surname></string-name>, <string-name><given-names>K.</given-names> <surname>Wirekoh</surname></string-name>, <string-name><given-names>E. N.</given-names> <surname>Aidoo</surname></string-name>, <string-name><given-names>S. D.</given-names> <surname>Oduro</surname></string-name>, and <string-name><given-names>Y. D.</given-names> <surname>Arthur</surname></string-name></person-group>, &#x201C;<article-title>A model-based clustering of expectation-maximization and K-means algorithms in crime hotspot analysis</article-title>,&#x201D; <source>Res. Math.</source>, vol. <volume>9</volume>, no. <issue>1</issue>, <year>2022, Art. no. 2073662</year>. doi: <pub-id pub-id-type="doi">10.1080/27684830.2022.2073662</pub-id>.</mixed-citation></ref>
<ref id="ref-20"><label>[20]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>B. R.</given-names> <surname>Prathap</surname></string-name></person-group>, &#x201C;<article-title>Geospatial crime analysis and forecasting with machine learning techniques</article-title>,&#x201D; <source>Artif. Intell. Mach. Learn. EDGE Comput.</source>, vol. <volume>1</volume>, no. <issue>1</issue>, pp. <fpage>87</fpage>&#x2013;<lpage>102</lpage>, <year>2022</year>. doi: <pub-id pub-id-type="doi">10.1016/B978-0-12-824054-0.00008-3</pub-id>.</mixed-citation></ref>
<ref id="ref-21"><label>[21]</label><mixed-citation publication-type="conf-proc"><person-group person-group-type="author"><string-name><given-names>K.</given-names> <surname>Malik</surname></string-name>, <string-name><given-names>M.</given-names> <surname>Pandey</surname></string-name>, <string-name><given-names>A.</given-names> <surname>Khan</surname></string-name>, and <string-name><given-names>M.</given-names> <surname>Srivastav</surname></string-name></person-group>, &#x201C;<article-title>Crime prediction by comparing machine learning and deep learning algorithms</article-title>,&#x201D; in <conf-name>2024 2nd Int. Conf. Disrup. Technol. (ICDT)</conf-name>, <publisher-loc>Noida, India</publisher-loc>, <year>Mar. 2024</year>, pp. <fpage>215</fpage>&#x2013;<lpage>219</lpage>.</mixed-citation></ref>
<ref id="ref-22"><label>[22]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>O. Z.</given-names> <surname>Apene</surname></string-name>, <string-name><given-names>N. V.</given-names> <surname>Blamah</surname></string-name>, and <string-name><given-names>G. I. O.</given-names> <surname>Aimufua</surname></string-name></person-group>, &#x201C;<article-title>Advancements in crime prevention and detection: From traditional approaches to artificial intelligence solutions</article-title>,&#x201D; <source>Euro. J. Appl. Sci., Eng. Technol.</source>, vol. <volume>2</volume>, no. <issue>2</issue>, pp. <fpage>285</fpage>&#x2013;<lpage>297</lpage>, <year>Feb. 2024</year>. doi: <pub-id pub-id-type="doi">10.59324/ejaset.2024.2(2).20</pub-id>.</mixed-citation></ref>
<ref id="ref-23"><label>[23]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Alsubayhin</surname></string-name>, <string-name><given-names>M. S.</given-names> <surname>Ramzan</surname></string-name>, and <string-name><given-names>B.</given-names> <surname>Alzahrani</surname></string-name></person-group>, &#x201C;<article-title>Crime prediction model using three classification techniques: Random forest, logistic regression, and LightGBM</article-title>,&#x201D; <source>Int. J. Adv. Comput. Sci. Appl.</source>, vol. <volume>15</volume>, no. <issue>1</issue>, pp. <fpage>698</fpage>&#x2013;<lpage>717</lpage>, <year>Jan. 2024</year>. doi: <pub-id pub-id-type="doi">10.14569/issn.2156-5570</pub-id>.</mixed-citation></ref>
<ref id="ref-24"><label>[24]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>R. M.</given-names> <surname>Aziz</surname></string-name>, <string-name><given-names>P.</given-names> <surname>Sharma</surname></string-name>, and <string-name><given-names>A.</given-names> <surname>Hussain</surname></string-name></person-group>, &#x201C;<article-title>Machine learning algorithms for crime prediction under Indian penal code</article-title>,&#x201D; <source>Ann. Data Sci.</source>, vol. <volume>11</volume>, no. <issue>1</issue>, pp. <fpage>379</fpage>&#x2013;<lpage>410</lpage>, <year>2024</year>. doi: <pub-id pub-id-type="doi">10.1007/s40745-022-00424-6</pub-id>.</mixed-citation></ref>
<ref id="ref-25"><label>[25]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Sharma</surname></string-name>, <string-name><given-names>R.</given-names> <surname>Agarwal</surname></string-name>, and <string-name><given-names>A. M.</given-names> <surname>Nancy</surname></string-name></person-group>, &#x201C;<article-title>Detecting pattern in crime analysis using machine learning</article-title>,&#x201D; <source>AIP Conf. Proc.</source>, vol. <volume>3075</volume>, no. <issue>1</issue>, pp. <fpage>1</fpage>&#x2013;<lpage>8</lpage>, <year>Jul. 2024</year>.</mixed-citation></ref>
<ref id="ref-26"><label>[26]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>A.</given-names> <surname>Alhudhaif</surname></string-name></person-group>, &#x201C;<article-title>A novel multi-class imbalanced EEG signals classification based on the adaptive synthetic sampling (ADASYN) approach</article-title>,&#x201D; <source>PeerJ Comput. Sci.</source>, vol. <volume>7</volume>, no. <issue>1</issue>, <year>2021, Art. no. e523</year>. doi: <pub-id pub-id-type="doi">10.7717/peerj-cs.523</pub-id>; <pub-id pub-id-type="pmid">34084928</pub-id></mixed-citation></ref>
<ref id="ref-27"><label>[27]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>T.</given-names> <surname>Hart</surname></string-name> and <string-name><given-names>P.</given-names> <surname>Zandbergen</surname></string-name></person-group>, &#x201C;<article-title>Kernel density estimation and hotspot mapping</article-title>,&#x201D; <source>Policing</source>, vol. <volume>37</volume>, no. <issue>2</issue>, pp. <fpage>305</fpage>&#x2013;<lpage>323</lpage>, <year>Jun. 2014</year>. doi: <pub-id pub-id-type="doi">10.1108/PIJPSM-04-2013-0039</pub-id>.</mixed-citation></ref>
<ref id="ref-28"><label>[28]</label><mixed-citation publication-type="journal"><person-group person-group-type="author"><string-name><given-names>Y.</given-names> <surname>Hu</surname></string-name>, <string-name><given-names>F.</given-names> <surname>Wang</surname></string-name>, <string-name><given-names>C.</given-names> <surname>Guin</surname></string-name>, and <string-name><given-names>H.</given-names> <surname>Zhu</surname></string-name></person-group>, &#x201C;<article-title>A spatio-temporal kernel density estimation framework for predictive crime hotspot mapping and evaluation</article-title>,&#x201D; <source>Appl. Geogr.</source>, vol. <volume>99</volume>, no. <issue>1</issue>, pp. <fpage>89</fpage>&#x2013;<lpage>97</lpage>, <year>May 2018</year>. doi: <pub-id pub-id-type="doi">10.1016/j.apgeog.2018.08.001</pub-id>.</mixed-citation></ref>
</ref-list>
</back></article>