@article{ 
author = {Rasekhi, Farnaz and Babaie, Shahram},  
title = {A new multi-hop clustering algorithm based on iterative delay to enhance QoS for Internet of Things}, 
abstract ={In general, Internet of Things (IoT) as a new technology refers to a network of physical things in which objects have a unique identity and are able to communicate with each other or with the end user via the Internet. &#160;The Internet of Things refers to a collection of sensor-embedded devices, processing ability, software, and other technologies that connect and exchange data with other devices and systems over the Internet or other communications networks. Due to the limited radio range of objects and also the reduction of energy consumption, information transmission is carried out through the intermediate objects, which highlights the necessity for routing. Routing algorithms can be classified into static and dynamic techniques as well as source initiated and destination initiated approaches. In general, routing algorithms can be classified into data centric, hierarchical, geographical, and quality of service-based mechanisms. A routing algorithm directly affects reliability, transmission latency, power consumption, network throughput, bandwidth utilization, and network lifetime. This paper proposes a new routing method based on distributed clustering and iterative latency to improve the Quality of Service (QoS) of IoT, which divides network things into a number of separate clusters. The proposed method consists of four stages, i.e. network clustering, steady state, multi-hop transmission based on delay estimation, and investigation of adjacent headers. Clustering is performed based on the different states of neighbors, and the iterative delay mechanism is used between the cluster heads. The simulation results conducted through Cooja tool indicate that the proposed method outperforms the LEACH, LEACH-E, NCACM, and distributed clustering techniques in terms of energy consumption and packet delivery ratio by 33% and 9%. Furthermore, simulation results illustrate that the proposed method outperforms in terms of the first node death time and the number of dead objects in scattered and dense networks by 14% and 12%, respectively.},  
Keywords = {Internet of Things (IoT), Distributed clustering, Iterative delay, Routing, Quality of Service (QoS)},
volume = {21},
Number = {1}, 
pages = {3-14}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.1.3},
url = {http://jsdp.rcisp.ac.ir/article-1-1279-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1279-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {pirgazi, jamshid and Ghanbarisorkhi, Ali and IranpourMobarakeh, maji},  
title = {Extracting and combination efficient feature from protein sequence for classify protein based on rotation forest}, 
abstract ={Abstract Protein function prediction is one of the main challenges in bioinformatics, which has many applications. In recent years, many researches in this field have been used machine learning methods. In these methods, First, different features should be extracted from the protein sequence and classification should be done based on the extracted features. The feature extraction methods are based on the physical and chemical properties of the protein sequence. Therefore, extracting suitable features from protein sequence increases and improves the performance of machine learning methods. In this paper, usage of a new set of features based on Position-Specific Scoring Matrix (PSSM), Pseudo-Position Specific Scoring Matrix (PsePSSM), K-gram, Amino Acid Composition (AAC) and the new Term Frequency and Category Relevancy Factor (TFCRF) method, which has not been used in this application so far, is proposed to extract suitable features. In the PSSM method for protein BLAST searches, a scoring matrix is used, in which amino acid substitution scores are given separately for each position in a multi-sequence protein alignment. The PsePSSM feature is described by considering different ranking correlation factors along a protein sequenc to preserve information about the amino acid sequence. The normalized occurrence frequency of a certain number of amino acids in the protein is calculated by the ACC method. An K-gram is a set of K successive items in a protein that&#160; include amino acid. In the TFCRF weighting method, in addition to paying attention to how these are distributed in different sequences, how these are distributed in different classes is also paid attention to.The features extracted using this method give machine learning models a good discriminating power between data in classes. In the next step, classification is done using the extracted features using the rotation forest method. This classifier is a successful ensemble method for a wide range of data mining applications. In this method, the feature space is changed through Principal Component Analysis (PCA), which increases the power of this classifier. The proposed method has been compared to different classifiers. The results show that the efficiency of the proposed method is much better than other state-of&#8211;the-art methods in this application.},  
Keywords = {Protein sequence, feature extraction, TFCRF,  rotation forest, relevancy factor},
volume = {21},
Number = {1}, 
pages = {15-26}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.1.15},
url = {http://jsdp.rcisp.ac.ir/article-1-1387-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1387-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Parvinnia, Elham and Safari, mohammad and khayami, seyedalirez},  
title = {Exploring on rotating machines abnormal state with data mining in protective parameters}, 
abstract ={In order to protect rotating machines and prevent their operation in unusual situations, protective control systems and process data are traditionally used. In this article, a method has been proposed to detect the indirect effects of abnormal operating modes using data mining methods. One of the dangerous conditions of abnormal operation in compressors, as one of the important rotating machines in industries, is the surge condition. In this article, the real data stored during three years of a three-stage refrigerant compressor in a gas refinery are used. the relationship between the surge state of the compressor and the amount of vibration in its different parts has been investigated. It has been proven with data mining methods that there is a direct relationship between the state of surge and the amount of vibration. Also, more sensitive points to vibration during the surges have been identified and it has been proven that by measuring these points, surges can be detected. Therefore, in addition to the existing and previous traditional methods that use process data, it is possible to use the amount of vibration of the points as an extension protection system for surge detection. in this way, more protection of the compressor against the state of surge can be achieved. In this study, various data mining methods have been evaluated, and the results of the nearest neighbor method with the number of neighbors of two have the best performance, and the effects of the number of records in the data set on the quality and accuracy of the results have been investigated.},  
Keywords = {Rotating machine, Datamining, Surge detection, Compressor, Protection parameters},
volume = {21},
Number = {1}, 
pages = {27-38}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.1.27},
url = {http://jsdp.rcisp.ac.ir/article-1-1351-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1351-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Daneshpour, Negi},  
title = {Presenting a new method for mixed data clustering based on the number of similar features}, 
abstract ={Clustering is an operation in which a set of data samples is categorized according to the degree of similarity. Examples of clustering data are numerical or a mixture of numerical and non-numerical (nominal) data. Finding similarities and measuring distances is one of the challenges of mixed data clustering. In the related works, to detect the degree of similarity and obtain the distance value, only the parameter of the distance value was considered and the cluster was selected based on its value. Clustering in this way, especially for mixed data, has not had very accurate results. In this paper, we have tried to pay attention to the parameter &#34;number of similar features&#34; in calculating the degree of similarity and determining the distance. In assigning each sample to a cluster in cases where the distances are equal or close, the number of common features of the samples will determine the appropriate cluster. That is, we will pay attention to the &#34;number of similar features&#34; in addition to the distance to select the cluster. This idea believes that in cases where the distance of the cluster centers is close to the data object, it is better to choose the cluster center that has more features similar to the data object. Logically and also according to the proposed algorithm, the amount of similarity should be in a larger number of features, not just a few limited features but with high similarity. The parameter of the &#34;number of similar features&#34; has a specific definition and is obtained with a suitable threshold. If the distance value of two features is less than the threshold, those two features are considered as similar features. To calculate the distance in the algorithm, the normalized numerical difference for numerical properties and the Hamming distance for non-numerical properties are used. Determining the initial cluster centers, like many methods, is done randomly, and in subsequent iterations of the algorithm, more appropriate samples are selected as the cluster centers. The algorithm is compared with 5 other algorithms in 5 datasets. In examining the results, three criteria of Accuracy, RI and F-Measure have been used. According to the test results, in the mixed and integer datasets, the algorithm performs at least two percent better than the two algorithms and one percent better than the other algorithm. In another data set, the proposed algorithm had results equal to or close to one percent better accuracy than the superior algorithm. In the last data set, the proposed algorithm was ranked second among 5 algorithms. In general, the proposed algorithm won the top rank in most of the results, and in the rest of the cases, it won the second rank out of the five tested algorithms.},  
Keywords = {Clustering, Mixed data, Distance of values, Similarity of values, Cluster Center.},
volume = {21},
Number = {1}, 
pages = {39-52}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.1.39},
url = {http://jsdp.rcisp.ac.ir/article-1-1329-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1329-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Ahmadi, Sayed Mohammad and Dianat, Rouhollah},  
title = {A two-stage clustering-based distributed framework for large-scale face identification}, 
abstract ={Face recognition poses challenges in accuracy, memory efficiency, and computational complexity. This study proposes a two-stage, three-module approach: Subnetwork modules, Cluster-Finder unit, and Final-Decision module. Unlike random distribution methods, our approach employs clustering for distribution. Each subnetwork, a supervised deep neural network, is trained with cluster-specific data. The Cluster-Finder unit compares test data similarity with each subnetwork&#8217;s representative. The Final-Decision module selects the best class. Results indicate superior accuracy, recall, and F1 score compared to competitive methods. The approach is faster and more accurate than non-distribution methods, with comparable speed and higher accuracy than random distribution methods. Experiments on VGGFace2, MS-Celeb-1M, and Glint360K datasets confirm both superior performance and scalability. The proposed method, using KMeans for distribution, outperforms Softmax Dissection and Dynamic Active Class Selection. It simplifies training without additional manipulations, offering efficiency over methodologies like Softmax Dissection and ArcFace parallelization. In conclusion, this study focuses on pre-processing and post-processing without added training complexity. A divide-and-conquer approach addresses accuracy and efficiency challenges. In this study, various sources leading to errors in face recognition systems have been examined. These sources include: imprecise features, overfitting, challenging classes, distribution issues, and decision-making complexities. Various classification scenarios are explored, including non-distributed and models with random and intelligent distributions. Inaccurate features uniformly impact all scenarios, with overfitting posing the greatest challenge in non-distributed scenarios. Challenging classes are better distinguished in intelligent distribution scenarios. Inappropriate distribution has less impact in intelligent scenarios, and decision-making challenges exist in both distributions},  
Keywords = {face recognition, face identification, clustering, deep learning, distributed learning.},
volume = {21},
Number = {1}, 
pages = {53-70}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.1.53},
url = {http://jsdp.rcisp.ac.ir/article-1-1362-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1362-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Mottaghi, Maryam Sadat and Minaei-Bidgoli, Behrouz},  
title = {A Graph-based Algorithm for Clustering Qur’anic Surahs}, 
abstract ={The Holy Qur&#39;an is revealed from God Almighty. Up to now many scholars and researchers have tried to understand the Holy Qur&#39;an and comprehend it. The availability of computer systems is a great opportunity to help researchers reach higher peaks by speeding them up in their way. Clustering is one of the methods has been used to understand the structure of the data. In clustering, we want to divide samples of data into groups so that the members of each cluster are similar together and are different from the members of the other clusters. Clustering of Qur&#39;anic surahs has been the subject of some computer studies on the Qur&#39;an. In these studies, different approaches have been considered to vectorizing the surahs. In a study, Thabet formed vectors of each surah by considering some stems of Qur&#39;anic words as features and the normalized probability of their occurrences in the surah as feature values and clustered just 24 surahs due to the sparseness of the obtained data matrix. With a similar approach in vectorizing the surahs, Moisl calculated the minimum surah length threshold per feature in order to solve the problem of shorter surahs by using some concepts of statistical sampling theory, and could cluster more surahs. Instead of using words as features, Sharaf considered 13 features including existence of referring to the story of Adam and Ebliys, number of the phrase &#171;یا أَیُّهَا الَّذینَ آمَنُوا&#187; (O you who believe), and determined the method of measuring each feature. Then, he formed data matrix and clustered the Qur&#39;anic surahs. In another study, Sufi et al. considered the topics identified for each verse in the Tafsir Rahnama as features and constructed a binary data matrix based on the presence or absence of that topic in the Tafsir of that surah and applied clustering. In this article, we have clustered the surahs of the Holy Qur&#39;an based on the co-occurrence of words in it. To achieve this goal, we have used an existing graph-based approach. In the present study, we first represent each surah as a weighted undirected graph. Then we form the vector of each surah by considering closed frequent sub-graphs as features and relative occurrence of them in each surah as feature values, and eventually cluster the surahs. We used the Silhouette score to evaluate the quality of clustering. Based on this criterion, in the best clustering among different runs, the Silhouette score of 0.91 was obtained. This research provide a proper structural infrastructure for specifying the semantic layer of Holy Qur&#39;an surahs for computational linguistics researchers in the domain of Qur&#39;anic studies.},  
Keywords = {Document Clustering, Text Graph, Frequent subgraph, Computational Qur'an mining},
volume = {21},
Number = {1}, 
pages = {71-88}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.1.71},
url = {http://jsdp.rcisp.ac.ir/article-1-1220-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1220-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Etesam, Mohammad and Sadeghi-Lotfabadi, Ashkan and Ghiasi-Shirazi, Kamaledi},  
title = {Designing L-BFGS inspired automatic optimizer network}, 
abstract ={Nowadays using features learned by machines is common and these types of features have excellent quality in comparison with hand-designed features. While many machine learning models are developed to extract features automatically, however, the optimizing algorithms are still designed manually. In this paper, we propose a method to cast the optimizing algorithm as a machine learning problem. This is a branch of machine learning which is named meta-learning or learning to learn. Gradient-based optimization algorithms (e.g. gradient descent and BFGS) receive the gradient vector in each step and, by using the information of the previous points and gradients, estimate the update vector at the current point. The inputs and outputs of these algorithms are vectors whose dimension is the same as the optimization problem. These algorithms are written solely based on vector addition, scalar-product, and inner-product operations. Therefore, we can say that these algorithms are executed in a Hilbert space whose dimension is determined by the optimization problem. In this paper, we propose a novel method for learning to optimize over a Hilbert space of unknown dimensionality. We introduce a new neural network module named Hilbert LSTM (HLSTM) which is based on a novel LSTM cell whose learning process is independent of the input data dimension. This independency is the result of restricting the network to the operations on a Hilbert space, prohibiting the network to work directly with the entries within a vector. To achieve this goal, we use a linear coefficients layer that linearly combines the input vectors based on coefficients computed by their inner products. Training the network based on the inner product between vectors leads to learning an optimization algorithm that is independent of the data dimension. Our experiments show that the proposed optimizer achieves better results in comparison with hand-designed algorithms.},  
Keywords = {Hilbert LSTM, LSTM, L-BFGS, meta-learning, automatic optimization},
volume = {21},
Number = {1}, 
pages = {89-100}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.1.89},
url = {http://jsdp.rcisp.ac.ir/article-1-1142-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1142-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Al-Aswad, Azal and Minaei-Bidgoli, Behrouz and Shenassa, Mohammad-Ebrahim and Hossayni, Sayyed-Ali and Seryani, Habib},  
title = {Noor-stem v.1 A Benchmark Dataset for Evaluating the Arabic Stemmers}, 
abstract ={The main task of the tokenization is to divide the sentences of the text into its constituent units and remove punctuation marks (dots, commas, etc.). Each unit is a continuous lexical or grammatical writing chain that is an independent semantic unit. Tokenization occurs at the word level and the extracted units can be used as input to other components such as stemmer. Stemming is the main step of several processing tasks such as text mining, information retrieval, and natural language processing.Arabic stemmers face many challenges, mostly caused by the complex nature of Arabic words and their different writing styles. To our knowledge, there is no gold stemming dataset, which contains a wide variety of different possible stemming challenges, so that, stemmers face numerous and different possible real-world challenges to stem the words. Thus, we find it valuable to develop a dataset for evaluating the sustainability of stemmers in such a variety of challenging situations. In this paper, we introduce Noor-Stem, a benchmark dataset with various writing styles for the evaluation of Arabic stemmers. We use two thousand Arabic words in this dataset. We choose the words from different sources such as holy Quran as well as the Arabic websites and assign them to two groups of human experts to determine the correct stem for each word. The first chosen collection of words includes non-repetitive words of the Quran according to their morphological structure. This collection, with more than 16,000 words, is completely by its Quranic usage, labeling only the words stems. The necessity of morphological analysis in Quranic texts as an example of the index of classical Arabic texts has given rise to this evaluation. The second word collection includes 10 thousand words from the non-repetitive words of the text data in general classic Arabic texts. Out of more than 2,600,000 non-repetitive words, considering that the dataset is going to be gold and each stem must be labeled/ensured by a couple of experts, 10,000 words are chosen, regarding the comprehensive and unique patterns to fully measure the length. The variety of patterns can face each stemmer with a serious challenge to demonstrate its performance in various processes. We evaluate the performance of three Arabic stemmers (Light 10, NLTK and Tashaphyne) on this dataset. The results show that the F-measure of Tashaphyne is better than the other stemmers, which re-proves the superiority of this stemmer in this type of problem, as well.},  
Keywords = {Benchmark Dataset, Stemmer, Noor-Stem, Infix, Information Retrieval},
volume = {21},
Number = {1}, 
pages = {101-112}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.1.101},
url = {http://jsdp.rcisp.ac.ir/article-1-1346-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1346-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Hamidzadeh, Javad and Moradi, Mo},  
title = {Improving Collaborative Recommender Systems by Integrating Fuzzy C-Ordered Means Clustering and Chaotic Self-Adaptive Particle Swarm Optimization Algorithm}, 
abstract ={Recommender systems are a subset of intelligent information filtering systems that discovers user interests and provide user-friendly recommendations. User-based collaborative filtering recommender systems is one of the most important types of recommender systems. However, they are faced with voluminous data and sparsity problems that have negative effects on the performance of the systems. In the proposed method, fuzzy C-ordered means clustering algorithm is integrated with a chaotic self-adaptive particle swarm evolutionary algorithm for clustering users. The proposed method aims to improve the rating prediction in large sparse datasets and reduce the negative impact of outliers and noisy data. Experiments have been conducted on real-world datasets to evaluate and prove the efficiency of the proposed method. Experimental results show the superiority of the proposed method that the state-of-the-art methods based on prediction error criteria, accuracy rates, and the computational time.},  
Keywords = {Recommender systems, Collaborative filtering, Fuzzy clustering, Evolutionary algorithm, Chaotic self-adaptive particle swarm optimization algorithm.},
volume = {21},
Number = {1}, 
pages = {113-124}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.1.113},
url = {http://jsdp.rcisp.ac.ir/article-1-1129-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1129-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Nourollahi, Seyedeh Fatemeh and Baradaran, Razieh and Amirkhani, Hossei},  
title = {Domain adaptation-based method for improving generalization of hate speech detection models}, 
abstract ={Today, with the growth of activity in social media, we see an increase in hate speech online and for this reason, the issue of recognizing hate in cyberspace is important. Also, domain adaptation is one of the important challenges in this task and in general in the field of natural language processing. In many issues, while changing the domain, we face a drop in performance, which is also true in the task hate speech. In this research, we try to increase the generalizability of hate detection models by using domain adaptation methods. For this purpose, we use Transformer-based methods, including domain adversarial training and mixture of experts, and we also use multi-source training. Experiments are conducted using four datasets in the domain of hate. At first, we evaluate the models in an in-domain and single-source manner. In the next step, by adding other domains to the education section, we see a drop in results and a negative transfer. Then we perform the out-of-domain tests first as a single source with the DistilBERT model, which significantly reduces the results by changing the domain. In order to increase the power of domain adaptation of the model in the out-of-domain part, we perform the training on several sources, leads to improve the results in about half of the cases, which is not significant. In the following, we try to increase the domain adaptation power of the models, using transformer-based methods including domain adversarial training and the mixture of experts, which leads to increase in performance in 87% of multi-source out-of-domain tests. Of course, these methods are also effective in the performance of in-domain tests. An important issue that sometimes causes a significant drop in results is datasets. The similarity of the data and the similarity of the distribution of some domains increase the power of domain adaptation of the model and on the contrary.},  
Keywords = {hate speech, classification, transformer, domain adaptation, generalization},
volume = {21},
Number = {1}, 
pages = {125-142}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.1.125},
url = {http://jsdp.rcisp.ac.ir/article-1-1341-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1341-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Mir, Marziye and Noferesti, Samir},  
title = {Using Data Augmentation Techniques for Sentiment Analysis of Users’ Opinions on Reopening of Schools During the Covid-19 Epidemic}, 
abstract ={Sentiment analysis, also called opinion mining, is one of the sub-areas of natural language processing that aims to classify texts according to the sentiments, beliefs and attitudes expressed in them. In the most current research, texts are divided into two &#34;positive&#34; and &#34;negative&#34; categories. However, there are also other categories such as good/bad&#34; and agree/disagree, every one of which has its applications. The purpose of this paper is to analyze the opinions expressed by users on social media about the reopening of schools during the Covid-19 outbreak using supervised machine learning techniques, and to classify them into two &#34;agree&#34; and &#34;disagree&#34; categories. Users&#39; opinions, in this paper, are in Persian. The lack of sufficient datasets and also the low accuracy of natural language processing tools are the most important problems of text processing in Persian. Due to the mentioned limitations, the use of supervised machine learning algorithms and also the extraction of effective features for training machine learning classifiers in Persian are facing a serious challenge. In this paper, first, a small dataset of the users&#39; opinions about the reopening of schools was collected and manually labeled. Then, a combined method was used for data augmentation of the dataset. In the proposed method, first, Persian sentences were translated into English. Then nouns, verbs and adjectives of the English sentences were replaced with their synonyms. Next, the English sentences were translated into Persian again. The new sentence with the class label of the initial sentence was added to the training set. Thus, the size of the training set increased by 97 percent. After that, the efficiency of employing the common pre-processing steps and using common feature sets in sentiment analysis of the English texts for Persian were evaluated and the best of them were selected. Considering the low accuracy of the Persian natural language processing tools, it was tried to select those features that were less dependent on the tools. Finally, machine learning classification was used to determine agree/disagree class of the user opinions of the test sets. The results of the experiments indicated that by applying the proposed method for data augmentation and using selected features in this paper, 81 and 79 percent precision was obtained for the polarity classification of opinions using SVM and CNN algorithms, respectively.},  
Keywords = {Sentiment Analysis, Opinion mining, Supervised learning, Deep learning, Data augmentation, Covid-19},
volume = {21},
Number = {2}, 
pages = {3-14}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.2.3},
url = {http://jsdp.rcisp.ac.ir/article-1-1385-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1385-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {sametomrani, moslem and sanieeabadeh, mohammad and moghaddamcharkari, nasrollah},  
title = {Rumor Detection on Twitter using tweet and user features}, 
abstract ={When every news item is posted on social media, reactions to it are different and arouse curiosity from different viewpoints. The most important part is to understand the accuracy of the news. A rumor is invalid news, meaning it has not yet been confirmed and it may cause irreparable damage if it is not valid. Therefore, it is very important to detect it. Rumor detection, or in other words, determining its validity, plays an essential role in preventing fake news. Naturally, every phenomenon of normal and anomaly is transmitted to people through social networks. Every News Reactions to that news are different. Depending on the importance of the news, it may be widely covered or it may not have a specific reaction. But if the news spreads widely, it arouses curiosity from different angles. The news is false or true, or the news is valid or invalid. In this work, an attempt was made to identify rumors on social networks by using Hand-Crafted features based on tweets, users and a combination of the two, oversampling and normalization, and by using machine learning classification. Using 4 machine learning classifiers, including Support vector machine, Logistic regression, K-nearest neighbors and Random forest, the two rumors on social networks were detected. Two data sets, PHEME 2017 and PHEME 2018, have been used. The results on these two datasets show that in PHEME 2017, the random forest classifier shows an accuracy of 0.988 using tweet and combination features. Also, these features show a precision of 0.987, which is better than other classifiers used in this work. This classifier has a better recall than other classifiers along with logistic regression with a value of 0.986. Also, this classifier obtained better results with the two mentioned features, with 0.987. In the PHEME 2018 dataset, it obtained the RF classifier with an accuracy of 0.969 using tweet and combination features, and it has better performance in precision, recall and F1. In addition, the user feature in the classifier of k nearest neighbors brings better results than the other two features.},  
Keywords = {rumor detection, machine learning, user Feature, tweet Feature, Hand-Crafted Feature},
volume = {21},
Number = {2}, 
pages = {15-28}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.2.15},
url = {http://jsdp.rcisp.ac.ir/article-1-1354-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1354-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Haj-Hosseini, Zeinab and Doostari, Mohammad-Ali and Yusefi, Hame},  
title = {Implementation of a countermeasure method against DPA on McEliece Post Quantum Cryptosystem}, 
abstract ={In recent years, embedded systems have continuously gained importance. This ubiquity is accompanied by an increased need for embedded security. Cryptography can address these security requirements. Many symmetric and asymmetric algorithms, such as AES, DES, RSA, ElGamal, and ECC, have been implemented on embedded devices. All frequently implemented public-key cryptosystems rely on the presumed hardness of either factoring the product of two large primes (FP) or computing discrete logarithms (DLP). These two problems are closely related. Therefore, solving these problems would have significant ramifications for classical public-key cryptography and, consequently, for all embedded devices that utilize these algorithms. Currently, both problems are believed to be computationally infeasible with a conventional computer. However, a quantum computer capable of performing computations on a few thousand qubits could solve both problems using Shor&#39;s algorithm[1]. Although a quantum computer of this scale has not been reported, it could become a reality within the next one to three decades. Consequently, the development and cryptanalysis of alternative post-quantum cryptosystems are crucial. Post-quantum cryptosystems refer to cryptosystems that are not susceptible to the critical security loss or complete compromise caused by quantum computers. One of the major security challenges is the development of quantum computers and the potential compromise of current cryptosystems in the future. Therefore, it is essential to consider post-quantum cryptosystem algorithms and the challenges of implementing and attacking them. Post-quantum cryptosystems encompass various types, including hash-based cryptography, multivariate-quadratic-equations cryptography, lattice-based cryptography, and code-based cryptography. In this study, our focus is on the QC-MDPC McEliece code-based algorithm. Post-quantum public keys must be designed to gain popularity in practice; they should be optimized for implementation and efficient in execution. McEliece encryption and decryption do not require computationally expensive processing, making it more suitable for implementation[2]. One of the implementation challenges for these algorithms is the large key length, which poses an important issue for implementation on embedded systems. Additionally, countering side-channel attacks caused by information leakage from hardware equipment is crucial. We have addressed this by reducing the key length from 1200 bytes to 180 bytes, providing 80-bit security, and introducing a new method for implementing the QC-MDPC McEliece cryptosystem. Differential power analysis attacks (DPA) exploit the relationship between power consumption and intermediate data to recover the key. In this study, we have used a masking technique for multiplication in the finite field in the syndrome computation part of the decryption algorithm. We have implemented the Threshold Implementation (TI) masking countermeasure for DPA to eliminate information leaks from the previous implementation.},  
Keywords = {Post-Quantum Cryptosystem, DPA, McEliece, QC-MDPC Codes},
volume = {21},
Number = {2}, 
pages = {29-42}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.2.29},
url = {http://jsdp.rcisp.ac.ir/article-1-1222-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1222-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {mihandoost, sar},  
title = {Atrial Fibrillation Classification Using PiCA-ESN Algorithm and Stockwell Transform}, 
abstract ={Atrial fibrillation (AF) is a prevalent cardiac arrhythmia characterized by irregular heartbeats, often without noticeable symptoms in patients. Diagnosing AF is challenging for cardiologists, requiring advanced methods for accurate identification using electrocardiogram (ECG) signals. Automated AF diagnosis can significantly aid cardiologists in prompt identification, potentially reducing the risks associated with acute heart disease and stroke. Various non-invasive techniques based on ECG signal processing have been suggested to better understand the mechanisms by analyzing the atrial fibrillatory waves (f-waves). Different signal processing methods for f-wave extraction have been explored, which may be classified as follows: average beat subtraction and its advanced variants, QT-interval interpolation, principal and independent component analysis, nonlinear adaptive filtering using an echo state network, diffusion geometry, and extended Kalman filtering. This study aims to extract the f-wave from the ECG signal using the PiCA-ESN algorithm, which yields better results compared to other methods. Additionally, the f-wave&#39;s time-frequency behavior was analyzed using the Stockwell transform to differentiate between terminated and non-terminated AF states for the first time in this study. First, the PiCA-ESN algorithm facilitated the extraction of the f-wave from the ECG signal. Subsequently, the Stockwell transform was used to compute the time-frequency maps of the extracted f-wave. Various features were derived from the amplitude of the Stockwell transform and utilized in conjunction with three classifiers: MLP, SVM, and AdaBoost. The findings reveal that the proposed method outperforms selected methodologies from the Physionet Challenge 2004, achieving an impressive 100% accuracy in both tasks. Additionally, an experiment was conducted to assess the robustness of the proposed features across consecutive signal segments, validating their stability during signal analysis.},  
Keywords = {Electroencephalogram (ECG), Atrial fibrilation (AF), f-wave, PiCA-ESN algorithm, Stockwell transform (S transform)},
volume = {21},
Number = {2}, 
pages = {43-54}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.2.43},
url = {http://jsdp.rcisp.ac.ir/article-1-1374-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1374-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {soleimani, mohsen and ChehelAmirani, Mahdi and Kabodian, Seied Jahanshah},  
title = {Steganalysis of Compressed Audio Files Based on Machine Learning}, 
abstract ={The science of hiding a message containing information in a carrier medium is called steganography, and the attempt to detect the presence or absence of a hidden message in a cover medium is called steganalysis. The MP3 compression format has been used among audio data as a suitable and comprehensive host for information encryption, and various encryption methods have been designed for this purpose. In this research, the aim is to present an algorithm for audio ateganalysis, specifically for compressed audio files in MP3 format, in which some data has been embedded using MP3stego software. To prepare encrypted data, text files with random texts have been used. First, by using the side information extracted from MP3 files, the necessary features are extracted and the audio data, which includes two categories of stego files and clean files, is divided into two parts: training data and test data. And then, using machine learning techniques (support vector machine), the detection system of infected files and clean files is designed, and finally, the efficiency of the system is measured using the test data. In this paper, a new feature called spectral peakiness (SPK) is extracted from the side information of MP3 file. The proposed system was tested using separate test data, which includes clean files and stego files with various encryption capacities, and it distinguished clean and stego files with 100% accuracy and without error. The results indicate the perfect classification of stego and clean files while reducing the computational complexity and increasing the speed of steganalysis compared to other methods. Instead of using the audio signal information stored in the MP3 file, the proposed method uses the side information of the MP3 file, which is less dependent on the audio content of the file. In this method, the MDB side information in the compressed audio file is assumed as a sequence, and then, using a feature extraction method, a new feature in the frequency domain called spectral peakiness is calculated. This simple yet powerful feature is combined with features such as temporal average and spectral average of the MDB sequence and forms a low-dimensional (three-dimensional) feature vector. This feature vector will then be classified by a support vector machine (SVM) classifier as a suspicious file or a normal file. The feature extraction method, while being simple and having very few calculations, has 100% accuracy (recognition without any error) for MP3 files, even when the amount of the hidden information in the audio file is very low.},  
Keywords = {Compressed Audio File, Audio Steganography, Audio Steganalysis, MP3, MP3stego},
volume = {21},
Number = {2}, 
pages = {55-66}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.2.55},
url = {http://jsdp.rcisp.ac.ir/article-1-1273-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1273-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {saneiarani, hassan and esmaili, mahdi and Afsharkazimi, Mohmmad ali},  
title = {Choosing the Best Installation Paths In the Development of Urban CCTV Cameras}, 
abstract ={Optimizing camera placement is a two-decade-old research problem. Many researches have solved the problem with different approaches. Some different methods such as genetic algorithm, reinforcement learning, and greedy algorithm have been developed to obtain the maximum surface coverage. Some researchers have considered specific applications in order to optimally cover a certain area such as a coastal area or a protected area under the coverage of CCTV cameras. Some researchers have also considered the camera&#39;s capabilities of vertical rotation or horizontal rotation or zooming in order to use these capabilities for optimization. With the development of drone manufacturing technology, this tool is also proposed for specific applications. But what is less discussed is the optimization of the placement of urban surveillance cameras in a real city map. Usually, due to the high cost, all city cameras are not installed at once, and cameras are added annually to develop the city traffic monitoring system. Therefore, it is necessary to prioritize the selection of the route and a very important factor in prioritization is traffic. Traffic is the most important factor in choosing the route for the placement of urban surveillance cameras because the streets with more traffic are exposed to more traffic accidents and should be the priority for video monitoring. Traffic data is usually big data, not available for all cities, and on the other hand, providing traffic data may violate citizens&#39; privacy. Therefore, there are many methods for creating virtual traffic, which are classified into two categories: macro and micro. Macro methods model traffic as a physical phenomenon such as fluid or gas, but micro models, which are mostly used in artificial intelligence methods, consider traffic as a set of individual trips. In this work, we use the second method to create virtual traffic so that routes with more traffic are prioritized for installation. Citizens usually make a lot of intra-city trips, and the function of city monitoring systems is to monitor these routes. Therefore, the placement of surveillance cameras should also be in such a way that it considers the observation of these routes. In the proposed method, the real map of the city is selected as a model. Then, by separating the main paths and obtaining the skeleton of the path, a graph of the paths is obtained, the intersection point of the paths will be its vertex and the distance between the vertices will be the weight of the connecting edges. Now by randomly selecting two vertices from the graph as the origin and destination of an intra-city trip and routing between them with Dijkstra&#39;s algorithm, a trip is made. By repeating this process, virtual traffic is simulated. To create virtual traffic similar to real traffic, the probability of choosing high-traffic points is considered more than other points. Therefore, the probability of selecting vertices in the graph is different according to their location in the city. By creating one hundred thousand paths for the studied model, the edges with the highest repetition can be found as the final results and suggested for camera installation. The evaluation of the final results is done by repeating random experiments and using the Jaccard similarity coefficient, and the degree of similarity of the output results is checked. The reliability of the proposed method is expressed by mathematical analysis and by drawing graphs, and the impact of influential parameters such as the number of city trips, the probability of choosing points, the impact of city topology, and the number of output results are expressed analytically, and the similarity of the results is 98%. The advantage of the proposed method is not depending on special tools such as special cameras for traffic measurement, as well as not depending on a specific location and topology.},  
Keywords = {placement of urban CCTV camera - Virtual traffic - Jaccard similarity coefficient - Dijkstra's algorithm - smart city},
volume = {21},
Number = {2}, 
pages = {67-78}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.2.67},
url = {http://jsdp.rcisp.ac.ir/article-1-1402-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1402-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Jaberi, Pouyan and Nemati, Shahla and Basiri, Mohammad Ehs},  
title = {Classification of skin cancer images using two-level ensemble deep learning}, 
abstract ={Today, despite the tremendous advances in medical science and technology, access to a specialist doctor is still considered a major challenge. This challenge is of great importance for diseases such as cancer. Skin cancer is the 13th most common cancer in men and the 15th most common cancer in women. While some skin problems are benign and harmless, some of them can be malignant masses, which will remain harmless if they are diagnosed in time. When consulting a specialist doctor may be time-consuming and expensive, an intelligent system can be a fast alternative or, at least, an efficient preliminary treatment solution. For skin cancer, such intelligent system may utilize the images of suspicious skin masses labeled according to their benign or malignant state by specialist physicians. These labeled images are useful for training intelligent systems which should diagnose the potential problems in unseen new images. In this research, a novel deep learning-based approach is proposed for the problem of classifying skin cancer images into two categories of benign and malignant images. In the proposed model, powerful deep learning models for image classification including VGG, ResNet, and Inception are used in two levels. Specifically, we formed two ensembles; VGG ensemble which consists of VGG-16 and VGG-19 models and ResNet ensemble which consists of ResNet152, ResNet50, and Inception models. CatBoost algorithm is used in each level to combine the models on that ensemble. Finally, at the next level, two ensembles were combined using the CatBoost algorithm. The proposed ensemble model tries to improve the accuracy and consistency of the results by aggregating the deep models at its two levels. In order to show the utility of the proposed model, a subset of ISIC public dataset for skin cancer images is used for training and evaluation of models. The performance of the proposed ensemble model is compared with several deep neural networks and previous similar researches. Specifically, we compared the results achieved by the proposed model with those obtained by existing similar deep models and those used as building blocks of the proposed model. The results show that the proposed model performs better in classifying skin cancer images. The performance of the proposed model, both in each of the classes and in general, has been better than all independent deep learning models. It has also been shown that using VGG ensemble along with this proposed model by combining its results with the help of CatBoost and forming a two-level ensemble has improved its independent performance in each class.},  
Keywords = {Deep Learning, Ensemble Learning, Skin Cancer, Benign, Malignant},
volume = {21},
Number = {2}, 
pages = {79-90}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.2.79},
url = {http://jsdp.rcisp.ac.ir/article-1-1350-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1350-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Ahmadnia, Mahdi and Maghrebi, Mojtaba and Ghanbari, Rez},  
title = {Providing an effective way to enhance low-light images: Enhanced Illumination Map Optimally}, 
abstract ={Low-light images often suffer from low brightness and contrast, which makes some scene details hard to see. This can affect the performance of many computer vision tasks, such as object recognition, tracking, scene understanding, and occlusion detection. Therefore, it is important and useful to enhance low-light images. One technique to enhance low-light images is based on the Retinex theory, which decomposes images into two components: reflection and illumination. Several mathematical models have been recently developed to estimate the illumination map using this theory. These methods first compute an initial illumination map and then refine it by solving a mathematical model. This paper introduces a novel method based on the Retinex theory to estimate the illumination map. The proposed method employs a new mathematical model with a differentiable objective function, unlike other similar models. This allows us to use more diverse methods to solve the proposed model, as classical optimization methods such as Newton, Gradient, and Trust-Region methods need the objective function to be differentiable. The proposed model also has linear constraints and is convex, which are desirable properties for optimization. We use the CPLEX solver to solve the proposed model, as it performs well and exploits the features of the model. Finally, we improve the illumination map obtained from the mathematical model using a simple linear transformation. This paper introduces a new method based on the Retinex theory for enhancing low-light images. The proposed method improves the illumination and the visibility of the scene details. We compare the performance of our method with six existing methods: AMSR, NPE, SRIE, DONG, MF, and LIME. We use four common metrics to evaluate the visual quality of the enhanced images: AMBE, LOE, SSIM, and NIQE. The results demonstrate that our method is competitive with many of the state-of-the-art methods for low-light image enhancement.},  
Keywords = {Enhance illumination, Enhance low-light images, Illumination map, Retinex theory, Optimization model, Image Processing},
volume = {21},
Number = {2}, 
pages = {91-104}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.2.91},
url = {http://jsdp.rcisp.ac.ir/article-1-1256-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1256-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Teymouri, Ahmad and Deypir, Mahmoo},  
title = {Two-level intrusion detection system for Internet of Things network based on deep learning}, 
abstract ={Along with the growth in the use of Internet of Things networks for various applications, threats and attacks related to these types of networks have also increased. Intrusion detection systems are designed and used to detect and identify attacks in this type of networks, and to identify intrusions or abuses that are going to take place from the network, and to inform the relevant authorities about this issue. In most intrusion detection systems, various methods and algorithms are used, including deep neural networks (DNNs), support vector machines (SVM), or multilayer perceptron (MLP), and other traditional machine learning models. Each method has advantages and disadvantages, but it usually has a lower accuracy rate than combined methods. In recent years, the idea of combining classifications has been used for anomaly-based diagnosis. In this research, to reach better accuracy, we used the combination of principal component analysis (PCA) and convolutional neural network (CNN) algorithms to design our intrusion detection system. In the initial step of the proposed method, after preprocessing including conversions and normalizations, valuable features for classification are extracted. In this study, the NSL-KDD dataset, which has been mentioned in many scientific articles as a valid reference dataset in the field of intrusion detection, has been used. In fact, due to the high number of data dimensions and the high dispersion of feature values, we used a dimension reduction method. The dimensionality reduction method used in this research is principal component analysis (PCA). In the PCA method, the dimensions of the data are reduced in such a way that the reduced dimension data also includes the vital information of the dataset. We used PCA in order to reduce the size and volume of the input data to help increase the efficiency of our main algorithm and the new data generated with this algorithm is provided to the CNN classifier. A convolutional neural network is a special type of neural network with multiple layers that processes data that has a grid arrangement and then extracts important features from them. Here, accurate pattern learning and deep insight from the given data are our two main reasons for using CNN. In the proposed approach, we have two level classification including binary CNN and multi-class CNN, for detecting attacks and exact type of them, respectively. That is, firstly attacks and normal data are identified by binary classification and then by multi-class classification, the types of attacks are identified and separated. In fact, the type of attacks which includes one of DoS, U2R, R2L and Probe cases is determined using second convolutional neural network. Based on the obtained results, we have witnessed the growth of the accuracy rate of the proposed method compared to many other popular methods. In the evaluation of accuracy parameter values for different phases of training and testing, competitive results are observed for binary classification phase. Here we consider the number of 15 rounds. As it is clear from the graph related to training, the accuracy values in the final courses have reached 0.94. The accuracy of the test has also approached the value of 0.9 in the last round. Also, the results obtained in multi-class CNN are such that the accuracy value is 0.99 in the classification of the training data samples and 0.97 in the classification of the test data samples. Moreover, the cost graphs for training and testing courses of multi-class CNN are shown. The cost of training and testing in the final round is 0.06 and 0.09, respectively.},  
Keywords = {Intrusion detection system, Convolutional neural network (CNN), Binary classifier, Multi-class classifier, Principal component analysis (PCA).},
volume = {21},
Number = {3}, 
pages = {3-22}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.3.3},
url = {http://jsdp.rcisp.ac.ir/article-1-1388-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1388-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Fateh, Haleh and Rezvani, Mohsen and Tahanian, Esmaeel},  
title = {A systematic review of the foundations, applications, and challenges of federated learning.}, 
abstract ={Federated Learning (FL) is an innovative machine learning paradigm that tackles the challenge of data island while safeguarding data privacy. It enables decentralized model training by allowing multiple clients&#8212;such as mobile devices, institutions, or organizations&#8212;to collaboratively build models without transferring local data to a central server. This paradigm gained significant attention following Google&#8217;s 2016 initiative to predict user text input on Android devices while maintaining the privacy of locally stored data. A core feature of FL is its distributed and encrypted framework, enabling participants to contribute to a collective learning process without revealing their original data to a central entity or other participants. In recent years, FL has evolved to encompass a broader spectrum of decentralized machine learning techniques, while still maintaining privacy as a central tenet. This evolution has positioned FL as a critical technology in sectors where data privacy, security, and sovereignty are paramount. This paper presents a systematic review of the literature on federated learning, synthesizing insights from review articles, Books, key documents, and published research. The review is structured as follows: Overview of Federated Learning: This section introduces the foundational concepts of FL, detailing its origins, core principles, and operational processes. The decentralized structure and privacy-preserving techniques employed in FL are examined, along with real-world applications as examples. Algorithms and Evolution: This section explores the state-of-the-art algorithms driving FL and traces their development over time. Key innovations in aggregation techniques, optimization methods, and client-server communication protocols are highlighted, demonstrating how they have enhanced FL&#39;s scalability and efficiency. Classification and Applications of FL Architectures: Federated learning architectures are categorized into three main types: horizontal federated learning, vertical federated learning, and federated transfer learning. This section analyzes the application of these architectures across various domains, highlighting their distinctive features and associated challenges. Applications in IoT, Smart Cities, and Healthcare: Using selected case studies, this section evaluates the deployment of FL in the Internet of Things (IoT), smart cities, and healthcare. It assesses how FL enhances data privacy, security, and operational efficiency in these domains, focusing on practical implementations. Comparative Analysis: This section offers a comparative evaluation of the various methods and algorithms used in the aforementioned fields, identifying their relative strengths and weaknesses. Special attention is given to the challenges posed by large-scale FL deployments, including communication overhead, data heterogeneity, and model convergence. Federated Learning and Related Technologies: This section explores the integration of FL with related technologies, such as federated deep learning and federated blockchain, particularly within the context of the Industrial Internet of Things (IIoT). The potential of these technologies to improve storage, data management, and resource optimization is discussed in detail. Challenges and Future Directions: The final section addresses the ongoing challenges facing FL, including scalability, model accuracy, communication costs, and compliance with regulatory frameworks. Additionally, it proposes future research directions aimed at improving the practicality and widespread adoption of FL in industrial and commercial applications. This systematic review provides a comprehensive examination of federated learning&#8217;s current state, including its foundational concepts, applications, and challenges. It also outlines a forward-looking perspective on the advancements needed to establish FL as a key technology in privacy-centric, decentralized machine learning.},  
Keywords = {Federated learning, decentralized machine learning, privacy-preserving, Distributed artificial intelligence, Internet of Things},
volume = {21},
Number = {3}, 
pages = {23-68}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.3.23},
url = {http://jsdp.rcisp.ac.ir/article-1-1389-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1389-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Jahani, Seyyed Ali and Mohebbi, Keyvan and ZamaniBoroujeni, Fars},  
title = {Improving Scene Recognition in Remote Sensing Using Deep Learning and Feature Selector}, 
abstract ={Remote sensing images as a valuable data source in Earth observation can help in measuring and observing detailed structures on the Earth&#39;s surface. Scene detection in remote sensing images has many applications in various fields such as urban planning, natural hazard detection, environmental monitoring, vegetation mapping, and geographic object detection. One of the key problems in the interpretation of remote sensing images is the scene classification of remote sensing images. Feature extraction is very important in scene detection and classification. Convolutional neural networks are one of the deep learning methods that have significantly increased the performance of tasks such as object recognition and scene classification, but their performance is highly dependent on the number of labeled images available, which are not available enough especially in the field of remote sensing. Recently, transfer learning, especially for the fine-tuning of pre-trained convolutional neural networks, has attracted more attention from researchers as a practical strategy for scene classification in remote sensing. However, the lack of use of local features and global deep model that is trained on the target data set is one of the limitations of current methods. Also, if these networks are not deep enough and the images do not pass through multiple filters, they cannot extract more semantic information, and the extracted features do not have high discrimination power, and as a result, scene recognition is not performed well. On the other hand, the features extracted through local features are very large, and not using feature selector methods reduces the accuracy of the model. In this research, to solve the mentioned limitations, a hybrid approach of feature extraction has been proposed in which three types of features including two types of deep local and global features and one type of manual local feature are combined with each other. To extract deep features, pre-trained convolutional networks have been used. The pre-trained networks used are: ResNet, InceptionNet, GoogleNet and EfficientNet_b0. In order to extract as much information as possible from the images, a convolutional network with 20 fully connected layers is proposed. Also, a combined feature selection stage consisting of two categories of filtering and packing algorithms is included in this model. Finally, scene detection is performed using several different classification algorithms. The different structure of pre-trained convolutional networks and their appropriate depth can be effective in improving the extraction of deep features. In addition, the combination of three categories of different features can provide a more comprehensive knowledge of images. The evaluation of the proposed solution on the UCM, AID, RSSCN7 and NWPU-RESISC45 datasets has obtained the accuracy of 99.27%, 97.91%, 99.09% and 93.09% respectively in identifying and classifying images. As a result, this solution has shown a better performance compared to the models that used the manual extraction of features, as well as the methods that use normal convolutional models.},  
Keywords = {Remote Sensing, Deep Learning, Deep Feature, Hybrid Learning, Pre-Trained},
volume = {21},
Number = {3}, 
pages = {69-84}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.3.69},
url = {http://jsdp.rcisp.ac.ir/article-1-1398-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1398-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Ebrahimi, Abolufazl and Shamsi, Mahboubeh and Mohajjel, Mortez},  
title = {Improvement of missing vital signs data estimation algorithm in wireless body sensor networks based on deep neural networks}, 
abstract ={In a wireless sensor network (WSN), due to various factors such as limited power, sensor transferability, hardware failure and network problems such as packet collisions, unreliable connection and unexpected damage, the amount sensed to the header or base station is not arrives. Therefore, data loss is very common in wireless sensor networks. Loss of measured data greatly reduces WBAN accuracy. Because WBAN deals with the vital signs of the human body, network reliability is very important. To solve this problem, missing data must be estimated. Many methods are used to reconstruct lost sensor data based on temporal correlation, spatial correlation, interpolation method, or sparse theory. Due to the characteristics of vital signs data, they can be considered as a series of sequential information. So far, various methods have been developed to estimate missing data in time series data in different fields. These methods can be divided into two categories: statistical methods and machine learning-based methods. In order to predict missing values, a missing data estimation model based on LSTM recurrent neural network whose network weights are optimized by particle swarm algorithm (PSO) is presented in this paper. In this paper, we use the MIMIC-III Waveform database to test the algorithm and determine the algorithm parameters. However, due to the large volume of data and the difficulty of testing the algorithm on all data, we suffice to test 500 patients with this data, whose vital signs included heart rate, respiration, blood oxygen, and so on. After data preprocessing, network training, predicting lost values and calculating error values, it is observed that the proposed technique of sgdm-LSTM By combining the PSO algorithm is a suitable method for estimating lost values. In addition, experimental results show that the mean square root error of the estimated value is lower than other methods. This value is 1.5898 with the best LSTM network hyperparameters.},  
Keywords = {WBAN, Deep Learning, Artificial Neural Network, Missing Data, Estimation},
volume = {21},
Number = {3}, 
pages = {85-96}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.3.85},
url = {http://jsdp.rcisp.ac.ir/article-1-1277-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1277-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Naghavi, Mehdi and HassaniAhangar, Mohmmad Reza and AmiriJezeh, Ali},  
title = {Identify the Named Entities Using Deep Learning and Reinforcement Approach}, 
abstract ={Named Entity Recognition (NER) has emerged as a critical and highly applicable task in the field of Natural Language Processing (NLP). Its significance stems from its essential role in numerous NLP applications, such as machine translation, question answering, text summarization, and information extraction. Recent studies highlight the substantial impact of advancements in Artificial Intelligence (AI), particularly Deep Neural Networks (DNNs), on improving the performance of NER systems. Deep Neural Networks, with their ability to learn complex patterns and extract rich features, have opened new horizons in addressing NLP challenges. These methods leverage advanced language models like BERT and GPT to enable deeper comprehension of linguistic structures and semantic relationships. One of their prominent capabilities is to capture long-term dependencies in complex sentences while reducing the reliance on manually engineered features. This research introduces a novel hybrid approach for Named Entity Recognition in both Persian and English languages, based on deep neural networks and semantic language models. To address the dependency on large datasets, the proposed method employs an iterative logic mechanism that facilitates effective learning with limited data. The proposed system was evaluated on three datasets: The CoNLL 2003 dataset for English, Two Persian datasets, Arman and Peyma. Experimental results demonstrate that the proposed method achieves F1-scores of 95.3, 96.32, and 94.72 on the CoNLL, Arman, and Peyma datasets, respectively. These scores reflect significant improvements over previous methods. The findings of this study suggest that combining advanced language models with deep neural networks can significantly enhance the accuracy and efficiency of NER systems. These achievements pave the way for developing effective NLP tools for low-resource languages, particularly Persian, and enable the application of this technology in both industrial and research contexts.},  
Keywords = {Named Entity, Named Entity Recognition, Language Model, Deep Learning},
volume = {21},
Number = {3}, 
pages = {97-110}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.3.97},
url = {http://jsdp.rcisp.ac.ir/article-1-1232-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1232-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Samimi, Navid and Nejatian, Samad and Parvin, Hamid and BagheriFard, Karamolah and Rezaei, Vahideh},  
title = {Presenting a Method based on Genetic Algorithm for finding the most Stable Clusters in Ensemble Clustering}, 
abstract ={Clustering is one of the fundamental tools in data analysis and data mining, enabling the extraction of hidden and meaningful structures from large datasets by grouping data based on intrinsic similarities. However, selecting optimal clusters in conventional clustering algorithms poses challenges, especially when clusters are dense or heterogeneous. In this study, a novel genetic algorithm-based method is proposed to identify the most stable clusters in ensemble clustering. By leveraging cluster stability criteria and a correlation matrix, the proposed approach improves the accuracy and stability of the final clustering results. The proposed method involves generating initial partitions of the data using six different clustering algorithms. Next, the Fisher criterion is applied to identify more stable clusters. These selected clusters are then evaluated and optimized using a genetic algorithm to construct an optimized correlation matrix. This matrix is subsequently fed into a hierarchical clustering algorithm, which produces the final consensus clustering. The proposed method was tested on standard datasets. Results demonstrated improvements of 12% and 5% in NMI and ARI metrics, respectively, compared to previous methods. The use of a genetic algorithm enabled the identification of clusters with higher stability and diversity, reducing the impact of noise and increasing the accuracy of the final clustering. Moreover, the method outperformed individual base clustering algorithms in providing more precise clustering results. Due to its ability to enhance the accuracy and stability of clustering, the proposed method holds potential for applications in domains such as big data analysis, machine learning, and information retrieval. The use of the Fisher criterion for selecting stable clusters and genetic algorithms for optimization are among the strengths of this research. This method not only preserves diversity among clusters but also significantly enhances clustering accuracy. Future studies could explore the combination of this approach with more advanced algorithms to assess its applicability to more complex datasets.},  
Keywords = {Ensemble clustering, Cluster Stability, Fisher Criterion, Correlation matrix, Genetic Algorithm},
volume = {21},
Number = {3}, 
pages = {111-136}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.3.111},
url = {http://jsdp.rcisp.ac.ir/article-1-1217-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1217-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Ghasemi, Masoume and horri, abbas and basiri, Mohammad Ehs},  
title = {Detecting Android malware with offloading approach in cloud computing}, 
abstract ={Today, the mobile phone is one of the smart devices that have become a necessity in everyday life and are used for various tasks such as shopping, banking, communicating with friends, family, etc. In recent years, the Android operating system has been able to gain more popularity than other mobile phone operating systems. The number of software related to this operating system is also expanding at a remarkable speed. Unfortunately, this issue is not hidden from the profit-seeking people, and the production of malware of this operating system has also grown in parallel with its development. Third-party Android app stores that have emerged in recent years have become a very strong source of malware distribution, as these stores have weak to non-existent measures to prevent malicious apps from being uploaded and distributed to users&#39; devices. Therefore, one of the challenges that programmers are dealing with in this field is to find solutions to establish security in these types of devices, in such a way that it provides powerful security analysis capabilities while consuming few resources on the device itself. Software products such as Lookout, Norton, and Comodo Mobile Security mainly use signature-based methods to detect malware threats. However, malware attackers use techniques such as repackaging and obfuscation to circumvent signatures and defeat attempts to analyze their internal mechanisms. The ever-increasing sophistication of Android malware requires new defense techniques that can protect users against new threats while not using up all of a mobile device&#39;s processing and storage resources. Therefore, in the current research, a computational offloading method is presented in the cloud structure to identify Android malware. The solution proposed by this research first extracts the features of Android applications during installation and execution on the mobile phone, then sends these extracted features to the cloud servers. On the cloud server side, these features are analyzed and using machine learning algorithms, malware is distinguished from clean programs. The proposed approach is trained and tested using the Drebin dataset. The obtained results show that the proposed approach has achieved 96.44% accuracy for malware detection.},  
Keywords = {Android Malware, Machine Learning, Cloud Computing},
volume = {21},
Number = {3}, 
pages = {137-148}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.3.137},
url = {http://jsdp.rcisp.ac.ir/article-1-1336-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1336-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {HassanPourAskari, Abbas and KhatibiBardsiri, Amid and MohammadiGhanatGhestani, Mokhtar},  
title = {IoT privacy for the transmission of data in the field of health using blockchain}, 
abstract ={Data transmission and storage through blockchain is something that many studies have suggested for various security issues. Due to the sensitivities in securing information from patients and medical professionals, the healthcare system has a favorable context for deploying a powerful blockchain system for privacy. Blockchain application in the healthcare system allows physicians to store patient records with high security and make them available to other hospitals and clinics as needed. Besides increasing data transmission and storage security, this reduces data management risks and expenses. The present study emphasizes the confidentiality of healthcare data in the cloud-based Internet of Things (IoT) using blockchain and edge computing. Also this method uses SHA2 hashing and PKI encryption. Therefore, it can optimally provide data confidentiality in medical settings, especially for patients under care. The system proposed in this study includes five parts: users, IoT devices, edge devices, security server, and cloud computing. The proposed method uses the Diffie &#8211; Hellman key exchange in the authentication process for anomality case to achieve the goal of essential security compliance. This method is a cryptographic (encryption) protocol allowing two people or two organizations to create a shared password key without the need for any prior acquaintance and exchange it through an insecure connection path. This study used data sensed by IoT network sensors with medical data for loading on IoT and cloud simulated networks. These data were related to monitoring patients with cardiovascular disease and were sampled on 300 patients. The research dataset belonged to the Cleveland Clinic Foundation for Heart Disease Dataset. The study simulation software was NS-2.35, which used C++ and TCL programming languages. The research findings revealed that if there were no method for data encryption, much of the data would be exposed. This rate reaches 50% for 50 attackers. Encryption using the proposed method causes the percentage of data disclosed to be very slight, and it equals zero, even though attackers may guess the password or data. As network traffic grows, the throughput difference between blocking and non-blocking access methods increases. It suggests that by blocking the attacking nodes&#8217; access, network traffic will not have a detrimental effect on attacking nodes by detecting and preventing their activity. However, if the access of these nodes is not blocked, the destructive impact is very high, and the network traffic will grow slightly. According to these results, in terms of SLA violation, the system is in a situation where even in case of an attack, there is no SLA violation and the efficiency is maintained. Also, the percentage of access to useful information by the hacker will be close to zero. By preventing the entry of malicious nodes, the throughput increases by about 30%. Some other advantages of this method are its high flexibility and comparability, robustness, and relatively low execution time and delay, which is caused by the use of cloud edge. It is estimated that the improvement rate of the proposed method is more than 5% compared to other related approaches. In the design presented in this study, the processes are highly simplified, and there will be a relatively low processing overhead. At the same time, the steps meet all the requirements for cloud and IoT data centers in healthcare applications.},  
Keywords = {cloud-based IoT, health record system, blockchain, data confidentiality, edge computing},
volume = {21},
Number = {3}, 
pages = {149-178}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.3.149},
url = {http://jsdp.rcisp.ac.ir/article-1-1314-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1314-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2024}  
}

@article{ 
author = {Sarabi, Homeyra and AbdaliMohammadi, Fardi},  
title = {Diagnosing Children\'s Developmental Disorders By A Transfer Learning Based Architecture Using Knowledge Distillation}, 
abstract ={Enhancing medical devices with Internet of Things (IOT) and diagnostic artificial intelligence technology, while considering the constraints of these systems, has the potential to modernize and enhance the diagnostic approach of future generations of Internet of Things systems in healthcare. The radiography device, commonly used in paraclinical settings, is widely used in various hospital departments. The automated assessment of abnormalities and bone age from radiographic images of the left hand assists radiologists, pediatricians, and forensic experts in determining the developmental stage of young individuals. The IoT devices in medicine are unable to process large amounts of data due to limited resources. This article uses a teacher-student network for bone age classification, using the decomposed knowledge distillation model of convolutional neural networks. This approach minimizes the computational resources needed for edge devices. The proposed method is comprised of two sequential steps.&#160; In the preprocessing step, the initial phase involves the elimination of non-clinical data and artifacts. This is followed by the extraction of region of interest (ROI). In this phase of the procedure, only the hand portion of the patient&#39;s X-ray remains for further evaluation. The subsequent phase involves the delineation of the boundaries of the region of interest. This is necessary because, in certain age groups, some bones are not ossified. Consequently, reliance on bones as landmarks is precluded.&#160; In the second step, The extracted ROI from the preceding step is utilized to train the teacher model. The student model utilizes the teacher model&#39;s knowledge to learn how to predict patient age. Therefore, the present study puts forth transfer learning methodologies founded on the distillation of knowledge, with the aim of facilitating the transference of knowledge between teacher and student models. &#160;The proposed method is based on the data set of the Digital Hand Atlas (DHA) database. The evaluation criteria used in this work are Accuracy, recall, permission and mean absolute error (MAE). The proposed model achieves 96/47% test accuracy for bone age classification.},  
Keywords = {Developmental Disorders, Knowledge Distillation, Bone Age, Deep Learning},
volume = {21},
Number = {4}, 
pages = {1-14}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.4.1},
url = {http://jsdp.rcisp.ac.ir/article-1-1399-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1399-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2025}  
}

@article{ 
author = {RahimiResketi, Mahsa and Motameni, Homayun and Akbari, Ebrahim and Nematzadeh, Hossei},  
title = {Tag recommendation in social networks with the help of text summarization and KNN}, 
abstract ={In recent years, the utilization of social networks has surged markedly, with interest in their use escalating daily. A pivotal concern is augmenting the number of views for individuals&#39; posts or messages to enhance their popularity. The most effective means to achieve this objective is through the use of tags. Tags significantly contribute to the organization and retrieval of existing data, and the automatic generation of tags has garnered substantial attention. Tag recommendation from textual sources can be approached as a text extraction issue. This paper endeavors to propose a comprehensive set of suggested keywords derived from data via advanced text summarization techniques, culminating in the presentation of a sophisticated tag recommender. Consequently, this research introduces an innovative and robust solution by integrating clustering, summarization, and recommendation methodologies. Initially, utilizing the Bag of Words (BoW) model, comprehensive word parsing and extraction of word roots are performed. This process yields a bag of words capable of facilitating deep semantic exploration. The data is meticulously simplified to its core elements, with prepositions and repetitions omitted. Verbs, due to their high frequency and significance depending on the context of the sentence or post, are mined separately. Other words are judiciously selected based on their frequency and importance, and stored with their repetition counts. Subsequently, employing the K-Nearest Neighbor (KNN) clustering algorithm, the data is clustered, and the cluster representatives serve as the output tags. A slight modification is made to the KNN algorithm by incorporating the Explicit Semantic Analysis (ESA) method for precise scale calculations. The proposed solution was rigorously evaluated on two public datasets: TPA, extracted by Aminer, and AG, extracted by ComeToMyHead. The AG dataset comprises 127,600 news articles, categorized into four distinct tag types. Each category contains 30,000 training samples and 1,900 test samples, with a total of 31,900 tags representing global, sports, business, and scientific concepts. The findings of this study were compared with those from 13 similar research papers, which fall into four distinct categories: machine learning, long-short-term memory (LSTM), convolutional neural network (CNN), and capsule-based models. The comparative analysis revealed that the proposed method demonstrates superior accuracy, comprehensive coverage, and an enhanced F-measure. The integration of advanced text analytics techniques underscores the significance of this study in the broader context of information retrieval and data mining. By harnessing the power of semantic analysis and machine learning, this research provides a novel framework that not only enhances the efficiency of tag recommendation systems but also contributes to the theoretical foundation of automated keyword extraction. The implications of these findings are far-reaching, with potential applications extending beyond social networks to other domains requiring efficient data organization and retrieval.},  
Keywords = {label recommendation, text summarization, word embedding, k-nearest neighbor, BoW},
volume = {21},
Number = {4}, 
pages = {15-28}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.4.15},
url = {http://jsdp.rcisp.ac.ir/article-1-1326-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1326-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2025}  
}

@article{ 
author = {Abdolrazzagh-Nezhad, Majid and Kherad, Mehdi},  
title = {Discovering Influential Nodes in Social Networks based on modified independent cascade model and Fuzzy Multi-Objective Genetic Algorithm}, 
abstract ={In recent years, social networks have become an integral part of people&#39;s lives and play a significant role in the real world. The primary aim of influence maximization problem is finding a set of nodes in the network which can maximize the influence if the diffusion process starts from them. Therefore, the problem&#8217;s goal is to find influential people in large scale real social networks. The penetration phenomenon is carried out according to an influence model in the network. Two independent cascade and linear threshold influence models, the most common of which is the independent cascade model, are utilized for broadcasting in the network. Theoreticaly, optimizing the selection influential nodes problem is NP-hard in both models. The problem will start by considering the social network&#8217;s graph, a specific influence model and a given number k. The problem&#8217;s goal is to select k nodes (users) from the graph (network) as influential nodes, so that the number of active nodes is maximized at the end of the diffusion process. Due to the influence maximization problem and finding influential people is an NP-hard optimization problem in the social network, meta-heuristic algorithms can be used to solve the problem. With regard to the privous researches, there is just one objective function as the problem&#8217;s goal and it is maximizing the number of effective nodes of the diffusion model. While other objective functions such as maximizing the number of effective nodes in the diffusion model and minimizing the budget value k (the number of initial nodes as seed) are not considered, minimizing the time required for effective diffusion can be achieved by having k initial nodes. Although various models for the problem and various optimization algorithms have been presented to discover influential nodes in social networks, paying attention to the multi-objective nature of the problem and improving the performance of the proposed optimization algorithms are a serious research challenge in this field. In this paper. A fuzzy version of the NSGA-II as a multi-objective genetic algorithm, whose mutation and crossover rates are adjusted by Fuzzy Inference System, is utilized to simultaneously optimize the three objectives of maximizing the number of effective nodes, minimizing the number of initial nodes and minimizing the required diffusion time. In the proposed method, the Expected Diffusion Value (EDV) of the diffusion model is replaced instead the simulation of the independent cascade diffusion model with heavy calculations to calculate the diffusion spread (the number of effective nodes of the diffusion model). Therefore, the EDV function is satisfied as the thied objective (minimizing the required diffusion time). The second objective function can also be converted into a maximization function by considering N-k nodes, where k is the number of selected initial nodesand N is the total graph nodes. The decimal numerical coding with fixed length is used in the proposed method. Based on the coding, each chromosome has k genes in the search space. The integer part of each gene is the selected node number. The decimal part is also used to determine whether that initial node exists or not. In the maximizing influence process using the fuzzy NSGA-II algorithm, the solution space (chromosomes) consists of k number of initial nodes, which should be encoded into the fuzzy NSGA-II comprehensible space. Also, a Fuzzy Inference System is proposed to adjust the mutation and recombination rates for filling up a serious challenge of genetic algorithms. In the fuzzy system, NF and FitBest are considered as two input variables, and Pm (mutation rate) and Pc (crossover rate) are returned as outputs NF is the number of chromosomes in the first frontier (F1) of the multi-objective genetic algorithm and FitBest is the average of the normalized objective functions for the chromosomes in F1. To analysis the efficiency of the proposed method, the obtained exprimental results have been compared with conventional maximizing influence methods, i.e., degree centrality, distance centrality, closeness centerality, betweenness, eigenvector and page rank methods, non-fuzzy version of NSGA-II, the latest methods presented for maximizing penetration based on multi-objective meta-heuristic algorithms, i.e. &#181;GP multi-objective evolutionary algorithm, multi-objective crow search algorithm (MOCSA), greedy randomized adaptive search process algorithm (GRASP) and the Multi-Transformation Evolutionary Framework (MTEF) on five benchmark graph datasets Arenasjazz, Canetscience, EgoFacebook, Higgs-Reply and Slashdot. This comparison is based on four criteria: EDV, cost (the number of nodes selected as seed), the influence expansion criterion &#963;(s), i.e. the number of active nodes with the independent cascade (IC) propagation model, and the execution time of the method in seconds. The obtained results show the superiority of the proposed method over the other methods.},  
Keywords = {Influence Maximization Problem, Social Network, Genetic Algorithm with Non-Dominant Sorting, Fuzzy System},
volume = {21},
Number = {4}, 
pages = {29-48}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.4.29},
url = {http://jsdp.rcisp.ac.ir/article-1-1380-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1380-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2025}  
}

@article{ 
author = {Gholizade, Masoume and soltanizadeh, hadi and Rahmanimanesh, Mohamm},  
title = {Multi-Source Transfer Learning Based on Fuzzy Rules for Improving Image Classification Accuracy}, 
abstract ={Image classification tasks often involve the challenge of acquiring a sufficient number of labeled training samples, a process that is not only expensive but also time-consuming. In response to this issue, researchers have focused on transfer learning algorithms, which capitalize on prior knowledge to enhance a model training. While numerous existing transfer learning methods concentrate on knowledge transfer between a single-source domain and a single-target domain, the complexity of real-world scenarios is often underestimated. Limited studies have delved into domain adaptation within multi-source environments, where transferring knowledge from multiple sources introduces ambiguity and uncertainty into the learning process. To address this challenge, this study proposes the application of fuzzy rule-based transfer learning, leveraging the inherent ability of fuzzy rules to effectively handle uncertainty. One notable aspect of fuzzy transfer learning, and transfer learning in general, is the unresolved question of effectively combining and utilizing knowledge when multiple source domains are available. This issue is particularly pertinent in scenarios involving diverse datasets from various sources. Consequently, the present study introduces a novel approach to multi-source transfer learning anchored in fuzzy rules. By integrating fuzzy logic, the proposed method aims to provide a robust solution to the challenges posed by knowledge transfer in scenarios with multiple source domains. This research contributes to advancing transfer learning methodologies, offering a nuanced perspective on handling uncertainty in multi-source environments by applying fuzzy rule-based techniques. In conclusion, the significance of transfer learning in image classification tasks is underscored by the inherent challenges of acquiring labeled training data. The conventional focus on single-source to single-target domain transfer has limitations, prompting a shift towards addressing the more realistic and challenging scenarios of multi-source domain adaptation. This study introduces a pioneering approach to multi-source transfer learning, utilizing fuzzy rule-based techniques to effectively navigate the complexities introduced by knowledge transfer from multiple sources. Through this contribution, the research aims to propel advancements in transfer learning methodologies and foster a more comprehensive understanding of handling uncertainty in multi-source environments.},  
Keywords = {Machine learning, transfer learning, fuzzy rules, multi-source domain adaptation.},
volume = {21},
Number = {4}, 
pages = {49-66}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.4.49},
url = {http://jsdp.rcisp.ac.ir/article-1-1401-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1401-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2025}  
}

@article{ 
author = {ghorbani, masoumeh and esmaeili, leil},  
title = {Application of web usage mining to investigate online shopping behavior via PC versus mobile devices: evidence from click-stream data}, 
abstract ={In recent years, the widespread use of smartphones and the rapid growth of mobile technologies have significantly transformed e-commerce, leading to the rise of mobile commerce (m-commerce). Mobile commerce services such as mobile banking, mobile payments, and mobile shopping have gained substantial traction. According to Statista, mobile retail sales in the United States exceeded $360 billion in 2021 and are projected to nearly double to approximately $710 billion by 2025. Furthermore, by the end of 2021, nearly one-third of U.S. internet users reported making weekly purchases online via their mobile devices. This unprecedented growth underscores the need to explore user behavior in online shopping, particularly through mobile platforms. Several factors influence online shopping behavior, with the device used for browsing and purchasing playing a critical role. As smartphones become the dominant means of internet access, understanding the behavioral differences between mobile and desktop users becomes increasingly important. Although mobile commerce is a subset of e-commerce and shares similarities such as convenience and speed, notable differences exist due to device characteristics. These include the constant availability of smartphones, their lower computational power compared to desktops, and their smaller screen sizes, which can negatively impact the user experience during complex transactions. Research has shown that mobile-specific features, including screen size, speed, security, and website optimization for mobile users, influence browsing and shopping behaviors. Despite the growing recognition of these differences, limited studies have compared user behavior between mobile and desktop platforms in e-commerce settings. This study addresses this gap by analyzing user behavior on the Basalam platform, a prominent Iranian social e-commerce marketplace that supports both desktop and mobile shopping. The primary objective is to empirically examine whether and how user browsing behaviors differ between mobile and desktop platforms. The analysis adopts a novel approach inspired by the work of Orit Raphaeli et al., utilizing sequential association rule mining to uncover frequent navigation patterns and their implications for user interaction and purchase likelihood. Unlike Raphaeli&#8217;s dataset, which focuses on specific web page content, this study employs server-side event logs from Basalam to generalize findings and enhance applicability across e-commerce platforms. The Basalam dataset represents user interactions captured through server logs, documenting user activities on the platform. The preprocessing steps differ from those in Raphaeli&#8217;s study due to variations in data structure, features, and timeframes. The proposed methodology creatively applies sequential association rule mining to episodes of user activity rather than specific web pages, identifying patterns that influence purchase outcomes without focusing on content-specific details. The findings reveal distinct behavioral trends between desktop and mobile users. Desktop sessions are task-oriented, resulting in higher conversion rates, while mobile users demonstrate exploratory browsing patterns. Notably, certain navigation sequences were associated with higher purchase probabilities across both platforms. This research contributes to the field in several ways: It is the first study of its kind focusing on the browsing behavior of Iranian e-commerce platforms, comparing mobile and desktop interactions. The methodology adapts and extends prior approaches to accommodate differences in dataset characteristics, providing a scalable framework for behavioral analysis. The results hold significant implications for e-commerce strategies, offering insights for enhancing user experience, optimizing platform design, and improving conversion rates across devices. By analyzing Basalam&#8217;s event logs, this study provides a comprehensive understanding of user behavior in mobile and desktop contexts, highlighting the strategic importance of platform-specific design in the evolving landscape of digital commerce.},  
Keywords = {M-commerce, E-commerce, Online browsing behavior, Navigation patterns, Footstep graph, Web usage mining, Sequential association rule mining},
volume = {21},
Number = {4}, 
pages = {67-84}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.4.67},
url = {http://jsdp.rcisp.ac.ir/article-1-1366-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1366-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2025}  
}

@article{ 
author = {rastgoo, mohammad and Ghaffari, Hamid Rez},  
title = {A location recommender in social networks based on location based on deep learning}, 
abstract ={The potential of social networks to extract valuable insights into user behavior has become a focal point of research. With the proliferation of social media platforms, people are increasingly sharing their experiences online. This wealth of user-generated data provides unique opportunities to understand movement patterns and predict future behavior. Location-based social networks like Foursquare exemplify this, allowing users to check in at various locations and enabling researchers to analyze these data points.By analyzing the data collected from these platforms, we can uncover patterns in user behavior, such as frequently visited locations and the factors influencing these choices. This information can be invaluable for businesses and urban planners.To improve the accuracy of predicting a user&#39;s next location, this study focuses on identifying the most influential friends or individuals in a user&#39;s social network. Factors such as the strength of these relationships, historical visit data, and temporal-spatial characteristics are considered. Additionally, the study emphasizes the importance of data quality, focusing on locations that have been visited more than 100 times to ensure reliability. A key aspect of this research is understanding the influence of social connections on individual behavior. By analyzing the overlap in visited locations between friends, the study aims to identify the most influential friends for each user. These influential friends are then used to predict the user&#39;s next location. The proposed method employs machine learning techniques, specifically RandomForest and recurrent neural networks (LSTM, RNN, and GRU), to predict user behavior. RandomForest is used to analyze the data and identify the most significant features, while recurrent neural networks are employed to model the sequential nature of user behavior. Among these, LSTM achieved the highest accuracy of 71% in predicting users&#39; next locations.This research demonstrates that combining artificial intelligence with spatial-temporal data can provide profound insights into human behavior in urban and digital environments. By understanding these patterns, businesses can tailor their offerings to individual customers, and urban planners can design more efficient and user-friendly cities.},  
Keywords = {Location-based social networks, recommender systems, spatial data mining},
volume = {21},
Number = {4}, 
pages = {85-96}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.4.85},
url = {http://jsdp.rcisp.ac.ir/article-1-1365-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1365-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2025}  
}

@article{ 
author = {Haghparast, Daniyal and Fotouhi, Ali Mohamm},  
title = {Fatigue and drowsiness detection of the car driver based on image processing and artificial intelligence on the mobile phone}, 
abstract ={One of the important factors in traffic accidents is the fatigue and drowsiness of the driver. In this paper, by using the driver&#39;s face detection and eye state recognition based on image processing and artificial intelligence, the driver&#39;s drowsiness is detected, and appropriate alarms sound to wake up the driver. The proposed method is implemented on the driver&#39;s mobile phone and uses the facilities of the phone, including processor, camera, and alarm, so it requires no additional hardware in the car. The method used and implemented in order to detect and determine the position of the face is based on the Hare-Cascade algorithm. In order to further speed up the algorithm by combining the two stages of eye detection and eye state detection, the Hare-Cascade method has been used to detect open eyes in the face area. The proposed algorithm, while providing the necessary accuracy, unlike the existing numerous and advanced algorithms, including algorithms based on deep learning, has a low computational cost and can be implemented in real time on different types of smart mobile phones. Also, by adjusting the sensitivity of the software by the user, based on the detection of one or two open eyes in the area of the face and the time between two consecutive frames of not detecting open eyes, increasing the number of correct alarms and reducing the number of false alarms can be controlled. In this research to train and increase the accuracy of the intelligent model used, a database of 500 suitable images in different driving situations was prepared and used. Experimental results on 20 test videos in different driving situations show the proper performance of the designed system by creating 95% of the expected alarms. Based on the results of numerous and various experimental tests with the acceptable performance of the product of this applied research in detecting driver drowsiness and creating correct alarms, it seems that if used by drivers, it can prevent many car accidents.},  
Keywords = {Driver fatigue and drowsiness, image processing, artificial intelligence, face and eye detection, mobile phone application},
volume = {21},
Number = {4}, 
pages = {97-112}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.4.97},
url = {http://jsdp.rcisp.ac.ir/article-1-1377-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1377-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2025}  
}

@article{ 
author = {Taheri, Mohammad Amin and Shenassa, Mohammad Ebrahim and Minaei-Bidgoli, Behrouz and Hossayni, Sayyed Ali},  
title = {Noor-Vajeh: A Benchmark Dataset for Keyword Extraction from Persian Papers}, 
abstract ={There are various ways to express the overall intention and focal points of a text, and keywords might be the most appropriate choice. Keywords are defined as the most prominent phrases in a document that convey its main message. By extracting relevant words and phrases from a text, keyword extraction can help to uncover meaningful patterns in the text and provide an overview of the content. It can also help highlight the most significant concepts in a text and focus the attention of a machine learning algorithm on them. Keyword extraction is an imperative subtask of natural language processing. By reducing the complexity of the text and making it easier to process, keyword extraction can be used as the basis for many other processing tasks such as text classification, clustering, and summarization. By extracting keywords from text, a machine can better understand the meaning and context of the text. This enables it to better analyze the text, recognize patterns, and make more accurate decisions. It can also reduce the amount of time it takes to process the text by eliminating unnecessary words and focusing on the most important words. Many datasets are proposed for evaluating keyword extraction methods in Persian, most of which only contain authors&#8217; keywords and do not cover all potential ones. Thus, using such datasets leads to incorrect judgments about the accuracy of the suggested supervised and unsupervised methods. In this paper, we introduce Noor-Vajeh, a Persian keyword extraction dataset of about 1400 scientific papers. We asked experts to extract potential keywords besides the authors&#8217; keywords to complete the keywords set for each article. The resulting dataset is a valuable resource for ongoing research into Persian keyword extraction. To evaluate the dataset to be used as a benchmark, we tested several unsupervised keyword extraction methods. We used these methods because, compared to supervised methods, they take less time to execute and require minimal to no manual tuning. Moreover, they are able to extract keywords with a high degree of accuracy and generalize well to different articles. Furthermore, due to the wide variety of categories of unsupervised learning methods, graph-based methods have been regularly applied in different projects, so we describe and use some of their most famous ones, such as TextRank, SingleRank, and PositionRank. These algorithms can identify important words and phrases in a given text, as well as identify relationships between them. Doing so can provide insights into the overall structure and meaning of the text. This makes them especially useful for finding patterns and making predictions in a variety of tasks, such as machine translation, text summarization, and sentiment analysis. Furthermore, graph-based methods are highly versatile and can be adapted to different datasets and tasks. This makes them ideal for use in benchmark datasets. The results inferred from these methods confirm the comparisons made between the methods employed in other papers.},  
Keywords = {Keyword Extraction, Persian Dataset, Unsupervised Learning, Graph-Based Methods, Information Retrieval},
volume = {21},
Number = {4}, 
pages = {113-123}, 
publisher = {Research Center on Developing Advanced Technologies},

doi = {10.61186/jsdp.21.4.113},
url = {http://jsdp.rcisp.ac.ir/article-1-1340-en.html},  
eprint = {http://jsdp.rcisp.ac.ir/article-1-1340-en.pdf},  
journal = {Signal and Data Processing},  
issn = {2538-4201}, 
eissn = {2538-421X}, 
year = {2025}  
}

