<?xml version="1.0" encoding="UTF-8"?><!DOCTYPE article PUBLIC "-//NLM//DTD Journal Publishing DTD v2.0 20040830//EN" "journalpublishing.dtd"><article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" dtd-version="2.0" xml:lang="en" article-type="research-article"><front><journal-meta><journal-id journal-id-type="nlm-ta">Online J Public Health Inform</journal-id><journal-id journal-id-type="publisher-id">ojphi</journal-id><journal-id journal-id-type="index">45</journal-id><journal-title>Online Journal of Public Health Informatics</journal-title><abbrev-journal-title>Online J Public Health Inform</abbrev-journal-title><issn pub-type="epub">1947-2579</issn><publisher><publisher-name>JMIR Publications</publisher-name><publisher-loc>Toronto, Canada</publisher-loc></publisher></journal-meta><article-meta><article-id pub-id-type="publisher-id">v18i1e84966</article-id><article-id pub-id-type="doi">10.2196/84966</article-id><article-categories><subj-group subj-group-type="heading"><subject>Original Paper</subject></subj-group></article-categories><title-group><article-title>Automating Diagnosis of Skin Neglected Tropical Diseases via Patient Metadata through Machine Learning Model with Adaptive Balancing and Dual Cross-Validation: Retrospective Diagnostic Accuracy Study</article-title></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name name-style="western"><surname>Minyilu</surname><given-names>Yohannes</given-names></name><degrees>MSc</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Yimer</surname><given-names>Mohammed Abebe</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff1">1</xref></contrib><contrib contrib-type="author"><name name-style="western"><surname>Meshesha</surname><given-names>Million</given-names></name><degrees>PhD</degrees><xref ref-type="aff" rid="aff2">2</xref></contrib></contrib-group><aff id="aff1"><institution>Faculty of Computing and Software Engineering, Arba Minch Institute of Technology, Arba Minch University</institution><addr-line>P.O. Box 21</addr-line><addr-line>Arba Minch</addr-line><country>Ethiopia</country></aff><aff id="aff2"><institution>School of Information Science, College of Natural and Computational Sciences, Addis Ababa University</institution><addr-line>Addis Ababa</addr-line><country>Ethiopia</country></aff><contrib-group><contrib contrib-type="editor"><name name-style="western"><surname>Mensah</surname><given-names>Edward</given-names></name></contrib></contrib-group><contrib-group><contrib contrib-type="reviewer"><name name-style="western"><surname>Famotire</surname><given-names>Akinwale</given-names></name></contrib><contrib contrib-type="reviewer"><name name-style="western"><surname>Tozammel Hossain</surname><given-names>K S M</given-names></name></contrib></contrib-group><author-notes><corresp>Correspondence to Yohannes Minyilu, MSc, Faculty of Computing and Software Engineering, Arba Minch Institute of Technology, Arba Minch University, P.O. Box 21, Arba Minch, Ethiopia, +251 911434681; <email>yohannes.minyilu@amu.edu.et</email></corresp></author-notes><pub-date pub-type="collection"><year>2026</year></pub-date><pub-date pub-type="epub"><day>21</day><month>8</month><year>2026</year></pub-date><volume>18</volume><elocation-id>e84966</elocation-id><history><date date-type="received"><day>28</day><month>09</month><year>2025</year></date><date date-type="rev-recd"><day>25</day><month>06</month><year>2026</year></date><date date-type="accepted"><day>18</day><month>07</month><year>2026</year></date></history><copyright-statement>&#x00A9; Yohannes Minyilu, Mohammed Abebe Yimer, Million Meshesha. Originally published in the Online Journal of Public Health Informatics (<ext-link ext-link-type="uri" xlink:href="https://ojphi.jmir.org/">https://ojphi.jmir.org/</ext-link>), 21.8.2026. </copyright-statement><copyright-year>2026</copyright-year><license license-type="open-access" xlink:href="https://creativecommons.org/licenses/by/4.0/"><p>This is an open-access article distributed under the terms of the Creative Commons Attribution License (<ext-link ext-link-type="uri" xlink:href="https://creativecommons.org/licenses/by/4.0/">https://creativecommons.org/licenses/by/4.0/</ext-link>), which permits unrestricted use, distribution, and reproduction in any medium, provided the original work, first published in the Online Journal of Public Health Informatics, is properly cited. The complete bibliographic information, a link to the original publication on <ext-link ext-link-type="uri" xlink:href="https://ojphi.jmir.org/">https://ojphi.jmir.org/</ext-link>, as well as this copyright and license information must be included.</p></license><self-uri xlink:type="simple" xlink:href="https://ojphi.jmir.org/2026/1/e84966"/><abstract><sec><title>Background</title><p>Skin neglected tropical diseases (NTDs) are the most prevalent diseases worldwide, affecting people living in resource-limited areas with low health care services and trained professionals. While machine learning (ML)&#x2013;based diagnostic tools can be used for initial clinical assessment and patient screening, especially in resource-limited areas (including in Ethiopia), little effort has been made in this area.</p></sec><sec><title>Objective</title><p>This pilot study develops a foundational ML model for the diagnosis of skin NTDs using patient metadata to analyze the feasibility of ML-based models for skin NTDs by identifying and experimentally evaluating 8 ML models.</p></sec><sec sec-type="methods"><title>Methods</title><p>For this study, we acquired a tabular skin NTD diagnostic dataset collected from a specific affected district in the southwest of Ethiopia. We used the data in 3 different structures, which include using the initial dataset (IDS) that contains huge null values, using a final dataset (FDS) created through preliminary preprocessing, and a third dataset created by applying feature engineering (FEFDS). Selecting 8 ML models, we trained the models in 4 major experimental settings: baseline training, handling structural missing values, handling severe class imbalance through conditional class weighting, and a hybrid approach based on robust dual cross-validation (CV) consisting of an outer repeated stratified <italic>k</italic>-fold and nested CV methods. We used the macro and class-specific metrics (such as precision, recall, and <italic>F</italic><sub>1</sub>-score), including balanced accuracy, due to the severe class imbalance. Feature importance scores are also used for evaluating overall model performance.</p></sec><sec sec-type="results"><title>Results</title><p>After the final training applying the hybrid approach, 4 models scored a perfect test score (1.0) across all the metrics and all experiments except na&#x00EF;ve Bayes and multilayer perceptron (MLP), similarly scoring 0.997 balanced accuracy, 0.97 recall, and 0.985 <italic>F</italic><sub>1</sub>-score. In the same experiment, the nested CV loop revealed a slight performance drop for light gradient boosting machine (LightGBM) and extreme gradient boosting (XGBoost), though both models similarly maintained higher mean scores of balanced accuracy and macro recall of 0.993 (SD 0.013), including the mean macro <italic>F</italic><sub>1</sub>-score of 0.996 (SD 0.007), scoring declining performance with a mean score of 0.993 (SD 0.013) in balanced accuracy, macro recall, and <italic>F</italic><sub>1</sub>-score, showing predictive biases. In terms of feature utilization, only CatBoost showed optimal feature ranking, while 4 models showed over-feature utilization (selection bias), with 3 models using small subsets of features (showing feature parsimony).</p></sec><sec sec-type="conclusions"><title>Conclusions</title><p>Overall, this study has been highly challenged by data scarcity, class imbalances, limited disease representation, specific geographic representation, and lack of more data modalities. Hence, further studies are suggested to confirm the results on larger datasets having a representative distribution of disease classes and geographic locations.</p></sec></abstract><kwd-group><kwd>skin NTDs</kwd><kwd>patient metadata</kwd><kwd>ML models</kwd><kwd>feature importance</kwd><kwd>class weighting</kwd><kwd>dual cross-validation</kwd><kwd>machine learning</kwd><kwd>neglected tropical diseases</kwd></kwd-group></article-meta></front><body><sec id="s1" sec-type="intro"><title>Introduction</title><sec id="s1-1"><title>Background</title><p>Neglected tropical diseases (NTDs) are the most prevalent diseases worldwide. According to the World Health Organization (WHO), NTDs represent about 20 different diseases, such as podoconiosis, scabies, tungiasis, buruli ulcer, cutaneous leishmaniasis (CL), leprosy, mycetoma, and rabies [<xref ref-type="bibr" rid="ref1">1</xref>-<xref ref-type="bibr" rid="ref3">3</xref>]. Ten of the NTDs, according to WHO, are known to exhibit primary skin symptoms (hence skin NTDs) or related clinical symptoms that often can cause bodily harm and disfigurement, which can cause disability [<xref ref-type="bibr" rid="ref4">4</xref>]. As a tropical country, the majority of NTDs are present in Ethiopia [<xref ref-type="bibr" rid="ref5">5</xref>], particularly in the remote areas of the country. Recent figures showed that Ethiopia remains one of the countries with a high burden of NTDs, where more than 75 million people are at risk of contracting at least one of the NTDs [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref7">7</xref>].</p><p>Clinically, different diagnostic procedures are used for the diagnosis of skin NTDs, where the dermatological approach is the primary method used. Marks et al [<xref ref-type="bibr" rid="ref8">8</xref>], on behalf of the WHO Diagnostic Technical Advisory Group (DTAG), highlighted that the integration of clinical and laboratory diagnostic services is the fundamental approach for skin NTDs, where initial clinical assessment of patients, followed by confirmatory tests (laboratory diagnostics), are the 2 critical steps toward enhanced diagnostics of skin NTDs. However, establishment of laboratory facilities in most NTD-vulnerable areas (based on current real-world scenarios) is hardly feasible due to higher resource requirements. Such scenarios guide efforts to look for alternative, technology-assisted solutions through decentralized diagnostic frameworks. Tadesse et al [<xref ref-type="bibr" rid="ref9">9</xref>], who studied the decentralized diagnosis of CL in Ethiopia, claimed that with minor adjustments to health care facilities and continuous training of primary health care professionals, the diagnosis and treatment of skin NTDs can be decentralized to the health care center levels.</p><p>Furthermore, recent studies such as [<xref ref-type="bibr" rid="ref6">6</xref>,<xref ref-type="bibr" rid="ref10">10</xref>] suggested the feasibility of mobile-based diagnostic platforms for resource-limited settings. The authors in [<xref ref-type="bibr" rid="ref11">11</xref>] and [<xref ref-type="bibr" rid="ref12">12</xref>] used advanced AI-based diagnostic tools based on machine learning (ML) and deep learning (DL) methods for the diagnosis of skin NTDs. The use of such tools would facilitate the creation of localized diagnostic platforms, as they provide the opportunity for patients to get health care services around their living areas, especially in resource-limited areas. This triggers the need to investigate and implement NTD diagnostics based on effective point-of-care tools for skin NTD diagnosis [<xref ref-type="bibr" rid="ref8">8</xref>,<xref ref-type="bibr" rid="ref9">9</xref>,<xref ref-type="bibr" rid="ref13">13</xref>]. Accordingly, ML-based diagnostic tools based on clinical patient data can be used for initial clinical assessment of patients and patient screening, as well as patient record organization and management, especially in resource-limited areas. Most importantly, in resource-constrained areas, ML-based diagnostic tools implemented based on the differential diagnostic approach can be used by middle-level health care workers for early diagnosis of patients with skin NTDs. Therefore, this study investigates the feasibility of an ML-based skin NTD diagnostic framework by proposing an ML-based diagnostic model for skin NTDs using patient metadata. Accordingly, the study is designed to be a pilot study that intends to establish the foundational basis for the realization of an intelligent diagnostic model for skin NTDs.</p><p>In this pilot study, we establish a robust and integrated framework for the systematic investigation of the feasibility of high-performing diagnostic ML models for skin NTDs based on tabular clinical data. This includes establishing integrated pipelines for data preprocessing, handling missing data, and addressing class imbalance, including the identification, synthesis, and evaluation of ML architectures. For this study, we used a novel dataset created using newly collected tabular patient metadata of 3 skin NTDs (podoconiosis, scabies, and tungiasis) to train and evaluate 8 selected ML models. We further conducted a comparative performance analysis of the trained models to identify the model with the highest performance and reliability, as well as the optimal ML strategy that resulted in the best performance. The study also analyzed the impact of missing values and strategies for handling missing data (imputation methods), which serves as a benchmark for further studies. However, as the study presents a foundational framework, it does not intend to build a full-fledged and deployment-ready diagnostic model; as such, complete systems require sufficiently large datasets, high computational resources, complex and multistaged evaluation, and regulatory procedures before deployment.</p><p>Overall, apart from being the first study that proposes the application of ML-based diagnostics for skin NTDs using tabular data, this study presents other novel contributions, which include presenting a novel tabular skin NTD dataset and the establishment of a harmonized ML pipeline. Accordingly, we identified and used a set of ML strategies and created a robust ML model development pipeline for the systematic application, evaluation, and analysis of methods and models, which include comprehensive baseline training setup, rigorous feature ranking and importance analysis, robust implementation of missing data handling, and the harmonized approach&#x2014;the robust ML pipeline that synergistically integrates conditional class weighting with a dual CV evaluation approach based on the integration of 2 CV methods (repeated and nested CV methods). The collective implementation of all these strategies helps in establishing an overall robust ML pipeline, clearly marking the novelty of our study.</p></sec><sec id="s1-2"><title>Related Works</title><sec id="s1-2-1"><title>Machine Learning Models for Structured Data</title><p>Developing ML-based diagnostic tools requires carefully devised strategies for the selection, utilization, evaluation, and optimization of the ML models and tools. Accordingly, the type of data to be used, computational resources available for model building and deployment, expected model performances, undesired model behaviors such as overfitting, model optimization strategies to be used, and the nature of the disease to be diagnosed are the major parameters that determine the ML methods and strategies to be used. For small-sized datasets, support vector machines (SVMs), decision trees, and random forests (RF) are the preferred ML algorithms, while the convolutional neural network (CNN)&#x2013;based DL models can also be used [<xref ref-type="bibr" rid="ref14">14</xref>,<xref ref-type="bibr" rid="ref15">15</xref>]. However, the tree-based model extreme gradient boosting (XGBoost) is the most suitable model for tabular data since it requires fewer hyperparameter tuning tasks compared to the DL-based models, as stated in [<xref ref-type="bibr" rid="ref16">16</xref>]. Generally, the gradient-boosted tree models of XGBoost, light gradient boosting machine (LightGBM), and CatBoost are well-suited for tabular data, although they can be challenged by model complexity [<xref ref-type="bibr" rid="ref17">17</xref>], higher computational resource requirements such as GPUs for large datasets [<xref ref-type="bibr" rid="ref18">18</xref>], and possible issues during real-time deployment on edge devices [<xref ref-type="bibr" rid="ref19">19</xref>].</p></sec><sec id="s1-2-2"><title>Machine Learning Methods for Data-Related Constraints</title><p>There are multiple ML strategies and models for handling dataset-related issues, as it is sometimes challenging to acquire complete datasets that are sufficient to train ML and DL models. For small-sized tabular datasets, traditional ML models and ensemble approaches are the most suitable methods due to their stability and interpretability, as indicated in recent studies. Accordingly, the tree-based ensemble models of RF and extra trees are most often the top-performing models for the classification of diseases in dermatology based on tabular patient metadata, including the situation of having small-sized tabular datasets [<xref ref-type="bibr" rid="ref20">20</xref>]. In terms of performance reliability, <italic>k</italic>-nearest neighbor (KNN) and SVM are the most stable models that resulted in superior performance compared to complex neural network&#x2013;based models for small-sized structured clinical datasets [<xref ref-type="bibr" rid="ref21">21</xref>,<xref ref-type="bibr" rid="ref22">22</xref>]. While the boosted tree-based model XGBoost requires a larger number of samples to reach performance stability, the RF model appears to be a more reliable model for small-sized datasets [<xref ref-type="bibr" rid="ref22">22</xref>]. Regarding missing data, studies suggested that the gradient-boosted tree models of CatBoost, XGBoost, and LightGBM are highly robust to missing data when trained on incomplete patient records. Accordingly, CatBoost (with nan_mode enabled) is the most robust model, as it has an internal mechanism to treat &#x201C;NaN&#x201D; values as a separate split direction [<xref ref-type="bibr" rid="ref23">23</xref>]. The skin disease detection study by Aquil et al [<xref ref-type="bibr" rid="ref20">20</xref>] highlighted that XGBoost and LightGBM models have the ability of learning &#x201C;default direction&#x201D; for missing values during training. Accordingly, if a given symptom (feature) has a null value, the model automatically assigns this feature to the branch that minimizes the loss based on other patients.</p></sec><sec id="s1-2-3"><title>Machine Learning Methods for Skin NTD Diagnosis</title><p>Most previous studies conducted for the diagnosis of skin NTDs heavily rely on skin images. Accordingly, the previous studies [<xref ref-type="bibr" rid="ref23">23</xref>-<xref ref-type="bibr" rid="ref25">25</xref>] used DL-based methods to apply diagnostic models for skin NTDs, where all these studies used only skin image-based approaches. On the other hand, previous works that implemented ML methods for the diagnosis of skin NTDs using only structured clinical data were difficult to find, except 2 studies that used tabular patient metadata along with skin image data. Accordingly, the study by Barbieri et al [<xref ref-type="bibr" rid="ref11">11</xref>] used structured clinical data of leprosy patients to integrate with skin lesion images and used ML algorithms that include RF and XGBoost for handling the clinical patient data in their multimodal data fusion study. Similarly, Achary et al [<xref ref-type="bibr" rid="ref12">12</xref>] used tabular clinical metadata to apply multimodal data fusion by combining the metadata with skin images of patients using generative models for text data augmentation. In both studies, the tabular clinical data of patients are not used independently for skin NTD diagnosis; instead, they are merged with image data based on multimodal data fusion methods.</p><p>Overall, the results of the literature review underscored that structured clinical metadata of patients are not fully used, let alone the independent utilization, demonstrating the development of ML-based diagnostic models for skin NTDs based on tabular clinical patient data, indicating a clear research gap. One of the biggest challenges in building AI-based diagnostic tools for skin NTDs is the lack of large-scale, complete, and properly annotated datasets, as affirmed by the study by Achary et al [<xref ref-type="bibr" rid="ref12">12</xref>]&#x2014;the study used generative models to generate textual data to fill the gap of data scarcity. Conversely, the selection and use of specific ML strategies require further investigation, given that skin NTD data are naturally scarce resources, including limited previous studies for comparative analysis of methods. All these factors clearly mark research-wide gaps that forced the authors of this study to conduct this study.</p></sec></sec></sec><sec id="s2" sec-type="methods"><title>Methods</title><sec id="s2-1"><title>Study Design</title><p>This pilot study develops a foundational benchmark diagnostic model for skin NTDs using structured (tabular) clinical patient data by selecting the best-performing model from a set of 8 ML models initially selected for the experiments. For a complete and fair comparison, all models are equally trained on similar datasets, using similar training settings. To achieve this objective, the study relies on capturing and analyzing model performance results by conducting a series of model training experiments. This highlights both the qualitative and quantitative aspects based on the experimental approach. Therefore, this study uses a mixed research strategy. However, this study does not currently intend to realize a fully deployable diagnostic model.</p></sec><sec id="s2-2"><title>Ethical Considerations</title><p>This study uses a novel dataset created by acquiring clinical records consisting of demographic information and disease-specific diagnostic data of patients with skin NTDs. The entire data collection and acquisition processes are carried out in a professional and ethical manner, which include acquiring an ethical clearance letter that authorizes the data collection and use of the collected data for this study, and confirming all legal and ethical issues, including written consent of patients. To achieve this, the authors of this study have acquired a proper ethical clearance letter from the Institutional Review Board committee at Arba Minch University (approval protocol number: YM23161).</p></sec><sec id="s2-3"><title>Data Acquisition and Dataset Preparation</title><sec id="s2-3-1"><title>Data Source</title><p>For this study, we created a new hand-crafted dataset using the data collected from patients with skin NTDs living in Gacho Baba district of the Gamo zone, southwest of Ethiopia, which is one of the remote areas representing underserved communities in terms of health care services. The data were collected by a team of researchers from the Collaborative Research and Training Center for NTDs, College of Medicine and Health Sciences of Arba Minch University, in a project-based research conducted for the assessment of skin NTD burden. For this study, the original (first-hand) data were acquired through institutional collaboration (acknowledgments section) and used to achieve the objective of this study.</p></sec><sec id="s2-3-2"><title>Initial Diagnostics and Confirmation</title><p>The entire data collection was initially conducted in a large-scale community-based mass drug administration (MDA) campaign in the specified community based on on-site clinical diagnostic procedures. At the MDA site, however, extended medical resources like laboratory facilities, specialized medical imaging, or any advanced medical infrastructure were not available. In such settings, the initial clinical diagnostic procedures were performed by trained frontline health care workers (mostly middle-level clinicians and nurses) who were actively involved in the day-to-day clinical diagnostics of patients around the area, and a professional dermatologist from Arba Minch General Hospital who participated during the MDA campaign. Therefore, the on-site diagnostics were performed entirely based on visual clinical examination procedures by the frontline clinicians, with strict diagnostic confirmation by the professional dermatologist.</p><p>As the data collection involved actual field-based diagnoses, the establishment of initial ground truth was challenged by limitations related to field logistics. To overcome this challenge, a hierarchical diagnostic verification protocol was applied by the data collection team to clinically confirm the diagnostics. Hence, each of the initial diagnostic procedures underwent a mandatory clinical review by a professional dermatologist from Arba Minch General Hospital to confirm the initial diagnostics. While this procedure confirms the correct classification of the skin NTD diagnosed based on the signs and symptoms presented by the patients, it also helped in creating higher-level diagnostic fidelity. Overall, the ground truth we used for the classification of the skin NTDs for this study complies with the WHO-standardized clinical case definitions [<xref ref-type="bibr" rid="ref26">26</xref>]. After all these clinical and ethical procedures were completed, the authors of this study used the clinically validated data for the ML-based research in this study after acquiring the proper authorization through ethical clearance.</p></sec><sec id="s2-3-3"><title>Data Collection</title><p>Overall, the data collection was conducted based on a community-based screening approach for assessing the burden of skin NTDs, which included on-site examination, treatment, and registration of patient information. To achieve this, the data collection team used a well-prepared data collection and patient registration form, which was prepared by consulting professional dermatologists from Arba Minch General Hospital. The data collection involved 4029 participants during the initial screening and diagnosis, targeting the identification and diagnosis of 8 skin NTDs that are endemic to Ethiopia, which include mycetoma, leprosy, CL, lymphatic filariasis, podoconiosis (podo), onchocerciasis, tungiasis, and head lice. Finally, according to the data collection report, the specified community was highly affected by 3 of the skin NTDs, namely scabies, tungiasis, and podoconiosis. Thus, we selected these 3 diseases for our study based on the prevalence and availability of sufficient data. The other 5 mentioned diseases have very little (or no data at all) in the specified community at the time of data collection; hence, they are not used in the final dataset.</p></sec><sec id="s2-3-4"><title>Limitations of the Data Collection</title><p>Generally, collecting large-scale skin NTD data from wider geographic areas (like Ethiopia) with complete representation of all endemic diseases is a complex and highly resource-intensive process, which is almost impossible to achieve in a single research project. Hence, our study is limited to a single geographic location in the Gacho Baba district, southwest of Ethiopia. This confinement to the specific location limited the number of endemic diseases to only 3 skin NTDs (scabies, tungiasis, and podoconiosis) out of the 8 overall endemic diseases, which forced us to exclude other skin NTDs like buruli ulcer, leishmaniasis, leprosy, and mycetoma. However, as a preliminary pilot study, we used the acquired skin NTD data to build a benchmark skin NTD diagnostic model, with the potential of extending the disease representation through further efforts.</p></sec><sec id="s2-3-5"><title>Dataset Creation</title><p>During initial collection, the patient data collected for each of the 3 diseases were recorded separately using individual Microsoft Excel files for each disease. Hence, we initially created 3 separate datasets, which represent (1) scabies, which consists of 955 instances having 10 features (6 demographic and 4 disease-specific); (2) tungiasis, consisting of 474 instances having 8 features (6 demographic and 2 disease-specific); and (3) podoconiosis, comprising 66 instances having 11 features (6 demographic and 5 disease-specific), all before data preprocessing. In each of these 3 separate datasets, all disease-specific features (symptoms) are represented in binary categorical form using either &#x201C;1&#x201D; or &#x201C;0&#x201D; to signify the presence or absence of a given disease-specific symptom in a patient, which were applied after clinical confirmation of all diagnoses. For instance, &#x201C;slowenlarging_swelling(feet&#x0026;legs)&#x201D; is a symptom specifically assessed while diagnosing podoconiosis, and all 66 podo patients were assessed if they manifest this particular symptom. Hence, after final clinical confirmation of the diagnosis, this feature is encoded as either &#x201C;1&#x201D; if a given patient showed this symptom or &#x201C;0&#x201D; otherwise, where a similar pattern was used for all symptoms of the 3 diseases.</p><p>Finally, we handcrafted our initial dataset (named it IDS) by merging the 3 individual disease-specific datasets. Originally, IDS comprises a total of 1495 instances with 17 trainable features (6 demographic and 11 diagnostic) and the target variable &#x201C;disease_diagnosed&#x201D; defining the class labels. After initial data preprocessing, IDS is transformed into a new structure having 1495 instances with 28 trainable features (17 demographic and 11 diagnostic) plus the target variable. However, the merger of the 3 separate disease-specific datasets introduced a different challenge due to the disease-specific symptoms, as not all symptoms are applicable to all diseases&#x2014;a specific symptom that can be used to diagnose scabies is not applicable to the other 2 diseases. This created 3 blocks of missing values in IDS, showing a structural missingness issue [<xref ref-type="bibr" rid="ref27">27</xref>]. Hence, using ML-based operations, we created 2 additional datasets through transformation: first, we created a final dataset (FDS), having 28 trainable features (17 demographic and 11 disease-specific features); next, we created a different dataset by applying feature engineering on the final dataset (FEFDS), having 38 trainable features (17 demographic and 11 disease-specific features, including 10 symptom applicability indicator flags introduced after feature engineering). We created all these 3 different datasets in 3 separate CSV files to address structural missingness, ensure reproducibility, and maintain dataset integrity across different experimental settings, thereby allowing us to establish a consistent training and evaluation pipeline.</p></sec></sec><sec id="s2-4"><title>Exploratory Data Analysis</title><sec id="s2-4-1"><title>Dataset Characteristics and Missingness Analysis</title><p><xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> presents the overall distribution of instances (n=1495), disease classes, and missing values in IDS. As shown in Figure S1 of <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, the scabies disease class represents the largest proportion with a total of 955 (63.9%) instances, while tungiasis represents the second largest proportion with 474 (31.7%) instances, and podoconiosis has only 66 (4.4%) instances. Gender-wise, 66.6% (995/1495) of the patients are male, while 33.4% (500/1495) are female. These statistics clearly highlight that the dataset exhibits a severe class imbalance among the 3 disease classes, having an extremely distorted distribution with an overall imbalance ratio of 14.5 between scabies (majority class) and podoconiosis (minority class).</p><p>Correspondingly, Figure S2 in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref> depicts the statistical summary of missing values in the dataset. As stated, our new dataset is created by merging 3 separate disease-specific datasets, each of which has a unique set of clinically relevant symptoms that are not applicable to the other diseases. For instance, the symptom &#x201C;PPSEs&#x201D; (which represents &#x201C;papule, pustule, scratches, and excoriations&#x201D;) is assessed only for scabies patients, where this field is structurally empty for podoconiosis and tungiasis patients. Ultimately, this created 3 blocks of missing values that define the case of structural missingness or missing by design [<xref ref-type="bibr" rid="ref27">27</xref>], where our new dataset contains a total of 11,347 such structural null values. The missing values in our dataset are not random omissions created as a result of the random absence or exclusion of symptoms that are called &#x201C;missing completely at random (MCAR)&#x201D; or &#x201C;missing at random (MAR)&#x201D; [<xref ref-type="bibr" rid="ref28">28</xref>]. Overall, as depicted in <xref ref-type="supplementary-material" rid="app1">Multimedia Appendix 1</xref>, the scabies class has the highest number of null values, with 58.9% (6685/11,347) of the total number of null values (having 7 null values per each of the 955 rows), while tungiasis has 37.6% (4266/11,347), and podoconiosis has 3.5% (396/11,347) since it has the least number of instances.</p></sec><sec id="s2-4-2"><title>Feature Description</title><p><xref ref-type="table" rid="table1">Table 1</xref> presents the overall summary of our first dataset, IDS, including its overall statistical summary, including its overall statistical summary of class distribution, feature categories, and the initial feature sets of IDS.</p><p>As shown, our new dataset (IDS) primarily contains two major categories of patient data: (1) demographics, which include 2 numerical features (age and weight) and 3 categorical features (gender, marital status, educational level, and occupation) of the patient information, and (2) clinical data, which comprises 11 disease-specific symptoms from the 3 classes. Class-wise, podoconiosis has 5 specific features (disease-specific symptoms) that are not applicable to the other 2 diseases, and the scabies class has 4 unique symptoms applicable only to scabies, while there are only 2 tungiasis-specific symptoms (features). Overall, our dataset has 2 features representing numerical data (age and weight), while the remaining features are categorical, which in turn include 3 ordinal categorical features, representing sex, educational level, and one disease-specific feature (the jigger infestation classification feature of tungiasis); 2 nominal categorical features (marital status and occupation), representing the unordered categorical data that include marital status occupation; and 10 binary categorical features, represented as &#x201C;1&#x201D; or &#x201C;0&#x201D; indicating the presence or absence of a given disease-specific symptom. Table S1 in <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref> presents the complete dataset information.</p><table-wrap id="t1" position="float"><label>Table 1.</label><caption><p>Summarized statistical description of the new dataset used for this study.</p></caption><table id="table1" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Features</td><td align="left" valign="top">Podoconiosis (n=66)</td><td align="left" valign="top">Scabies (n=955)</td><td align="left" valign="top">Tungiasis (n=474)</td><td align="left" valign="top"><italic>P</italic> value</td></tr></thead><tbody><tr><td align="left" valign="top">Demographic</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Age (years), mean (SD)</td><td align="left" valign="top">44.68 (13.59)</td><td align="left" valign="top">17.81 (15.17)</td><td align="left" valign="top">21.90 (20.28)</td><td align="char" char="." valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Sex, n (%)</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="char" char="." valign="top">.104</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Female</td><td align="left" valign="top">30 (45.5)</td><td align="left" valign="top">316 (33.1)</td><td align="left" valign="top">154 (32.5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Male</td><td align="left" valign="top">36 (54.5)</td><td align="left" valign="top">639 (66.9)</td><td align="left" valign="top">320 (67.5)</td><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Weight (kg), mean (SD)</td><td align="left" valign="top">52.02 (7.73)</td><td align="left" valign="top">33.31 (15.74)</td><td align="left" valign="top">32.47 (16.70)</td><td align="char" char="." valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Educational level, mean (SD)</td><td align="left" valign="top">1.26 (0.54)</td><td align="left" valign="top">2.34 (1.38)</td><td align="left" valign="top">2.37 (1.61)</td><td align="char" char="." valign="top">&#x003C;.001</td></tr><tr><td align="left" valign="top">Clinical</td><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/><td align="left" valign="top"/></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Disease-specific clinical symptoms</td><td align="left" valign="top">5 (specific symptoms)<list list-type="bullet"><list-item><p>Slow-enlarging swelling</p></list-item><list-item><p>History of barefoot</p></list-item><list-item><p>Living podo in endemic area</p></list-item><list-item><p>Acute attacks</p></list-item><list-item><p>Involving bilateral legs</p></list-item></list></td><td align="left" valign="top">4 (specific symptoms)<list list-type="bullet"><list-item><p>Night-worsened itchy lesion</p></list-item><list-item><p>Family history of similar condition</p></list-item><list-item><p>PPSEs<sup><xref ref-type="table-fn" rid="table1fn1">a</xref></sup></p></list-item><list-item><p>FWABBP<sup><xref ref-type="table-fn" rid="table1fn2">b</xref></sup></p></list-item></list></td><td align="left" valign="top">2 (specific symptoms)<list list-type="bullet"><list-item><p>Body part infested (hand/leg)</p></list-item><list-item><p>Jigger infestation classification</p></list-item></list></td><td align="left" valign="top"/></tr></tbody></table><table-wrap-foot><fn id="table1fn1"><p><sup>a</sup>Papule, pustule, scratches, and excoriations.</p></fn><fn id="table1fn2"><p><sup>b</sup>Finger webs, axilla, breast, buttock, and penis.</p></fn></table-wrap-foot></table-wrap></sec></sec><sec id="s2-5"><title>Data Preparation and Preprocessing</title><sec id="s2-5-1"><title>Data Labeling and Initial Data Preprocessing</title><p>In this study, we used IDS in 3 different settings. First, IDS is used with minimal preprocessing operations for developing baseline models. This dataset is used to mimic the original real-world scenario, which is sometimes characterized by incomplete datasets having multiple missing values. As a base dataset, the majority of the essential data preprocessing operations are applied to IDS, while we applied only normalization for FDS and FEFDS, as these datasets were derived from the feature-encoded IDS. Overall, for data preprocessing, we applied data encoding, which mainly included data encoding using label encoding and one-hot encoding methods, including normalization.</p><p>For encoding the ordinal categorical features (&#x201C;sex,&#x201D; &#x201C;educational_level,&#x201D; &#x201C;JI_classification(#Jig-Inf),&#x201D; including the target variable &#x201C;disease_diagnosed&#x201D;), the Label Encoder method is used. As the remaining 12 features represent nominal categorical data, which include &#x201C;marital_status,&#x201D; &#x201C;occupation,&#x201D; and 10 disease-specific symptoms, we applied the one-hot encoding method to transform the data under these features. Data normalization was the other preprocessing operation we performed. Accordingly, the numerical features of age (with minimum 2 and maximum 77) and weight (with minimum 8 and maximum 80) are normalized using the &#x201C;standard scaler&#x201D; technique to bring these features on the same scale with the other features. This allowed us to maintain a standard normal distribution with zero mean and unit variance within the dataset [<xref ref-type="bibr" rid="ref29">29</xref>].</p></sec><sec id="s2-5-2"><title>Handling Structural Missing Data: Feature Engineering</title><p>As stated, our new dataset (IDS) contains a total of 11,347 structural null values, which are not random omissions created as a result of the random absence or exclusion of symptoms that are called MCAR or MAR [<xref ref-type="bibr" rid="ref28">28</xref>]. Over reliance on conventional imputation techniques such as simple imputation or using other methods like mean, median, or multivariate imputation by chained equation (MICE) would introduce biases that could potentially cause both statistical and clinical issues [<xref ref-type="bibr" rid="ref27">27</xref>,<xref ref-type="bibr" rid="ref28">28</xref>], having contradictory implications: positively, imputing missing values with &#x201C;0&#x201D; values is logically correct, easy to apply, and does not cause any loss of information; while negatively, imputing the missing values with &#x201C;0&#x201D; values will introduce hidden danger if a patient develops a secondary (additional) symptom due to a different cause. In our case, for instance, if our models are trained using the fact that &#x201C;if the values of the &#x2018;Slowenlarging_swelling(feet&#x0026;legs)&#x2019; column are non-zero then, the disease cannot be scabies or tungiasis.&#x201D; However, such restrictions will affect the generalizability of our models when dealing with patients having mixed symptoms if a patient also has swelling on the feet due to different causes. Therefore, this requires careful analysis and selection of strategies to be used for the imputation, which adds robustness to our models.</p><p>Therefore, we created the final dataset by applying a hybrid data encoding strategy: (1) we encoded all symptoms as binary, putting 1 if a given symptom is present (applicable) to a particular disease and putting 0 otherwise; and (2) we introduced a &#x201C;symptom missingness indicator&#x201D; parameters, which flags 1 if a given symptom applies to a particular disease and 0 if that symptom does not apply. Therefore, we created a new column called &#x201C;DISEASE-SYMPTOM_applicable&#x201D; as a missingness indicator, indicating if that &#x201C;SYMPTOM&#x201D; is applicable to the &#x201C;DISEASE&#x201D; or not. For instance, since &#x201C;PPSEs&#x201D; is a symptom applicable only to scabies and not applicable to podoconiosis or tungiasis, the &#x201C;PPSEs_applicable&#x201D; column for all scabies patients is encoded as &#x201C;1&#x201D; whether a patient has &#x201C;PPSEs&#x201D; or not, while this column is encoded as &#x201C;0&#x201D; for podoconiosis and tungiasis patients. The repeated application of this method across 10 disease-specific features adds 10 additional confirmatory columns, which expanded the number of trainable columns of the dataset to 38.</p></sec></sec><sec id="s2-6"><title>Experimental Setup</title><sec id="s2-6-1"><title>Overall Experimental Configurations</title><p>This study is conducted based on the stratified <italic>k</italic>-fold cross-validation (CV) method to evaluate possible performance variations resulting from changes in the dataset and ML models and methods. The hold-out method is excluded from the experimental setup due to high-level overfitting recorded during preliminary experiments. Therefore, we conducted extensive experiments using our new dataset in 3 structures (IDS, FDS, and FEFDS) with an 80/20 train-test split based on the stratified 5-fold CV method. We designed 4 separate experimental settings that include (1) baseline training, where IDS (the original dataset) is used with minimal data preprocessing (only data encoding); (2) handling structural missingness, which includes 2 separate training experiments conducted for handling the structural missing values&#x2014;the first experiment includes training using FDS that applies the simple imputer method, while the next experiment was conducted on FEFDS that applies the symptom applicability indicator flags created after feature engineering&#x2014;each of these two experiments exactly reflects the other except for the datasets used, aiming to comparatively evaluate performance impacts between the two methods; (3) conditional class weighting; and (4) the hybrid approach, the experiment that harmoniously combines class weighting and the robust dual validation methods, all applied on the dataset created using feature engineering (experiment 3).</p></sec><sec id="s2-6-2"><title>Conditional Class Weighting</title><p>To address the severe class imbalance, we applied a class weighting method. Due to the difference in the internal nature of one of our selected models in accepting the &#x201C;sample_weight&#x201D; parameter of the scikit-learn library [<xref ref-type="bibr" rid="ref30">30</xref>], we conditionally applied the class weighting method. Therefore, for all 7 models that inherently accept the &#x201C;sample_weight&#x201D; parameter, these sample weights are automatically computed using the built-in method with the &#x201C;class_weight&#x201D; option set to &#x201C;balanced,&#x201D; then directly passed to their fit methods. Conversely, as the multilayer perceptron (MLP)&#x2013;based model does not accept the &#x201C;sample_weight&#x201D; parameter, we implemented a fallback strategy based on resampling to replicate the instances of the minority class. In this setting, instances from the training set are resampled with replacement based on probabilities equal to the sample weights computed at the beginning of the class weighting method. We applied this method to facilitate the consistent application of the class weighting method across all the selected models for a complete and fair performance comparison.</p></sec><sec id="s2-6-3"><title>The Robust Dual Validation Framework</title><p>Given the size and distribution of our dataset, using a single <italic>k</italic>-fold (5-fold) split might not help to critically assess the stability and reliability of the models, where the repeated stratified <italic>k</italic>-fold CV method plays vital roles in overcoming performance instability issues [<xref ref-type="bibr" rid="ref29">29</xref>,<xref ref-type="bibr" rid="ref31">31</xref>]. Therefore, to establish robust and unbiased performance estimates, we harmoniously applied the repeated and nested stratified CV methods.</p><sec id="s2-6-3-1"><title>Repeated Stratified <italic>k</italic>-Fold CV (Outer Loop)</title><p>For a severely imbalanced dataset like ours, the distribution of classes across each fold is preserved through stratification. Accordingly, the repeated stratified CV method allows each fold to hold a consistent distribution among the 3 disease classes as a result of the repeated shuffling [<xref ref-type="bibr" rid="ref32">32</xref>]. Hence, to critically assess the average performance and stability of selected models, we applied the repeated stratified <italic>k</italic>-fold CV method by repeating our previous 5-fold CV loops 10 times to randomly repartition the training set into 5-folds across each repeat while maintaining the class distribution.</p></sec><sec id="s2-6-3-2"><title>Nested CV (Inner Loop)</title><p>The nested stratified CV is the other ML-based strategy we applied in this study, which is a dual validation strategy consisting of 2 main loops, an inner loop for hyperparameter tuning and an outer loop for validation, collaboratively operating to achieve unbiased performance estimates [<xref ref-type="bibr" rid="ref30">30</xref>,<xref ref-type="bibr" rid="ref33">33</xref>]. Accordingly, we implemented the nested CV that uses a stratified <italic>k</italic>-fold loop for the outer loop with 5 folds and another loop for the inner loop with 3 folds. The inner folds are sequentially used by each of the outer folds for selecting the best hyperparameters by running the grid search (GridSearchCV) method on the training portion of each of the outer folds. To ensure unbiased performance estimates, we implemented a procedure to evaluate the best model on the outer test fold, where all the scores on the outer test fold are aggregated.</p><p>Overall, in this strategy, we applied the stratification method to maintain the proportional representation of the instances of the minority class (podo) across all training and testing phases. Additionally, to reduce data partitioning bias and evaluate the stability of model prediction results, the complete nested process is repeated 10 times (the repeated stratified <italic>k</italic>-fold CV with <italic>k</italic>=10 is applied) using different random seeds. In this setting, the conditional class balancing method is applied inside the inner loop (inside the inner training folds).</p></sec></sec></sec><sec id="s2-7"><title>Model Development</title><sec id="s2-7-1"><title>Model Selection</title><p>This study uses datasets having 1495 instances with only 38 trainable features, where all the features are mostly linear, as shown by the principal component analysis (PCA) projection in <xref ref-type="fig" rid="figure1">Figure 1</xref>.</p><fig position="float" id="figure1"><label>Figure 1.</label><caption><p>The PCA projection analysis of the features of the datasets. PCA: principal component analysis.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ojphi_v18i1e84966_fig01.png"/></fig><p>As shown, the features of the 3 disease classes are highly linear, which shows the tendency of the features to be linearly separable. Hence, given the PCA projection analysis and the datasets to be used, 6 parameters are used to select the models. The parameters used include the model&#x2019;s ability in handling small-sized datasets, the nature of the data used (mostly linear), the robustness of the model in handling higher-level class imbalances, the robustness of the model in handling data with missing values, higher model performance expected, and model simplicity and interpretability for future deployment of the models on portable devices. Therefore, based on the extended analysis of size, complexity, and overall nature of our datasets, ML models, and the intent of the study, we selected 8 ML models: 3 nontree models, namely na&#x00EF;ve Bayes (NB), SVM (with linear kernel), and a simple CNN-based model using MLP; and 5 tree-based models, which include RF (ensemble tree-based model) and 4 boosted tree models of CatBoost, AdaBoost, LightGBM, and XGBoost.</p></sec><sec id="s2-7-2"><title>Model Evaluation Methods</title><p>For this study, we applied a multifaceted evaluation strategy, given the nature of our new dataset. First, the study entirely relies on the stratified <italic>k</italic>-fold CV method, as it allowed us to achieve more robust and lower-variance estimates. Hence, we applied a 5-fold (<italic>k</italic>=5) stratified <italic>k</italic>-fold CV method, as the hold-out method resulted in total overfitting during preliminary experimental trainings. For this study, using the standard accuracy might be misleading given the small-sized dataset used, the simplicity or complexity of models with respect to the dataset used, and the tendency of models to favor the majority class due to the extreme class imbalance [<xref ref-type="bibr" rid="ref34">34</xref>]. To address this, we used &#x201C;balanced accuracy,&#x201D; which is a metric used to reliably evaluate models by giving equal weights for all classes by averaging sensitivity (recall) and specificity (ie, rate of correctly predicted negative classes) of the models [<xref ref-type="bibr" rid="ref28">28</xref>,<xref ref-type="bibr" rid="ref29">29</xref>].</p><p>The study highly focuses on the <italic>F</italic><sub>1</sub>-score metric to balance between the precision and recall. The G-mean score, which represents the geometric mean of TPR (true positive rate or recall) and TNR (true negative rate or specificity) [<xref ref-type="bibr" rid="ref35">35</xref>], is used to evaluate the class-wise accuracy of the models on the 3 disease classes. We used the G-mean along with AUPRC (area under the precision-recall curve) as these metrics are not affected by class imbalances [<xref ref-type="bibr" rid="ref36">36</xref>], where the AUPRC metric is used to robustly assess the models&#x2019; predictive performance on the minority class. To evaluate the models&#x2019; class-wise separation performance, the ROC AUC metrics are computed based on the one-vs-one and one-vs-rest classification approaches. Apart from the macro average scores, the class-specific performance of each model is also evaluated to assess how well each model performs on each class, given the severe class imbalance. Hence, the class-specific precision (positive predictive value), recall (sensitivity), specificity (true negative rate), <italic>F</italic><sub>1</sub>-score, and NPV (negative predictive value) are used to individually assess the models on each class. Confusion matrices are heavily used to compute and visualize the class-specific scores for each model across all experimental settings. Further, we used the feature importance (FI) scores of each model to evaluate model performance and stability, monitor overfitting due to imbalance, include clinical interpretability, and identify noise and redundant features in the dataset. To apply the FI evaluation, we implemented 3 different but complementary techniques that include using built-in importance for the tree-based models, absolute coefficient magnitude for the linear model (SVM), and permutation importance for MLP and NB. Finally, based on the detailed evaluation of the models&#x2019; performance, we conducted deeper analysis using descriptive statistical metrics, which include computing average or mean (&#x03BC;) values, computing and evaluating numerical variations or differences (&#x0394;) of models&#x2019; performance, and analyzing performance stability of the models using SD or &#x03C3;, across the different experimental settings.</p></sec></sec></sec><sec id="s3" sec-type="results"><title>Results</title><p>For this study, we conducted the training experiments based on 4 logical experimental settings, as presented by the results in the following subsection.</p><sec id="s3-1"><title>Baseline Training</title><p>During the first training using IDS based on the 5-fold CV method, all the selected nontree models (NB, SVM, and MLP) and AdaBoost (tree-based model) failed to train due to the null values in the dataset and terminated the training with error messages. These results verified that these 3 nontree and AdaBoost models were highly sensitive to missing values [<xref ref-type="bibr" rid="ref37">37</xref>], due to the models&#x2019; inherent mathematical paradigms of reliance on the availability of complete data compared to the tree-based models [<xref ref-type="bibr" rid="ref38">38</xref>]. Accordingly, the 4 tree-based models&#x2014;RF, CatBoost, LightGBM, and XGBoost&#x2014;were able to classify the diseases with evaluation results, as presented in <xref ref-type="table" rid="table2">Table 2</xref>.</p><table-wrap id="t2" position="float"><label>Table 2.</label><caption><p>Overall model training results on the initial dataset using the <italic>k</italic>-fold cross-validation method.</p></caption><table id="table2" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom" colspan="2">Accuracy</td><td align="left" valign="bottom" colspan="7">Macro average</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Overall accuracy</td><td align="left" valign="bottom">Balanced accuracy</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">G-mean</td><td align="left" valign="bottom">ROC-AUC<sup><xref ref-type="table-fn" rid="table2fn1">a</xref></sup> (OVO<sup><xref ref-type="table-fn" rid="table2fn2">b</xref></sup>)</td><td align="left" valign="bottom">ROC-AUC (OVR<sup><xref ref-type="table-fn" rid="table2fn3">c</xref></sup>)</td><td align="left" valign="bottom">AUPRC<sup><xref ref-type="table-fn" rid="table2fn4">d</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">Random forest</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">CatBoost<sup><xref ref-type="table-fn" rid="table2fn5">e</xref></sup></td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">XGBoost<sup><xref ref-type="table-fn" rid="table2fn6">f</xref></sup></td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">LightGBM<sup><xref ref-type="table-fn" rid="table2fn7">g</xref></sup></td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr></tbody></table><table-wrap-foot><fn id="table2fn1"><p><sup>a</sup>ROC-AUC: area under the receiver operating characteristic curve.</p></fn><fn id="table2fn2"><p><sup>b</sup>OVO: one-vs-one.</p></fn><fn id="table2fn3"><p><sup>c</sup>OVR: one-vs-rest.</p></fn><fn id="table2fn4"><p><sup>d</sup>AUPRC: area under the precision-recall curve.</p></fn><fn id="table2fn5"><p><sup>e</sup>CatBoost: categorical boosting.</p></fn><fn id="table2fn6"><p><sup>f</sup>XGBoost: extreme gradient boosting.</p></fn><fn id="table2fn7"><p><sup>g</sup>LightGBM: light gradient boosting machine.</p></fn></table-wrap-foot></table-wrap><p>As shown in <xref ref-type="table" rid="table2">Table 2</xref>, all the 4 models achieved perfect scores (1.0) in accuracy and across all the macro average metrics of precision, recall, <italic>F</italic><sub>1</sub>-score, G-mean, and ROC-AUC. Class-wise, all these 4 models similarly scored perfect scores (1.0) across all metrics, as shown in <xref ref-type="table" rid="table3">Table 3</xref>.</p><table-wrap id="t3" position="float"><label>Table 3.</label><caption><p>Class-specific results of the 4 trainable models on IDS<sup><xref ref-type="table-fn" rid="table3fn1">a</xref></sup> using the <italic>k</italic>-fold cross-validation method.</p></caption><table id="table3" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Class</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">TNR<sup><xref ref-type="table-fn" rid="table3fn2">b</xref></sup></td><td align="left" valign="bottom">NPV<sup><xref ref-type="table-fn" rid="table3fn3">c</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">Podo</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">Scabies</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">Tungiasis</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr></tbody></table><table-wrap-foot><fn id="table3fn1"><p><sup>a</sup>IDS: initial dataset.</p></fn><fn id="table3fn2"><p><sup>b</sup>TNR: true negative rate or specificity.</p></fn><fn id="table3fn3"><p><sup>c</sup>NPV: negative predictive value.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-2"><title>Handling Structural Missingness</title><p><xref ref-type="table" rid="table4">Table 4</xref> presents overall model performance across the 2 experiments, demonstrating methods for handling missing values.</p><p>Starting from the second phase, all 8 models (including NB, SVM, MLP, and AdaBoost that were previously unable to train) were able to run properly as a result of replacing the missing values using the simple imputation method. However, as the null values are structural, an additional experiment demonstrating feature engineering techniques by creating missingness indicator flags was conducted to identify the suitable method for our dataset. Hence, during this experimental setting, we conducted two separate experiments demonstrating the methods we applied to handle structural missingness: (1) experiment A, which was conducted by applying the simple imputation method for the missing disease-specific values, and (2) experiment B, which applied a feature engineering technique and introduced missingness indicator flags for the missing disease-specific values. <xref ref-type="table" rid="table4">Table 4</xref> presents the overall results recorded in these 2 trainings. As shown in <xref ref-type="table" rid="table4">Table 4</xref>, with the use of the simple imputation method, 7 models showed perfect scores (1.0) across all evaluation metrics of balanced accuracy and macro average scores of precision, recall, <italic>F</italic><sub>1</sub>-score, G-mean, and AUPRC scores. Only MLP scored lower performance compared to the other models by achieving 0.90 in balanced accuracy and macro recall, and 0.77 in precision and <italic>F</italic><sub>1</sub>-score, while achieving 0.93 G-mean and approximately 1.0 (0.999) AUPRC scores.</p><table-wrap id="t4" position="float"><label>Table 4.</label><caption><p>Model performance results when trained using the <italic>k</italic>-fold cross-validation method.</p></caption><table id="table4" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom" colspan="6">Experiment A: simple imputing</td><td align="left" valign="bottom" colspan="6">Experiment B: use of missingness indicators</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Balanced accuracy</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">G-mean</td><td align="left" valign="bottom">AUPRC<sup><xref ref-type="table-fn" rid="table4fn1">a</xref></sup></td><td align="left" valign="bottom">Balanced accuracy</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">G-mean</td><td align="left" valign="bottom">AUPRC</td></tr></thead><tbody><tr><td align="left" valign="top">Na&#x00EF;ve Bayes</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">0.974</td><td align="left" valign="top">0.996</td><td align="left" valign="top">0.974</td><td align="left" valign="top">0.985</td><td align="left" valign="top">0.986</td><td align="left" valign="top">0.972</td></tr><tr><td align="left" valign="top">SVM<sup><xref ref-type="table-fn" rid="table4fn2">b</xref></sup> (linear)</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">MLP<sup><xref ref-type="table-fn" rid="table4fn3">c</xref></sup></td><td align="left" valign="top">0.90</td><td align="left" valign="top">0.77</td><td align="left" valign="top">0.90</td><td align="left" valign="top">0.77</td><td align="left" valign="top">0.93</td><td align="left" valign="top">0.999</td><td align="left" valign="top">0.997</td><td align="left" valign="top">0.956</td><td align="left" valign="top">0.997</td><td align="left" valign="top">0.974</td><td align="left" valign="top">0.997</td><td align="left" valign="top">0.999</td></tr><tr><td align="left" valign="top">Random forest</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">AdaBoost<sup><xref ref-type="table-fn" rid="table4fn4">d</xref></sup></td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">CatBoost<sup><xref ref-type="table-fn" rid="table4fn5">e</xref></sup></td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">XGBoost<sup><xref ref-type="table-fn" rid="table4fn6">f</xref></sup></td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">LightGBM<sup><xref ref-type="table-fn" rid="table4fn7">g</xref></sup></td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr></tbody></table><table-wrap-foot><fn id="table4fn1"><p><sup>a</sup>AUPRC: area under the precision-recall curve.</p></fn><fn id="table4fn2"><p><sup>b</sup>SVM: support vector machine.</p></fn><fn id="table4fn3"><p><sup>c</sup>MLP: multilayer perceptron.</p></fn><fn id="table4fn4"><p><sup>d</sup>AdaBoost: adaptive boosting.</p></fn><fn id="table4fn5"><p><sup>e</sup>CatBoost: categorical boosting.</p></fn><fn id="table4fn6"><p><sup>f</sup>XGBoost: extreme gradient boosting.</p></fn><fn id="table4fn7"><p><sup>g</sup>LightGBM: light gradient boosting machine.</p></fn></table-wrap-foot></table-wrap><p>Regarding class-specific performance, MLP scored the worst per-class precision of 0.31 for podo, while scoring 1.0 for scabies and tungiasis, as shown in <xref ref-type="table" rid="table5">Table 5</xref>. MLP also has the lowest <italic>F</italic><sub>1</sub>-score for podo (0.47), while scoring higher for tungiasis (0.83) and an almost perfect score (0.997) for scabies. However, MLP scored the highest recall and NPV scores for podo compared to scabies and tungiasis.</p><table-wrap id="t5" position="float"><label>Table 5.</label><caption><p>Model performance results when trained using the <italic>k</italic>-fold cross-validation method.</p></caption><table id="table5" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Class</td><td align="left" valign="bottom" colspan="5">NB<sup><xref ref-type="table-fn" rid="table5fn1">a</xref></sup>, SVM<sup><xref ref-type="table-fn" rid="table5fn2">b</xref></sup>, RF<sup><xref ref-type="table-fn" rid="table5fn3">c</xref></sup>, AdaBoost<sup><xref ref-type="table-fn" rid="table5fn4">d</xref></sup>, CatBoost<sup><xref ref-type="table-fn" rid="table5fn5">e</xref></sup>, XGBoost<sup><xref ref-type="table-fn" rid="table5fn6">f</xref></sup>, LightGBM<sup><xref ref-type="table-fn" rid="table5fn7">g</xref></sup></td><td align="left" valign="bottom" colspan="5">Per-class scores: MLP<sup><xref ref-type="table-fn" rid="table5fn8">h</xref></sup></td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Specificity</td><td align="left" valign="bottom">NPV<sup><xref ref-type="table-fn" rid="table5fn9">i</xref></sup></td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Specificity</td><td align="left" valign="bottom">NPV</td></tr></thead><tbody><tr><td align="left" valign="top">Podo</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.31</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.470</td><td align="char" char="." valign="top">0.899</td><td align="char" char="." valign="top">1.0</td></tr><tr><td align="left" valign="top">Scabies</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.995</td><td align="char" char="." valign="top">0.997</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.991</td></tr><tr><td align="left" valign="top">Tungiasis</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.705</td><td align="char" char="." valign="top">0.827</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.879</td></tr></tbody></table><table-wrap-foot><fn id="table5fn1"><p><sup>a</sup>NB: Na&#x00EF;ve Bayes.</p></fn><fn id="table5fn2"><p><sup>b</sup>SVM: support vector machine.</p></fn><fn id="table5fn3"><p><sup>c</sup>RF: random forest.</p></fn><fn id="table5fn4"><p><sup>d</sup>AdaBoost: adaptive boosting.</p></fn><fn id="table5fn5"><p><sup>e</sup>CatBoost: categorical boosting.</p></fn><fn id="table5fn6"><p><sup>f</sup>XGBoost: extreme gradient boosting.</p></fn><fn id="table5fn7"><p><sup>g</sup>LightGBM: light gradient boosting machine.</p></fn><fn id="table5fn8"><p><sup>h</sup>MLP: multilayer perceptron.</p></fn><fn id="table5fn9"><p><sup>i</sup>NPV: negative predictive value.</p></fn></table-wrap-foot></table-wrap><p>In experiment B, all models scored similar results as the previous simple imputation experiment, except NB and MLP, as shown in <xref ref-type="table" rid="table4">Table 4</xref>. With the use of the missingness indicators, the NB model achieved declining performance by achieving 0.974 in balanced accuracy, including the 0.97 and 0.985 macro recall and <italic>F</italic><sub>1</sub>-scores. Conversely, MLP scored significantly improved results compared to the previous experiment with simple imputation, while the remaining 6 models scored perfect scores across all metrics. Class-wise, as shown in <xref ref-type="table" rid="table6">Table 6</xref>, while MLP achieved improved per-class scores compared to the previous per-class scores (<xref ref-type="table" rid="table5">Table 5</xref>), NB scored declining performance in the class-specific scores of recall (podo: 0.92), <italic>F</italic><sub>1</sub>-score (podo: 0.96; tungiasis: 0.995), specificity (tungiasis: 0.995), and NPV (podo: 0.997).</p><table-wrap id="t6" position="float"><label>Table 6.</label><caption><p>Model performance results when trained using the <italic>k</italic>-fold cross-validation method.</p></caption><table id="table6" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Class</td><td align="left" valign="bottom" colspan="5">Na&#x00EF;ve Bayes</td><td align="left" valign="bottom" colspan="5">MLP<sup><xref ref-type="table-fn" rid="table6fn1">a</xref></sup></td></tr><tr><td align="left" valign="top"/><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Specificity</td><td align="left" valign="bottom">NPV<sup><xref ref-type="table-fn" rid="table6fn2">b</xref></sup></td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Specificity</td><td align="left" valign="bottom">NPV</td></tr></thead><tbody><tr><td align="left" valign="top">Podo</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.92</td><td align="char" char="." valign="top">0.96</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.997</td><td align="char" char="." valign="top">0.87</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.93</td><td align="char" char="." valign="top">0.993</td><td align="char" char="." valign="top">1.0</td></tr><tr><td align="left" valign="top">Scabies</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.990</td><td align="char" char="." valign="top">0.995</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.982</td></tr><tr><td align="left" valign="top">Tungiasis</td><td align="char" char="." valign="top">0.989</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.995</td><td align="char" char="." valign="top">0.995</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td></tr></tbody></table><table-wrap-foot><fn id="table6fn1"><p><sup>a</sup>MLP: multilayer perceptron.</p></fn><fn id="table6fn2"><p><sup>b</sup>NPV: negative predictive value.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-3"><title>Class Imbalance Handling</title><p>During the third experimental setting, demonstrating the class balancing method, 7 models (including NB) showed stable performance compared to the previous experiment with a missingness indicator. MLP, however, achieved improved results with perfect scores across all metrics, as shown in <xref ref-type="table" rid="table7">Table 7</xref>.</p><table-wrap id="t7" position="float"><label>Table 7.</label><caption><p>Model performance results when trained using the <italic>k</italic>-fold cross-validation method.</p></caption><table id="table7" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom" colspan="2">Accuracy</td><td align="left" valign="bottom" colspan="7">Macro average</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Overall accuracy</td><td align="left" valign="bottom">Balanced accuracy</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">G-mean</td><td align="left" valign="bottom">ROC-AUC<sup><xref ref-type="table-fn" rid="table7fn1">a</xref></sup> (OVO<sup><xref ref-type="table-fn" rid="table7fn2">b</xref></sup>)</td><td align="left" valign="bottom">ROC-AUC (OVR<sup><xref ref-type="table-fn" rid="table7fn3">c</xref></sup>)</td><td align="left" valign="bottom">AUPRC<sup><xref ref-type="table-fn" rid="table7fn4">d</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">Na&#x00EF;ve Bayes</td><td align="char" char="." valign="top">0.997</td><td align="char" char="." valign="top">0.974</td><td align="char" char="." valign="top">0.996</td><td align="char" char="." valign="top">0.974</td><td align="char" char="." valign="top">0.985</td><td align="char" char="." valign="top">0.986</td><td align="char" char="." valign="top">0.981</td><td align="char" char="." valign="top">0.986</td><td align="char" char="." valign="top">0.972</td></tr><tr><td align="left" valign="top">SVM<sup><xref ref-type="table-fn" rid="table7fn5">e</xref></sup> (linear)</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td></tr><tr><td align="left" valign="top">MLP<sup><xref ref-type="table-fn" rid="table7fn6">f</xref></sup></td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td></tr><tr><td align="left" valign="top">Random forest</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td></tr><tr><td align="left" valign="top">AdaBoost<sup><xref ref-type="table-fn" rid="table7fn7">g</xref></sup></td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td></tr><tr><td align="left" valign="top">CatBoost<sup><xref ref-type="table-fn" rid="table7fn8">h</xref></sup></td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td></tr><tr><td align="left" valign="top">XGBoost<sup><xref ref-type="table-fn" rid="table7fn9">i</xref></sup></td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td></tr><tr><td align="left" valign="top">LightGBM<sup><xref ref-type="table-fn" rid="table7fn10">j</xref></sup></td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td></tr></tbody></table><table-wrap-foot><fn id="table7fn1"><p><sup>a</sup>ROC-AUC: area under the receiver operating characteristic curve.</p></fn><fn id="table7fn2"><p><sup>b</sup>OVO: one-vs-one. </p></fn><fn id="table7fn3"><p><sup>c</sup>OVR: one-vs-rest. </p></fn><fn id="table7fn4"><p><sup>d</sup>AUPRC: area under the precision-recall curve.</p></fn><fn id="table7fn5"><p><sup>e</sup>SVM: support vector machine.</p></fn><fn id="table7fn6"><p><sup>f</sup>MLP: multilayer perceptron.</p></fn><fn id="table7fn7"><p><sup>g</sup>AdaBoost: adaptive boosting.</p></fn><fn id="table7fn8"><p><sup>h</sup>CatBoost: categorical boosting.</p></fn><fn id="table7fn9"><p><sup>i</sup>XGBoost: extreme gradient boosting.</p></fn><fn id="table7fn10"><p><sup>j</sup>LightGBM: light gradient boosting machine.</p></fn></table-wrap-foot></table-wrap><p>In terms of per-class scores, all 7 models (except NB) scored 1.0 across all 5 per-class metrics, as shown in <xref ref-type="table" rid="table8">Table 8</xref>. NB showed lower per-class scores compared to the other 7 models.</p><table-wrap id="t8" position="float"><label>Table 8.</label><caption><p>Model performance results when trained using the <italic>k</italic>-fold cross-validation method.</p></caption><table id="table8" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Class</td><td align="left" valign="bottom" colspan="5">SVM<sup><xref ref-type="table-fn" rid="table8fn1">a</xref></sup>, RF<sup><xref ref-type="table-fn" rid="table8fn2">b</xref></sup>, AdaBoost<sup><xref ref-type="table-fn" rid="table8fn3">c</xref></sup>, CatBoost<sup><xref ref-type="table-fn" rid="table8fn4">d</xref></sup>, XGBoost<sup><xref ref-type="table-fn" rid="table8fn5">e</xref></sup>, LightGBM<sup><xref ref-type="table-fn" rid="table8fn6">f</xref></sup>, MLP<sup><xref ref-type="table-fn" rid="table8fn7">g</xref></sup></td><td align="left" valign="bottom" colspan="5">Na&#x00EF;ve Bayes</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Specificity</td><td align="left" valign="bottom">NPV<sup><xref ref-type="table-fn" rid="table8fn8">h</xref></sup></td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Specificity</td><td align="left" valign="bottom">NPV</td></tr></thead><tbody><tr><td align="left" valign="top">Podo</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.92</td><td align="char" char="." valign="top">0.96</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.997</td></tr><tr><td align="left" valign="top">Scabies</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td></tr><tr><td align="left" valign="top">Tungiasis</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.989</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.995</td><td align="char" char="." valign="top">0.995</td><td align="char" char="." valign="top">1.0</td></tr></tbody></table><table-wrap-foot><fn id="table8fn1"><p><sup>a</sup>SVM: support vector machine.</p></fn><fn id="table8fn2"><p><sup>b</sup>RF: random forest.</p></fn><fn id="table8fn3"><p><sup>c</sup>AdaBoost: adaptive boosting.</p></fn><fn id="table8fn4"><p><sup>d</sup>CatBoost: categorical boosting.</p></fn><fn id="table8fn5"><p><sup>e</sup>XGBoost: extreme gradient boosting.</p></fn><fn id="table8fn6"><p><sup>f</sup>LightGBM: light gradient boosting machine.</p></fn><fn id="table8fn7"><p><sup>g</sup>MLP: multilayer perceptron.</p></fn><fn id="table8fn8"><p><sup>h</sup>NPV: negative predictive value.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-4"><title>The Harmonized Approach</title><p>In this last experimental setting, we applied our proposed harmonized approach that integrates data balancing and robust validation methods, which presents 2 separate evaluation results: first, the performance scored on the test set, and second, the results scored on the nested CV method that incorporates the hyperparameter tuning process. The results in <xref ref-type="table" rid="table9">Table 9</xref> present the overall test performance of the models of the final experiment using the hybrid approach. As shown in <xref ref-type="table" rid="table9">Table 9</xref>, NB and MLP scored similar performance scores of 0.974 in balanced accuracy, with the macro average scores of 0.996, 0.974, 0.985, and 0.986 in precision, recall, <italic>F</italic><sub>1</sub>-score, and G-mean, respectively. MLP scored higher results in ROC-AUC and AUPRC, while the remaining 6 models scored perfect scores (1.0) across all evaluation metrics.</p><table-wrap id="t9" position="float"><label>Table 9.</label><caption><p>Model performance results when trained using the <italic>k</italic>-fold cross-validation method.</p></caption><table id="table9" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom" colspan="2">Accuracy</td><td align="left" valign="bottom" colspan="7">Macro average</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Overall accuracy</td><td align="left" valign="bottom">Balanced accuracy</td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">G-mean</td><td align="left" valign="bottom">ROC-AUC<sup><xref ref-type="table-fn" rid="table9fn1">a</xref></sup> (OVO<sup><xref ref-type="table-fn" rid="table9fn2">b</xref></sup>)</td><td align="left" valign="bottom">ROC-AUC (OVR<sup><xref ref-type="table-fn" rid="table9fn3">c</xref></sup>)</td><td align="left" valign="bottom">AUPRC<sup><xref ref-type="table-fn" rid="table9fn4">d</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">Na&#x00EF;ve Bayes</td><td align="left" valign="top">0.997</td><td align="left" valign="top">0.974</td><td align="left" valign="top">0.996</td><td align="left" valign="top">0.974</td><td align="left" valign="top">0.985</td><td align="left" valign="top">0.986</td><td align="left" valign="top">0.981</td><td align="left" valign="top">0.986</td><td align="left" valign="top">0.972</td></tr><tr><td align="left" valign="top">SVM<sup><xref ref-type="table-fn" rid="table9fn5">e</xref></sup> (linear)</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">MLP<sup><xref ref-type="table-fn" rid="table9fn6">f</xref></sup></td><td align="left" valign="top">0.997</td><td align="left" valign="top">0.974</td><td align="left" valign="top">0.996</td><td align="left" valign="top">0.974</td><td align="left" valign="top">0.985</td><td align="left" valign="top">0.986</td><td align="left" valign="top">0.997</td><td align="left" valign="top">1.0</td><td align="left" valign="top">0.999</td></tr><tr><td align="left" valign="top">Random forest</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">AdaBoost<sup><xref ref-type="table-fn" rid="table9fn7">g</xref></sup></td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">CatBoost<sup><xref ref-type="table-fn" rid="table9fn8">h</xref></sup></td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">XGBoost<sup><xref ref-type="table-fn" rid="table9fn9">i</xref></sup></td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr><tr><td align="left" valign="top">LightGBM<sup><xref ref-type="table-fn" rid="table9fn10">j</xref></sup></td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td><td align="left" valign="top">1.0</td></tr></tbody></table><table-wrap-foot><fn id="table9fn1"><p><sup>a</sup>ROC-AUC: area under the receiver operating characteristic curve.</p></fn><fn id="table9fn2"><p><sup>b</sup>OVO: one-vs-one. </p></fn><fn id="table9fn3"><p><sup>c</sup>OVR: one-vs-rest. </p></fn><fn id="table9fn4"><p><sup>d</sup>AUPRC: area under the precision-recall curve.</p></fn><fn id="table9fn5"><p><sup>e</sup>SVM: support vector machine.</p></fn><fn id="table9fn6"><p><sup>f</sup>MLP: multilayer perceptron.</p></fn><fn id="table9fn7"><p><sup>g</sup>AdaBoost: adaptive boosting.</p></fn><fn id="table9fn8"><p><sup>h</sup>CatBoost: categorical boosting.</p></fn><fn id="table9fn9"><p><sup>i</sup>XGBoost: extreme gradient boosting.</p></fn><fn id="table9fn10"><p><sup>j</sup>LightGBM: light gradient boosting machine.</p></fn></table-wrap-foot></table-wrap><p>Class-wise, 6 models (SVM, RF, AdaBoost, CatBoost, XGBoost, and LightGBM) achieved perfect scores (1.0) across all per-class metrics of precision, recall, <italic>F</italic><sub>1</sub>-score, specificity, and NPV, as presented in <xref ref-type="table" rid="table10">Table 10</xref>. Conversely, the MLP and NB models scored exactly the same per-class metrics, scoring the least sensitivity for podoconiosis with a disease-specific recall of 0.92, while scoring almost perfect scores in precision (0.99), specificity (0.995), and <italic>F</italic><sub>1</sub>-score (0.995) for the tungiasis class.</p><table-wrap id="t10" position="float"><label>Table 10.</label><caption><p>Model performance results when trained using the <italic>k</italic>-fold cross-validation method.</p></caption><table id="table10" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Class</td><td align="left" valign="bottom" colspan="5">Models: SVM<sup><xref ref-type="table-fn" rid="table10fn1">a</xref></sup>, RF<sup><xref ref-type="table-fn" rid="table10fn2">b</xref></sup>, AdaBoost<sup><xref ref-type="table-fn" rid="table10fn3">c</xref></sup>, CatBoost<sup><xref ref-type="table-fn" rid="table10fn4">d</xref></sup>, XGBoost<sup><xref ref-type="table-fn" rid="table10fn5">e</xref></sup>, and LightGBM<sup><xref ref-type="table-fn" rid="table10fn6">f</xref></sup></td><td align="left" valign="bottom" colspan="5">Models: NB<sup><xref ref-type="table-fn" rid="table10fn7">g</xref></sup> and MLP<sup><xref ref-type="table-fn" rid="table10fn8">h</xref></sup></td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Specificity</td><td align="left" valign="bottom">NPV<sup><xref ref-type="table-fn" rid="table10fn9">i</xref></sup></td><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">Specificity</td><td align="left" valign="bottom">NPV</td></tr></thead><tbody><tr><td align="left" valign="top">Podo</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.923</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.997</td></tr><tr><td align="left" valign="top">Scabies</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td></tr><tr><td align="left" valign="top">Tungiasis</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.989</td><td align="char" char="." valign="top">1.0</td><td align="char" char="." valign="top">0.995</td><td align="char" char="." valign="top">0.995</td><td align="char" char="." valign="top">1.0</td></tr></tbody></table><table-wrap-foot><fn id="table10fn1"><p><sup>a</sup>SVM: support vector machine.</p></fn><fn id="table10fn2"><p><sup>b</sup>RF: random forest.</p></fn><fn id="table10fn3"><p><sup>c</sup>AdaBoost: adaptive boosting.</p></fn><fn id="table10fn4"><p><sup>d</sup>CatBoost: categorical boosting.</p></fn><fn id="table10fn5"><p><sup>e</sup>XGBoost: extreme gradient boosting.</p></fn><fn id="table10fn6"><p><sup>f</sup>LightGBM: light gradient boosting machine.</p></fn><fn id="table10fn7"><p><sup>g</sup>NB: Na&#x00EF;ve Bayes.</p></fn><fn id="table10fn8"><p><sup>h</sup>MLP: multilayer perceptron.</p></fn><fn id="table10fn9"><p><sup>i</sup>NPV: negative predictive value.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-5"><title>Nested CV and Hyperparameter Tuning</title><p>Nested CV, the method we applied for hyperparameter tuning and to achieve unbiased performance estimates, resulted in slight deviations compared to the test performance, as summarized in <xref ref-type="table" rid="table11">Table 11</xref>.</p><p>As shown, 5 models scored perfect scores across all metrics with no performance variability (&#x03C3;=0) on the outer folds. However, 2 models (LightGBM and XGBoost) exhibited slight drops in their average scores of balanced accuracy and macro recall (&#x0394;&#x2212;0.007, &#x03C3;=&#x2212;0.013), and <italic>F</italic><sub>1</sub>-score (&#x0394; 0.004, &#x03C3;=&#x2212;0.007), while the other scores remain stable.</p><table-wrap id="t11" position="float"><label>Table 11.</label><caption><p>Summary of models&#x2019; performance on the nested cross-validation method, with mean (SD) of key performance metrics.</p></caption><table id="table11" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Balanced accuracy</td><td align="left" valign="bottom">Recall (macro)</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score (macro)</td><td align="left" valign="bottom">Average precision</td><td align="left" valign="bottom">ROC-AUC<sup><xref ref-type="table-fn" rid="table11fn1">a</xref></sup> (OVO<sup><xref ref-type="table-fn" rid="table11fn2">b</xref></sup>)</td></tr></thead><tbody><tr><td align="left" valign="top">Na&#x00EF;ve Bayes</td><td align="left" valign="top">0.999 (0.003)</td><td align="left" valign="top">0.999 (0.003)</td><td align="left" valign="top">0.999 (0.003)</td><td align="left" valign="top">0.998 (0.004)</td><td align="left" valign="top">0.999 (0.002)</td></tr><tr><td align="left" valign="top">SVM<sup><xref ref-type="table-fn" rid="table11fn3">c</xref></sup> (linear)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td></tr><tr><td align="left" valign="top">MLP<sup><xref ref-type="table-fn" rid="table11fn4">d</xref></sup></td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td></tr><tr><td align="left" valign="top">Random forest</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td></tr><tr><td align="left" valign="top">AdaBoost<sup><xref ref-type="table-fn" rid="table11fn5">e</xref></sup></td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td></tr><tr><td align="left" valign="top">CatBoost<sup><xref ref-type="table-fn" rid="table11fn6">f</xref></sup></td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td></tr><tr><td align="left" valign="top">XGBoost<sup><xref ref-type="table-fn" rid="table11fn7">g</xref></sup></td><td align="left" valign="top">0.993 (0.013)</td><td align="left" valign="top">0.993 (0.013)</td><td align="left" valign="top">0.996 (0.007)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td></tr><tr><td align="left" valign="top">LightGBM<sup><xref ref-type="table-fn" rid="table11fn8">h</xref></sup></td><td align="left" valign="top">0.993 (0.013)</td><td align="left" valign="top">0.993 (0.013)</td><td align="left" valign="top">0.996 (0.008)</td><td align="left" valign="top">1.0 (0.0)</td><td align="left" valign="top">1.0 (0.0)</td></tr></tbody></table><table-wrap-foot><fn id="table11fn1"><p><sup>a</sup>ROC-AUC: area under the receiver operating characteristic curve.</p></fn><fn id="table11fn2"><p><sup>b</sup>OVO: one-vs-one.</p></fn><fn id="table11fn3"><p><sup>c</sup>SVM: support vector machine.</p></fn><fn id="table11fn4"><p><sup>d</sup>MLP: multilayer perceptron.</p></fn><fn id="table11fn5"><p><sup>e</sup>AdaBoost: adaptive boosting.</p></fn><fn id="table11fn6"><p><sup>f</sup>CatBoost: categorical boosting.</p></fn><fn id="table11fn7"><p><sup>g</sup>XGBoost: extreme gradient boosting.</p></fn><fn id="table11fn8"><p><sup>h</sup>LightGBM: light gradient boosting machine.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-6"><title>Analysis of Results</title><sec id="s3-6-1"><title>Overall Performance and Model Stability</title><p>The second experiment that applied the simple imputation method for the structural missing values introduced the first performance variation, allowing the 4 models (NB, SVM, MLP, and AdaBoost) to have full performance scores. The third experiment brought the next variations, creating 3 divergent performance score groups after the application of disease-specific symptom applicability indicator flags. This experiment resulted in substantial performance changes, particularly to MLP, as evidenced by the model&#x2019;s absolute improvements of &#x0394;+0.097 in balanced accuracy and macro recall, &#x0394;+0.560 (&#x03C3;=0.396) in podo precision, &#x0394;+0.460 (&#x03C3;=0.325) in podo <italic>F</italic><sub>1</sub>-score, &#x0394;+0.295 (&#x03C3;=0.209) in tungiasis precision, and &#x0394;+0.173 (&#x03C3;=0.122) in tungiasis <italic>F</italic><sub>1</sub>-score. According to the results, all improvements, particularly the class-specific scores, were achieved at a slight expense of the scabies class recall &#x0394;&#x2212;0.005 (&#x03C3;=&#x2212;0.004) and <italic>F</italic><sub>1</sub>-score &#x0394;&#x2212;0.002 (&#x03C3;=&#x2212;0.001). All these improved scores of MLP underscore that the feature engineering technique was significantly effective in helping the model identify the minority classes. The confusion matrices in <xref ref-type="fig" rid="figure2">Figures 2A and 2B</xref> depict all these changes shown by MLP across the 2 experiments. The other 6 models showed all perfect per-class scores with zero misclassification.</p><p>Unlike MLP, the generative model NB exhibited overall performance declines of &#x0394;&#x2212;0.026 in balanced accuracy and macro recall, &#x0394;&#x2212;0.015 in macro <italic>F</italic><sub>1</sub>-score, including precision (&#x0394;&#x2212;0.004) and G-mean (&#x0394;&#x2212;0.014), including the per-class scores. The remaining 6 models maintained the perfect scores across all metrics. The sample weighting method further improved the performance of MLP with an absolute increase of &#x0394;+0.003 in balanced accuracy, recall, and G-mean, &#x0394;+0.026 in macro <italic>F</italic><sub>1</sub>-score, and &#x0394;+0.044 in precision, while the remaining 7 models remained consistent.</p><p>Overall, across all the experimental phases, 4 models (SVM, RF, CatBoost, and AdaBoost) showed absolute stability of performance, with no performance variation (&#x03C3;=0.0), while the other 4 models (NB, MLP, XGBoost, and LightGBM) exhibited slight performance variations. <xref ref-type="table" rid="table12">Table 12</xref> presents an analysis of models&#x2019; performance variance across the 6 major evaluation metrics throughout the 5 experiments. As shown, MLP has the highest overall performance variance in its balanced accuracy and macro recall (&#x03C3;=0.043), macro <italic>F</italic><sub>1</sub>-score and precision (&#x03C3;=0.099), including G-mean (&#x03C3;=0.03) and macro AUPRC (&#x03C3;=0.001). Conversely, NB exhibited dropping patterns in balanced accuracy and macro recall (&#x03C3;=&#x2212;0.014), macro <italic>F</italic><sub>1</sub>-score (&#x03C3;=&#x2212;0.008), and precision, G-mean, and AUPRC (&#x03C3;=&#x2212;0.002, &#x2212;0.007, and &#x2212;0.015, respectively). These overall declines were introduced due to the introduction of symptom applicability indicator flags, which created a set of highly correlated features, leading to incompatibility with the model&#x2019;s probabilistic assumption [<xref ref-type="bibr" rid="ref39">39</xref>].</p><p>According to the results of the nested CV method (<xref ref-type="table" rid="table12">Table 12</xref>), LightGBM and XGBoost exceptionally exhibited marginal performance drops. These 2 models dropped their performance by &#x0394;&#x2212;0.007 (&#x03C3;=&#x2212;0.003) in balanced accuracy and macro recall, &#x0394;&#x2212;0.004 (&#x03C3;=&#x2212;0.002) in <italic>F</italic><sub>1</sub>-score and G-mean, while showing no change in precision and AUPRC.</p><fig position="float" id="figure2"><label>Figure 2.</label><caption><p>Confusion matrices of MLP showing model performance after the second and third experiments: (A) confusion matrix of MLP after the second experiment with simple imputation and (B) confusion matrix of MLP after the third experiment with feature engineering. MLP: multilayer perceptron.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ojphi_v18i1e84966_fig02.png"/></fig><table-wrap id="t12" position="float"><label>Table 12.</label><caption><p>Analysis of models&#x2019; performance based on 6 evaluation metrics across 5 experiments.</p></caption><table id="table12" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Balanced accuracy</td><td align="left" valign="bottom" colspan="5">Macro average</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="bottom"/><td align="left" valign="bottom">Precision</td><td align="left" valign="bottom">Recall</td><td align="left" valign="bottom"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="bottom">G-mean</td><td align="left" valign="bottom">AUPRC<sup><xref ref-type="table-fn" rid="table12fn1">a</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top">Na&#x00EF;ve Bayes</td><td align="char" char="." valign="top">0.014</td><td align="char" char="." valign="top">0.002</td><td align="char" char="." valign="top">0.014</td><td align="char" char="." valign="top">0.008</td><td align="char" char="." valign="top">0.007</td><td align="char" char="." valign="top">0.015</td></tr><tr><td align="left" valign="top">SVM<sup><xref ref-type="table-fn" rid="table12fn2">b</xref></sup> (linear)</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td></tr><tr><td align="left" valign="top">MLP<sup><xref ref-type="table-fn" rid="table12fn3">c</xref></sup></td><td align="char" char="." valign="top">0.043</td><td align="char" char="." valign="top">0.099</td><td align="char" char="." valign="top">0.043</td><td align="char" char="." valign="top">0.099</td><td align="char" char="." valign="top">0.03</td><td align="char" char="." valign="top">0.001</td></tr><tr><td align="left" valign="top">Random forest</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td></tr><tr><td align="left" valign="top">AdaBoost<sup><xref ref-type="table-fn" rid="table12fn4">d</xref></sup></td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td></tr><tr><td align="left" valign="top">CatBoost<sup><xref ref-type="table-fn" rid="table12fn5">e</xref></sup></td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td></tr><tr><td align="left" valign="top">XGBoost<sup><xref ref-type="table-fn" rid="table12fn6">f</xref></sup></td><td align="char" char="." valign="top">0.003</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.003</td><td align="char" char="." valign="top">0.002</td><td align="char" char="." valign="top">0.002</td><td align="char" char="." valign="top">0.0</td></tr><tr><td align="left" valign="top">LightGBM<sup><xref ref-type="table-fn" rid="table12fn7">g</xref></sup></td><td align="char" char="." valign="top">0.003</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.003</td><td align="char" char="." valign="top">0.002</td><td align="char" char="." valign="top">0.002</td><td align="char" char="." valign="top">0.0</td></tr></tbody></table><table-wrap-foot><fn id="table12fn1"><p><sup>a</sup>AUPRC: area under the precision-recall curve.</p></fn><fn id="table12fn2"><p><sup>b</sup>SVM: support vector machine.</p></fn><fn id="table12fn3"><p><sup>c</sup>MLP: multilayer perceptron.</p></fn><fn id="table12fn4"><p><sup>d</sup>AdaBoost: adaptive boosting.</p></fn><fn id="table12fn5"><p><sup>e</sup>CatBoost: categorical boosting.</p></fn><fn id="table12fn6"><p><sup>f</sup>XGBoost: extreme gradient boosting.</p></fn><fn id="table12fn7"><p><sup>g</sup>LightGBM: light gradient boosting machine.</p></fn></table-wrap-foot></table-wrap><p>Similar patterns were seen in the class-wise performance of these 4 models, as shown in <xref ref-type="table" rid="table13">Table 13</xref>.</p><p>MLP has the highest per-class performance variance with &#x03C3;=0.33, &#x03C3;=0.26, &#x03C3;=0.49, and &#x03C3;=0.39 in podo-precision, <italic>F</italic><sub>1</sub>-score, specificity, and recall, respectively. These higher variations are caused by the lowest per-class scores achieved during the third experimental training that demonstrates the structural missing values using missingness indicator features. NB exhibited smaller variance in tungiasis precision (&#x03C3;=&#x2212;0.006), podo recall (&#x03C3;=&#x2212;0.04), podo-<italic>F</italic><sub>1</sub>-score (&#x03C3;=&#x2212;0.023), tungiasis-<italic>F</italic><sub>1</sub>-score, and specificity (&#x03C3;=&#x2212;0.003). However, the 6 models maintained a totally stable per-class performance (&#x03C3;=0).</p><table-wrap id="t13" position="float"><label>Table 13.</label><caption><p>Summary of models&#x2019; performance on the nested cross-validation method, with mean (SD) of key performance metrics.</p></caption><table id="table13" frame="hsides" rules="groups"><thead><tr><td align="left" valign="top">Model and disease class</td><td align="left" valign="top" colspan="5">Per-class metrics</td></tr><tr><td align="left" valign="bottom"/><td align="left" valign="top">Precision</td><td align="left" valign="top">Recall</td><td align="left" valign="top"><italic>F</italic><sub>1</sub>-score</td><td align="left" valign="top">Specificity</td><td align="left" valign="top">NPV<sup><xref ref-type="table-fn" rid="table13fn1">a</xref></sup></td></tr></thead><tbody><tr><td align="left" valign="top" colspan="6">SVM<sup><xref ref-type="table-fn" rid="table13fn2">b</xref></sup>, RF<sup><xref ref-type="table-fn" rid="table13fn3">c</xref></sup>, AdaBoost<sup><xref ref-type="table-fn" rid="table13fn4">d</xref></sup>, CatBoost<sup><xref ref-type="table-fn" rid="table13fn5">e</xref></sup>, XGBoost<sup><xref ref-type="table-fn" rid="table13fn6">f</xref></sup>, LightGBM<sup><xref ref-type="table-fn" rid="table13fn7">g</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Podo</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Scabies</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Tungiasis</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td></tr><tr><td align="left" valign="top" colspan="6">Na&#x00EF;ve Bayes</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Podo</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.040</td><td align="char" char="." valign="top">0.023</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.002</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Scabies</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.0</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Tungiasis</td><td align="char" char="." valign="top">0.006</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.003</td><td align="char" char="." valign="top">0.003</td><td align="char" char="." valign="top">0.0</td></tr><tr><td align="left" valign="top" colspan="6">MLP<sup><xref ref-type="table-fn" rid="table13fn8">h</xref></sup></td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Podo</td><td align="char" char="." valign="top">0.329</td><td align="char" char="." valign="top">0.039</td><td align="char" char="." valign="top">0.255</td><td align="char" char="." valign="top">0.049</td><td align="char" char="." valign="top">0.002</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Scabies</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.005</td><td align="char" char="." valign="top">0.002</td><td align="char" char="." valign="top">0.0</td><td align="char" char="." valign="top">0.009</td></tr><tr><td align="left" valign="top"><named-content content-type="indent">&#x00A0;&#x00A0;&#x00A0;&#x00A0;</named-content>Tungiasis</td><td align="char" char="." valign="top">0.006</td><td align="char" char="." valign="top">0.147</td><td align="char" char="." valign="top">0.086</td><td align="char" char="." valign="top">0.003</td><td align="char" char="." valign="top">0.061</td></tr></tbody></table><table-wrap-foot><fn id="table13fn1"><p><sup>a</sup>NPV: negative predictive value.</p></fn><fn id="table13fn2"><p><sup>b</sup>SVM: support vector machine.</p></fn><fn id="table13fn3"><p><sup>c</sup>RF: random forest.</p></fn><fn id="table13fn4"><p><sup>d</sup>AdaBoost: adaptive boosting.</p></fn><fn id="table13fn5"><p><sup>e</sup>CatBoost: categorical boosting.</p></fn><fn id="table13fn6"><p><sup>f</sup>XGBoost: extreme gradient boosting.</p></fn><fn id="table13fn7"><p><sup>g</sup>LightGBM: light gradient boosting machine.</p></fn><fn id="table13fn8"><p><sup>h</sup>MLP: multilayer perceptron.</p></fn></table-wrap-foot></table-wrap></sec><sec id="s3-6-2"><title>Analysis of Feature Importance</title><p>While the complete FI scores of the 38 features across 8 models are presented in Table S3 of <xref ref-type="supplementary-material" rid="app2">Multimedia Appendix 2</xref>, in this section, we present an overall summary of the FI scores. Accordingly, the FI evaluation results revealed that 4 models (NB, SVM, MLP, and RF) used 37 features out of the 38 trainable features, as shown in <xref ref-type="fig" rid="figure3">Figure 3A</xref>.</p><p>Regarding individual features used, <xref ref-type="fig" rid="figure3">Figure 3B</xref> shows the top 10 features according to their use by the 8 models across all experiments. Accordingly, the jigger infestation classification feature (JI_classification(#jig-inf)) appears to be the top feature having the highest mean FI value. This feature appears in 6 models in their top 5 models list. Model-wise, we conducted deeper analyses of the feature relevance evaluation scores, presented in <xref ref-type="table" rid="table14">Table 14</xref>, that reveal a diverse set of feature distribution, utilization, and selectivity patterns among models.</p><fig position="float" id="figure3"><label>Figure 3.</label><caption><p>Feature importance evaluation results of the MLP model: (A) overall analysis of the number of features used by each model and (B) top features based on the average importance scores across models. MLP: multilayer perceptron.</p></caption><graphic alt-version="no" mimetype="image" position="float" xlink:type="simple" xlink:href="ojphi_v18i1e84966_fig03.png"/></fig><p>As shown, all the 5 tree-based models exhibited a right-skewed FI distribution, which signifies that smaller subsets of the entire feature set are highly relevant, with multiple remaining features being least or nonrelevant [<xref ref-type="bibr" rid="ref40">40</xref>]. This demonstrated that these 5 tree-based models are highly selective in using features, as evidenced by the feature preference and sparsity ratio scores. Conversely, NB and MLP exhibited a left-skewed FI distribution pattern, highlighting that these models equally designated the majority of the features as highly relevant&#x2014;MLP assigned the same FI score of 0.665 for 27 features, whereas NB assigned a uniform 1.0 FI score for 36 features. These models have higher nondiscriminative tendencies to identify and ignore less relevant features, as shown by the higher feature preference and lower sparsity ratio scores (<xref ref-type="table" rid="table14">Table 14</xref>). Finally, SVM is the only model that demonstrated an approximately balanced (symmetric) FI score distribution, as shown by its lower sparsity and information decay scores. Overall, the model-specific FI distributions displayed 3 different distribution trends, which is confirmed by the aggregate summary of FI distribution results (mean 0.311, SD 0.390)&#x2014;verifying that the overall distribution of the skin NTD diagnostic features is right-skewed. Given the higher mean (&#x03BC;=0.311) having an SD that exceeds the mean FI score (&#x03C3;=0.390), which creates higher aggregate variance factors, the right-skewed distribution is naturally expected to be reflected [<xref ref-type="bibr" rid="ref40">40</xref>].</p><table-wrap id="t14" position="float"><label>Table 14.</label><caption><p>Summary of feature importance scores and overall feature usage analysis of the 8 models.</p></caption><table id="table14" frame="hsides" rules="groups"><thead><tr><td align="left" valign="bottom">Model</td><td align="left" valign="bottom">Diagnostic features (n=21)</td><td align="left" valign="bottom">Demographic features (n=17)</td><td align="left" valign="bottom">Distribution of FI<sup><xref ref-type="table-fn" rid="table14fn1">a</xref></sup> score</td><td align="left" valign="bottom">FI distribution type</td><td align="left" valign="bottom">Feature preference</td><td align="left" valign="bottom">Sparsity ratio</td><td align="left" valign="bottom">Information decay</td><td align="left" valign="bottom">Information decay type</td></tr></thead><tbody><tr><td align="left" valign="top">Na&#x00EF;ve Bayes</td><td align="left" valign="top">21</td><td align="left" valign="top">16</td><td align="left" valign="top">Mean 0.974, SD 0.160, range 0-1</td><td align="left" valign="top">Left-skewed</td><td align="left" valign="top">0.9737</td><td align="left" valign="top">0.0263</td><td align="left" valign="top">&#x2212;0.004</td><td align="left" valign="top">Slow decay</td></tr><tr><td align="left" valign="top">SVM<sup><xref ref-type="table-fn" rid="table14fn2">b</xref></sup> (linear)</td><td align="left" valign="top">21</td><td align="left" valign="top">16</td><td align="left" valign="top">Mean 0.310, SD 0.334, range 0-1</td><td align="left" valign="top">Approximately symmetric</td><td align="left" valign="top">0.31</td><td align="left" valign="top">0.0526</td><td align="left" valign="top">&#x2212;0.0284</td><td align="left" valign="top">Slow decay</td></tr><tr><td align="left" valign="top">MLP<sup><xref ref-type="table-fn" rid="table14fn3">c</xref></sup></td><td align="left" valign="top">21</td><td align="left" valign="top">16</td><td align="left" valign="top">Mean 0.660, SD 0.160, range 0-1</td><td align="left" valign="top">Left-skewed</td><td align="left" valign="top">0.6604</td><td align="left" valign="top">0.0263</td><td align="left" valign="top">&#x2212;0.0085</td><td align="left" valign="top">Slow decay</td></tr><tr><td align="left" valign="top">Random forest</td><td align="left" valign="top">21</td><td align="left" valign="top">16</td><td align="left" valign="top">Mean 0.232, SD 0.292, range 0-1</td><td align="left" valign="top">Right-skewed</td><td align="left" valign="top">0.2321</td><td align="left" valign="top">0.2105</td><td align="left" valign="top">&#x2212;0.0239</td><td align="left" valign="top">Slow decay</td></tr><tr><td align="left" valign="top">CatBoost<sup><xref ref-type="table-fn" rid="table14fn4">d</xref></sup></td><td align="left" valign="top">17</td><td align="left" valign="top">10</td><td align="left" valign="top">Mean 0.069, SD 0.200, range 0-1</td><td align="left" valign="top">Right-skewed</td><td align="left" valign="top">0.0692</td><td align="left" valign="top">0.2895</td><td align="left" valign="top">&#x2212;0.0096</td><td align="left" valign="top">Slow decay</td></tr><tr><td align="left" valign="top">AdaBoost<sup><xref ref-type="table-fn" rid="table14fn5">e</xref></sup></td><td align="left" valign="top">11</td><td align="left" valign="top">0</td><td align="left" valign="top">Mean 0.105, SD 0.201, range 0-1</td><td align="left" valign="top">Right-skewed</td><td align="left" valign="top">0.1045</td><td align="left" valign="top">0.7105</td><td align="left" valign="top">&#x2212;0.013</td><td align="left" valign="top">Slow decay</td></tr><tr><td align="left" valign="top">LightGBM<sup><xref ref-type="table-fn" rid="table14fn6">f</xref></sup></td><td align="left" valign="top">5</td><td align="left" valign="top">3</td><td align="left" valign="top">Mean 0.061, SD 0.178, range 0-1</td><td align="left" valign="top">Right-skewed</td><td align="left" valign="top">0.061</td><td align="left" valign="top">0.7895</td><td align="left" valign="top">&#x2212;0.0086</td><td align="left" valign="top">Slow decay</td></tr><tr><td align="left" valign="top">XGBoost<sup><xref ref-type="table-fn" rid="table14fn7">g</xref></sup></td><td align="left" valign="top">3</td><td align="left" valign="top">0</td><td align="left" valign="top">Mean 0.079, SD 0.270, range 0-1</td><td align="left" valign="top">Right-skewed</td><td align="left" valign="top">0.0789</td><td align="left" valign="top">0.9211</td><td align="left" valign="top">&#x2212;0.0115</td><td align="left" valign="top">Slow decay</td></tr><tr><td align="left" valign="top">Overall aggregated summary</td><td align="left" valign="top">21</td><td align="left" valign="top">17</td><td align="left" valign="top">Mean 0.311, SD 0.390, range 0-1</td><td align="left" valign="top">Right-skewed</td><td align="left" valign="top">0.311 (0.313)</td><td align="left" valign="top">0.378 (0.347)</td><td align="left" valign="top">&#x2212;0.0134 (0.0078)</td><td align="left" valign="top">Slow decay</td></tr></tbody></table><table-wrap-foot><fn id="table14fn1"><p><sup>a</sup>FI: feature importance.</p></fn><fn id="table14fn2"><p><sup>b</sup>SVM: support vector machine.</p></fn><fn id="table14fn3"><p><sup>c</sup>MLP: multilayer perceptron.</p></fn><fn id="table14fn4"><p><sup>d</sup>CatBoost: categorical boosting.</p></fn><fn id="table14fn5"><p><sup>e</sup>AdaBoost: adaptive boosting.</p></fn><fn id="table14fn6"><p><sup>f</sup>LightGBM: light gradient boosting machine.</p></fn><fn id="table14fn7"><p><sup>g</sup>XGBoost: extreme gradient boosting.</p></fn></table-wrap-foot></table-wrap></sec></sec></sec><sec id="s4" sec-type="discussion"><title>Discussion</title><p>While prior studies proposing AI-assisted diagnosis for skin NTDs primarily focused on image-based approaches, the use of structured clinical skin NTD data, the methodological implications, and enhancement strategies remained to be the major gaps. This preliminary pilot study demonstrated ML-based diagnosis of skin NTDs based on structured patient data, marking a high contribution to the problem domain, through the findings and their implications highlighted hereunder.</p><sec id="s4-1"><title>Important Findings</title><sec id="s4-1-1"><title>Novel Dataset and Diversified Data-Related Challenges</title><p>In this study, we crafted a new structured skin NTD clinical dataset, confronted by multifaceted data-related challenges (sample size limitation, severe class imbalance, and lack of full disease representation). We used our novel dataset in 3 different structures with the aim of experimentally analyzing, comparing, and identifying the best-performing ML model for the diagnosis of skin NTDs based on tabular patient data. In achieving the objectives, the study recorded multiple findings with different implications. The use of the initial dataset (having the structural missing values) clearly demonstrated the resiliency of the 4 tree-based models (RF, CatBoost, LightGBM, and XGBoost) to null values due to their built-in mechanisms of handling missing values, which include considering the missing values as a separate group [<xref ref-type="bibr" rid="ref41">41</xref>,<xref ref-type="bibr" rid="ref42">42</xref>]. These results underlined the effectiveness of these tree-based architectures to establish robust real-time diagnostic tools, including the scenarios of having incomplete diagnostic data.</p></sec><sec id="s4-1-2"><title>Methodological Rigor</title><p>Given the size and distribution of the new dataset, our study underlined that the selection and utilization of sound ML methods is the key to beat expected generalizability challenges, as demonstrated by the progressive performance gains across experiments. As an initial experimental finding, the &#x0394;+0.097 (10%) and &#x0394;+0.204 (24%) macro recall and <italic>F</italic><sub>1</sub>-score boosts of MLP validate the adoption of feature engineering as one big initial strategy in addressing and demonstrating our methodological success. Another important finding highlighted that some models, specifically LightGBM and XGBoost, exhibited unreliable performance due to predictive bias, a situation where the standard validation methods (the test CV performance in our case) were pretentiously presenting biased prediction estimates [<xref ref-type="bibr" rid="ref38">38</xref>,<xref ref-type="bibr" rid="ref39">39</xref>]. Such performance traps were exposed by the nested CV scores, proving the robustness of the dual CV-based approach in establishing unbiased disease prediction estimates, given the highly constrained dataset used in this study.</p><p>Overall, this study highlighted that stable model performances with higher predictive accuracies are achievable through a synergistic integration of multiple ML strategies working in harmony, instead of individual and isolated methods&#x2014;first, establishing a reliable data preprocessing pipeline followed by a robust validation pipeline incorporating a synchronized implementation of class, all applied on top of the robust data preprocessing pipeline. However, as all these results are achieved against all the odds of data-related constraints, it is mandatory to have further investigations with extended resources to maintain the high achievements.</p></sec><sec id="s4-1-3"><title>Overall Performance of Models</title><p>Based on the overall test performance scores, the majority of the models (except NB and MLP) exhibited consistent performance scores across all metrics. While desirable, perfect scores are not always good signs, given the dataset constraints of this study. Specifically, the perfect scores achieved by the highly feature-selective tree-based models, such as XGBoost, that used only 3 features for disease prediction, raise critical performance concerns. Positively, the perfect scores indicate the maximum disease discriminative power of the model. However, given the fact that the model makes disease predictions based on only 3 features out of 38, and these scores showed declines during the nested CV training, these perfect scores have negative implications, indicating that these scores are fragile and misleading. Such scores indicate that the models are facing the problem of over-optimism in their overall test performance. The use of such models for the diagnosis of diseases using only 3 diagnostic features (3 different symptoms of 3 different diseases) is highly risky and clinically infeasible, as the model would fail if one symptom is missing or misrepresented.</p></sec></sec><sec id="s4-2"><title>Implications of Results for Model Selection and Clinical Practices</title><sec id="s4-2-1"><title>Performance vs Characteristics of Models</title><p>According to the FI analysis reports, the diagnostic features have the highest importance for the classification of the skin NTDs and are most widely used by all models, while the demographic features are the least frequently used set of features for the classification of the skin NTDs. These results highlight that models using almost the whole feature set, such as random forest, achieved 100% performance scores across all metrics, while LightGBM (8 features) and XGBoost using only 3 features scored the same 100% performance. These tendencies of the models in using the least proportions of the trainable features could suggest that the models are applying complex architectures and are suffering from overfitting, given the limited dimensionality of our dataset used. Conversely, the tendencies of models to use the highest proportion of the trainable features could also point to the same problem, though such challenges can be alleviated through the use of ML-based strategies, as in the case of this study. Therefore, all these factors suggest that tradeoffs need to be made for the final model selection. Overall performance analysis and FI scores revealed that the 8 models exhibited different behaviors, forming 3 groups of different model characteristics.</p></sec><sec id="s4-2-2"><title>All-Inclusiveness</title><p>Four models (NB, SVM, MLP, and RF) generally fall under this group due to the behavior they commonly share by equally ranking 37 features, while each model has specific determining characteristics. Accordingly, NB showed a different (outlier) behavior due to its invariable selection character that considers all features are equally important, having no interdependency or highly correlated relationship. This high-level feature-inseparability raises an issue, as it is not feasible for 2 or more features to be equally important in diagnosing skin NTDs. For instance, according to the results of our experimental results, a given value of the &#x201C;educational level&#x201D; of all patients has never been equally important as the diagnostic features, as confirmed by the dominant diagnostic feature &#x201C;JI_classification(#Jig-Inf)&#x201D; and its applicability flag. Further, the introduction of the symptom applicability flags created performance distortions. Both these factors diluted the reliability of the NB models, making them unable to differentiate between symptoms. This, in turn, results in a model inability to differentiate between diseases, which led to perfect scores, where such models cannot be reliable candidates for clinical diagnostics. MLP is the other all-inclusive model with a slightly different feature ranking behavior compared to NB, where 27 features (14 diagnostic and 13 demographic) showed an equal feature rank of 0.665. This depicts invariable ranks (&#x03C3;=0) among the 27 features, highlighting that the model failed to effectively differentiate disease signal from noise, which also proved that this neural network-based model showed overfitting due to the model&#x2019;s complex representational ability, given our small-sized tabular dataset. The other 2 models (SVM and RF) showed similar patterns of feature ranking scores with increasing feature sparsity, where the variance in FI scores increased among features.</p></sec><sec id="s4-2-3"><title>Shortcut Learners</title><p>Results showed that 3 of the tree-based models (XGBoost, LightGBM, and AdaBoost) exhibited a &#x201C;lazy learning&#x201D; pattern by using a limited number of features (3, 8, and 11 features), while all these 3 models achieved perfect scores. Especially with XGBoost showing absolute feature parsimony (showing high feature selectivity), the perfect scores using only 3 diagnostic features highlight a potential tendency of following shortcut paths by memorizing the feature patterns with the highest performance. Given the dataset constraints and the complexity of these models, this can underscore the plausibility of this shortcut-based (lazy) learning tendency by these 3 models, as exceptionally exposed by the overall performance declines recorded during the nested CV experiment.</p></sec><sec id="s4-2-4"><title>Final Model Selection: Optimal Classifier</title><p>After the extensive analysis of the models&#x2019; performance, the study concludes by identifying the model that achieved the optimal performance as the final skin NTD diagnostic model based on structured clinical data. According to the analytic results of the feature preference scores, CatBoost demonstrated the average performance standard compared to all models in our study. Statistically, based on the number of features identified as important (27 features, 71%), the distribution of the features (17 diagnostic and 10 demographic), and performance stability across experimental phases including nested CV, CatBoost showed an optimal separation pattern, unlike the other 2 groups of models. Additionally, its overall internal working logic of ordered boosting (to reduce prediction variances) and inherent categorical feature handling [<xref ref-type="bibr" rid="ref40">40</xref>], including its fast and stable inference capabilities [<xref ref-type="bibr" rid="ref43">43</xref>], could also play vital roles in helping the model achieve higher and more stable performance based on an optimal set of features. Therefore, we identified and proposed the CatBoost model as the potential benchmark skin NTD diagnostic model.</p></sec></sec><sec id="s4-3"><title>Conclusions</title><p>This study developed an ML-based diagnostic model for skin NTDs using a new tabular skin NTD diagnostic dataset (IDS). While the use of FDS (dataset created by applying only data preprocessing) resulted in perfect scores in all models except MLP, the third dataset architecture (created by applying feature engineering on FDS) created a new challenge for NB (due to statistical assumptions of the model), while stabilizing the performance of the other models. These data are used to experimentally analyze, compare, and identify the dataset structure with an optimal set of features, handle structural missing data, control overfitting, and achieve optimal model performance. The class weighting method we used helped in controlling the impact of the severe class imbalance by stabilizing the models&#x2019; performance. Though all 6 models scored perfect scores across all metrics (except NB and MLP), the dual CV method proved its robustness, as the nested CV spotted hidden weaknesses in 2 of the tree-based models, XGBoost and LightGBM, showing slight performance declines while these models showed 100% test scores. These two models exhibited a high level of feature parsimony, while NB, SVM (Linear), MLP, and RF were highly feature inclusive. However, CatBoost showed optimal feature ranking and usage patterns. Overall, the boosting tree-based model CatBoost exhibited consistently higher performance, demonstrating optimal feature ranking and usage behavior, which substantiates the robustness of the model. Hence, we recommend further studies with extended resources using the CatBoost model, including RF and SVM as secondary alternatives, to demonstrate reliable diagnostic performance for further deployment.</p><p>While this study successfully established a robust ML pipeline as a methodological framework and developed a robust benchmark diagnostic model for skin NTDs, several inherent constraints created restraints to the study, limiting its contribution to its full potential. Data scarcity was the primary limitation of this study, while a severe class imbalance was also a constraint. Other limitations of the study include limited disease representation, inclusion of only one data modality, and specific geographic representation. Therefore, we recommend that further studies be conducted by collecting more patient data from multiple affected and potential areas, with the data being representative of all disease classes that are endemic to Ethiopia. This study also suggests the use of DL-based methods and the inclusion of other patient data (such as images and laboratory results), through proper balancing between diagnostic accuracy and computational expenses, to transform the diagnosis of skin NTDs into higher-level technology-assisted platforms and deliver quality health care services for affected areas, especially for resource-limited areas.</p></sec></sec></body><back><ack><p>This study uses a novel dataset created using the data collected from patients with skin neglected tropical diseases (NTDs) living in Gacho Baba District of the Gamo Zone, Southwest Ethiopia. The data were initially collected for a project-based research intended to assess the burden of skin NTDs through community screening, led by Mr Alemayehu Bekele (Collaborative Research and Training Center for Neglected Tropical Diseases, Arba Minch University Medical College). We acquired the data for this study through institutional collaboration after having the required ethical clearance letter.</p><p>Therefore, the authors of this study gratefully acknowledge Mr Alemayehu Bekele and his team for providing the data to be used in this study, including the technical support he provided us.</p><p>All authors declared that they had insufficient or no funding to support open access publication of this manuscript, including from affiliated organizations or institutions, funding agencies, or other organizations. JMIR Publications provided APF support for the publication of this article. Finally, we (the authors this study) would like to declare that no generative tools and generated contents are used for the preparation of any part of this study.</p></ack><notes><sec><title>Funding</title><p>The authors declared no financial support was received for this work.</p></sec></notes><fn-group><fn fn-type="conflict"><p>None declared.</p></fn></fn-group><glossary><title>Abbreviations</title><def-list><def-item><term id="abb1">AUPRC</term><def><p>area under the precision-recall curve</p></def></def-item><def-item><term id="abb2">CL</term><def><p>cutaneous leishmaniasis</p></def></def-item><def-item><term id="abb3">CNN</term><def><p>convolutional neural network</p></def></def-item><def-item><term id="abb4">CV</term><def><p>cross-validation</p></def></def-item><def-item><term id="abb5">DL</term><def><p>deep learning</p></def></def-item><def-item><term id="abb6">DTAG</term><def><p>Diagnostic Technical Advisory Group</p></def></def-item><def-item><term id="abb7">FDS</term><def><p>final dataset</p></def></def-item><def-item><term id="abb8">FEFDS</term><def><p>feature engineering on final dataset</p></def></def-item><def-item><term id="abb9">FI</term><def><p>feature importance</p></def></def-item><def-item><term id="abb10">IDS</term><def><p>initial dataset</p></def></def-item><def-item><term id="abb11">KNN</term><def><p><italic>k</italic>-nearest neighbors</p></def></def-item><def-item><term id="abb12">LightGBM</term><def><p>light gradient boosting model</p></def></def-item><def-item><term id="abb13">MAR</term><def><p>missing at random</p></def></def-item><def-item><term id="abb14">MCAR</term><def><p>missing completely at random</p></def></def-item><def-item><term id="abb15">MDA</term><def><p>mass drug administration</p></def></def-item><def-item><term id="abb16">MICE</term><def><p>multivariate imputation by chained equation</p></def></def-item><def-item><term id="abb17">ML</term><def><p>machine learning</p></def></def-item><def-item><term id="abb18">MLP</term><def><p>multilayer perceptron</p></def></def-item><def-item><term id="abb19">NB</term><def><p>na&#x00EF;ve Bayes</p></def></def-item><def-item><term id="abb20">NPV</term><def><p>negative predictive value</p></def></def-item><def-item><term id="abb21">NTD</term><def><p>neglected tropical disease</p></def></def-item><def-item><term id="abb22">PCA</term><def><p>principal component analysis</p></def></def-item><def-item><term id="abb23">RF</term><def><p>random forest</p></def></def-item><def-item><term id="abb24">SVM</term><def><p>support vector machine</p></def></def-item><def-item><term id="abb25">TNR</term><def><p>true negative rate</p></def></def-item><def-item><term id="abb26">TPR</term><def><p>true positive rate</p></def></def-item><def-item><term id="abb27">WHO</term><def><p>World Health Organization</p></def></def-item><def-item><term id="abb28">XGBoost</term><def><p>extreme gradient boosting</p></def></def-item></def-list></glossary><ref-list><title>References</title><ref id="ref1"><label>1</label><nlm-citation citation-type="report"><article-title>Elimination of neglected tropical diseases (NTDs) in Ethiopia &#x2013; WOREDA level coordination toolkit for the WASH and NTD sectors</article-title><year>2019</year><access-date>2026-08-01</access-date><publisher-name>Ethiopian Federal Ministry of Health</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.susana.org/knowledge-hub/resources?id=3709">https://www.susana.org/knowledge-hub/resources?id=3709</ext-link></comment></nlm-citation></ref><ref id="ref2"><label>2</label><nlm-citation citation-type="web"><article-title>Neglected tropical diseases</article-title><source>World Health Organization</source><year>2025</year><month>12</month><day>14</day><access-date>2026-08-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/news-room/questions-and-answers/item/neglected-tropical-diseases">https://www.who.int/news-room/questions-and-answers/item/neglected-tropical-diseases</ext-link></comment></nlm-citation></ref><ref id="ref3"><label>3</label><nlm-citation citation-type="report"><article-title>Ending the neglect to attain the sustainable development goals: a road map for neglected tropical diseases 2021&#x2013;2030</article-title><year>2020</year><month>01</month><day>28</day><access-date>2026-08-01</access-date><publisher-name>World Health Organization</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/publications/i/item/9789240010352">https://www.who.int/publications/i/item/9789240010352</ext-link></comment></nlm-citation></ref><ref id="ref4"><label>4</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Abdela</surname><given-names>SG</given-names> </name><name name-style="western"><surname>Diro</surname><given-names>E</given-names> </name><name name-style="western"><surname>Zewdu</surname><given-names>FT</given-names> </name><etal/></person-group><article-title>Looking for NTDs in the skin; an entry door for offering patient centered holistic care</article-title><source>J Infect Dev Ctries</source><year>2020</year><month>06</month><day>29</day><volume>14</volume><issue>6.1</issue><fpage>16S</fpage><lpage>21S</lpage><pub-id pub-id-type="doi">10.3855/jidc.11707</pub-id><pub-id pub-id-type="medline">32614791</pub-id></nlm-citation></ref><ref id="ref5"><label>5</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Deribe</surname><given-names>K</given-names> </name><name name-style="western"><surname>Meribo</surname><given-names>K</given-names> </name><name name-style="western"><surname>Gebre</surname><given-names>T</given-names> </name><etal/></person-group><article-title>The burden of neglected tropical diseases in Ethiopia, and opportunities for integrated control and elimination</article-title><source>Parasit Vectors</source><year>2012</year><month>10</month><day>24</day><volume>5</volume><issue>1</issue><fpage>240</fpage><pub-id pub-id-type="doi">10.1186/1756-3305-5-240</pub-id><pub-id pub-id-type="medline">23095679</pub-id></nlm-citation></ref><ref id="ref6"><label>6</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Carrion</surname><given-names>C</given-names> </name><name name-style="western"><surname>Robles</surname><given-names>N</given-names> </name><name name-style="western"><surname>Sola-Morales</surname><given-names>O</given-names> </name><name name-style="western"><surname>Aymerich</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ruiz Postigo</surname><given-names>JA</given-names> </name></person-group><article-title>Mobile health strategies to tackle skin neglected tropical diseases with recommendations from innovative experiences: systematic review</article-title><source>JMIR Mhealth Uhealth</source><year>2020</year><month>12</month><day>31</day><volume>8</volume><issue>12</issue><fpage>e22478</fpage><pub-id pub-id-type="doi">10.2196/22478</pub-id><pub-id pub-id-type="medline">33382382</pub-id></nlm-citation></ref><ref id="ref7"><label>7</label><nlm-citation citation-type="report"><article-title>The third national neglected tropical diseases strategic plan 2021-2025</article-title><year>2021</year><month>11</month><access-date>2026-08-01</access-date><publisher-name>Ethiopian Federal Ministry of Health</publisher-name><fpage>1</fpage><lpage>116</lpage><comment><ext-link ext-link-type="uri" xlink:href="https://espen.afro.who.int/sites/default/files/content/document/Third%20NTD%20national%20Strategic%20Plan%202021-2025_0.pdf">https://espen.afro.who.int/sites/default/files/content/document/Third%20NTD%20national%20Strategic%20Plan%202021-2025_0.pdf</ext-link></comment></nlm-citation></ref><ref id="ref8"><label>8</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Marks</surname><given-names>M</given-names> </name><name name-style="western"><surname>Vedithi</surname><given-names>SC</given-names> </name><name name-style="western"><surname>van de Sande</surname><given-names>WWJ</given-names> </name><etal/></person-group><article-title>A pathway for skin NTD diagnostic development</article-title><source>PLoS Negl Trop Dis</source><year>2024</year><month>11</month><volume>18</volume><issue>11</issue><fpage>e0012661</fpage><pub-id pub-id-type="doi">10.1371/journal.pntd.0012661</pub-id><pub-id pub-id-type="medline">39585842</pub-id></nlm-citation></ref><ref id="ref9"><label>9</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Tadesse</surname><given-names>D</given-names> </name><name name-style="western"><surname>van Henten</surname><given-names>S</given-names> </name><name name-style="western"><surname>Batire</surname><given-names>S</given-names> </name><etal/></person-group><article-title>Decentralizing care for cutaneous leishmaniasis and other skin diseases to primary health facilities in Southern Ethiopia: what are the needs?</article-title><source>BMC Infect Dis</source><year>2025</year><volume>26</volume><fpage>206</fpage><pub-id pub-id-type="doi">10.1186/s12879-025-12324-0</pub-id></nlm-citation></ref><ref id="ref10"><label>10</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yotsu</surname><given-names>RR</given-names> </name><name name-style="western"><surname>Almamy</surname><given-names>D</given-names> </name><name name-style="western"><surname>Vagamon</surname><given-names>B</given-names> </name><etal/></person-group><article-title>An mHealth app (eSkinHealth) for detecting and managing skin diseases in resource-limited settings: mixed methods pilot study</article-title><source>JMIR Dermatol</source><year>2023</year><month>06</month><day>14</day><volume>6</volume><fpage>e46295</fpage><pub-id pub-id-type="doi">10.2196/46295</pub-id><pub-id pub-id-type="medline">37632977</pub-id></nlm-citation></ref><ref id="ref11"><label>11</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Barbieri</surname><given-names>RR</given-names> </name><name name-style="western"><surname>Xu</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Setian</surname><given-names>L</given-names> </name><etal/></person-group><article-title>Reimagining leprosy elimination with AI analysis of a combination of skin lesion images with demographic and clinical data</article-title><source>Lancet Reg Health Am</source><year>2022</year><month>05</month><volume>9</volume><fpage>100192</fpage><pub-id pub-id-type="doi">10.1016/j.lana.2022.100192</pub-id><pub-id pub-id-type="medline">36776278</pub-id></nlm-citation></ref><ref id="ref12"><label>12</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Achary</surname><given-names>R</given-names> </name><name name-style="western"><surname>Shelke</surname><given-names>CJ</given-names> </name><name name-style="western"><surname>Lekhya</surname><given-names>A</given-names> </name></person-group><article-title>A GAN-enhanced multimodal diagnostic framework utilizing an ensemble of BiLSTM, BiGRU, and RNN models for malaria and dengue detection</article-title><source>Procedia Comput Sci</source><year>2025</year><volume>252</volume><fpage>381</fpage><lpage>393</lpage><pub-id pub-id-type="doi">10.1016/j.procs.2024.12.039</pub-id></nlm-citation></ref><ref id="ref13"><label>13</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Yotsu</surname><given-names>RR</given-names> </name></person-group><article-title>Integrated management of skin NTDs&#x2014;lessons learned from existing practice and field research</article-title><source>Trop Med Infect Dis</source><year>2018</year><month>11</month><day>14</day><volume>3</volume><issue>4</issue><fpage>120</fpage><pub-id pub-id-type="doi">10.3390/tropicalmed3040120</pub-id><pub-id pub-id-type="medline">30441754</pub-id></nlm-citation></ref><ref id="ref14"><label>14</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Kokol</surname><given-names>P</given-names> </name><name name-style="western"><surname>Kokol</surname><given-names>M</given-names> </name><name name-style="western"><surname>Zagoranski</surname><given-names>S</given-names> </name></person-group><article-title>Machine learning on small size samples: a synthetic knowledge synthesis</article-title><source>Sci Prog</source><year>2022</year><volume>105</volume><issue>1</issue><pub-id pub-id-type="doi">10.1177/00368504211029777</pub-id><pub-id pub-id-type="medline">35220816</pub-id></nlm-citation></ref><ref id="ref15"><label>15</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Xu</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ji</surname><given-names>X</given-names> </name><name name-style="western"><surname>Li</surname><given-names>M</given-names> </name><name name-style="western"><surname>Lu</surname><given-names>W</given-names> </name></person-group><article-title>Small data machine learning in materials science</article-title><source>npj Comput Mater</source><year>2023</year><volume>9</volume><issue>1</issue><fpage>1</fpage><lpage>15</lpage><pub-id pub-id-type="doi">10.1038/s41524-023-01000-z</pub-id></nlm-citation></ref><ref id="ref16"><label>16</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Shwartz-Ziv</surname><given-names>R</given-names> </name><name name-style="western"><surname>Armon</surname><given-names>A</given-names> </name></person-group><article-title>Tabular data: Deep learning is not all you need</article-title><source>Inf Fusion</source><year>2022</year><month>05</month><volume>81</volume><fpage>84</fpage><lpage>90</lpage><pub-id pub-id-type="doi">10.1016/j.inffus.2021.11.011</pub-id></nlm-citation></ref><ref id="ref17"><label>17</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Florek</surname><given-names>P</given-names> </name><name name-style="western"><surname>Zagda&#x0144;ski</surname><given-names>A</given-names> </name></person-group><article-title>Benchmarking state-of-the-art gradient boosting algorithms for classification</article-title><source>arXiv</source><comment>Preprint posted online on  May 26, 2023</comment><pub-id pub-id-type="doi">10.48550/arXiv.2305.17094</pub-id></nlm-citation></ref><ref id="ref18"><label>18</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Wen</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Li</surname><given-names>Q</given-names> </name><name name-style="western"><surname>He</surname><given-names>B</given-names> </name><name name-style="western"><surname>Cui</surname><given-names>B</given-names> </name></person-group><article-title>Challenges and opportunities of building fast GBDT systems</article-title><source>Proceedings of the 30th International Joint Conference on Artificial Intelligence</source><year>2021</year><publisher-name>IJCAI</publisher-name><fpage>4661</fpage><lpage>4668</lpage><pub-id pub-id-type="doi">10.24963/ijcai.2021/632</pub-id></nlm-citation></ref><ref id="ref19"><label>19</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Wang</surname><given-names>X</given-names> </name><name name-style="western"><surname>Jia</surname><given-names>W</given-names> </name></person-group><article-title>Optimizing edge AI: a comprehensive survey on data, model, and system strategies</article-title><source>arXiv</source><comment>Preprint posted online on  Jan 4, 2025</comment><pub-id pub-id-type="doi">10.48550/arXiv.2501.03265</pub-id></nlm-citation></ref><ref id="ref20"><label>20</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Aquil</surname><given-names>A</given-names> </name><name name-style="western"><surname>Saeed</surname><given-names>F</given-names> </name><name name-style="western"><surname>Baowidan</surname><given-names>S</given-names> </name><name name-style="western"><surname>Ali</surname><given-names>AM</given-names> </name><name name-style="western"><surname>Elmitwally</surname><given-names>NS</given-names> </name></person-group><article-title>Early detection of skin diseases across diverse skin tones using hybrid machine learning and deep learning models</article-title><source>Information</source><year>2025</year><month>02</month><volume>16</volume><issue>2</issue><fpage>152</fpage><pub-id pub-id-type="doi">10.3390/info16020152</pub-id></nlm-citation></ref><ref id="ref21"><label>21</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jeong</surname><given-names>HK</given-names> </name><name name-style="western"><surname>Park</surname><given-names>C</given-names> </name><name name-style="western"><surname>Henao</surname><given-names>R</given-names> </name><name name-style="western"><surname>Kheterpal</surname><given-names>M</given-names> </name></person-group><article-title>Deep learning in dermatology: a systematic review of current approaches, outcomes, and limitations</article-title><source>JID Innov</source><year>2023</year><month>01</month><volume>3</volume><issue>1</issue><fpage>100150</fpage><pub-id pub-id-type="doi">10.1016/j.xjidi.2022.100150</pub-id><pub-id pub-id-type="medline">36655135</pub-id></nlm-citation></ref><ref id="ref22"><label>22</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Silvey</surname><given-names>S</given-names> </name><name name-style="western"><surname>Liu</surname><given-names>J</given-names> </name></person-group><article-title>Sample size requirements for popular classification algorithms in tabular clinical data: empirical study</article-title><source>J Med Internet Res</source><year>2024</year><month>12</month><day>17</day><volume>26</volume><fpage>e60231</fpage><pub-id pub-id-type="doi">10.2196/60231</pub-id><pub-id pub-id-type="medline">39689306</pub-id></nlm-citation></ref><ref id="ref23"><label>23</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Namburu</surname><given-names>A</given-names> </name><name name-style="western"><surname>Selvaraj</surname><given-names>P</given-names> </name><name name-style="western"><surname>Varsha</surname><given-names>M</given-names> </name></person-group><article-title>Product pricing solutions using hybrid machine learning algorithm</article-title><source>Innov Syst Softw Eng</source><year>2022</year><month>07</month><day>25</day><fpage>1</fpage><lpage>12</lpage><pub-id pub-id-type="doi">10.1007/s11334-022-00465-3</pub-id><pub-id pub-id-type="medline">35910813</pub-id></nlm-citation></ref><ref id="ref24"><label>24</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Pattnayak</surname><given-names>P</given-names> </name><name name-style="western"><surname>Mohanty</surname><given-names>A</given-names> </name><name name-style="western"><surname>Das</surname><given-names>T</given-names> </name><name name-style="western"><surname>Patnaik</surname><given-names>S</given-names> </name></person-group><article-title>Applying artificial intelligence and deep learning to identify neglected tropical skin disorders</article-title><source>2024 3rd International Conference for Innovation in Technology (INOCON)</source><year>2024</year><publisher-name>IEEE</publisher-name><pub-id pub-id-type="doi">10.1109/INOCON60754.2024.10511323</pub-id></nlm-citation></ref><ref id="ref25"><label>25</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Steyve</surname><given-names>N</given-names> </name><name name-style="western"><surname>Steve</surname><given-names>P</given-names> </name><name name-style="western"><surname>Ghislain</surname><given-names>M</given-names> </name><name name-style="western"><surname>Ndjakomo</surname><given-names>S</given-names> </name><name name-style="western"><surname>pierre</surname><given-names>E</given-names> </name></person-group><article-title>Optimized real-time diagnosis of neglected tropical diseases by automatic recognition of skin lesions</article-title><source>Inform Med Unlocked</source><year>2022</year><volume>33</volume><fpage>101078</fpage><pub-id pub-id-type="doi">10.1016/j.imu.2022.101078</pub-id></nlm-citation></ref><ref id="ref26"><label>26</label><nlm-citation citation-type="report"><article-title>Recognizing neglected tropical diseases through changes on the skin: a training guide for front-line health workers</article-title><year>2018</year><month>06</month><day>1</day><access-date>2026-08-01</access-date><publisher-name>World Health Organization</publisher-name><comment><ext-link ext-link-type="uri" xlink:href="https://www.who.int/publications/i/item/9789241513531">https://www.who.int/publications/i/item/9789241513531</ext-link></comment></nlm-citation></ref><ref id="ref27"><label>27</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Jackson</surname><given-names>J</given-names> </name><name name-style="western"><surname>Hofert</surname><given-names>M</given-names> </name><name name-style="western"><surname>Andrinopoulou</surname><given-names>E</given-names> </name><etal/></person-group><article-title>The challenge of handling structured missingness in integrated data sources</article-title><source>Adv Intell Discov</source><year>2025</year><fpage>e202500089</fpage><pub-id pub-id-type="doi">10.1002/aidi.202500089</pub-id></nlm-citation></ref><ref id="ref28"><label>28</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Newman</surname><given-names>DA</given-names> </name></person-group><article-title>Missing data: five practical guidelines</article-title><source>Organ Res Methods</source><year>2014</year><month>10</month><volume>17</volume><issue>4</issue><fpage>372</fpage><lpage>411</lpage><pub-id pub-id-type="doi">10.1177/1094428114548590</pub-id></nlm-citation></ref><ref id="ref29"><label>29</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>de Amorim</surname><given-names>LBV</given-names> </name><name name-style="western"><surname>Cavalcanti</surname><given-names>GDC</given-names> </name><name name-style="western"><surname>Cruz</surname><given-names>RMO</given-names> </name></person-group><article-title>The choice of scaling technique matters for classification performance</article-title><source>Appl Soft Comput</source><year>2023</year><month>01</month><volume>133</volume><fpage>109924</fpage><pub-id pub-id-type="doi">10.1016/j.asoc.2022.109924</pub-id></nlm-citation></ref><ref id="ref30"><label>30</label><nlm-citation citation-type="web"><article-title>Metadata routing (&#x201C;sample_weight&#x201D;)</article-title><source>scikit-learn</source><access-date>2026-08-01</access-date><comment><ext-link ext-link-type="uri" xlink:href="https://scikit-learn.org/stable/metadata_routing.html">https://scikit-learn.org/stable/metadata_routing.html</ext-link></comment></nlm-citation></ref><ref id="ref31"><label>31</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Waljee</surname><given-names>AK</given-names> </name><name name-style="western"><surname>Mukherjee</surname><given-names>A</given-names> </name><name name-style="western"><surname>Singal</surname><given-names>AG</given-names> </name><etal/></person-group><article-title>Comparison of imputation methods for missing laboratory data in medicine</article-title><source>BMJ Open</source><year>2013</year><month>08</month><day>1</day><volume>3</volume><issue>8</issue><fpage>e002847</fpage><pub-id pub-id-type="doi">10.1136/bmjopen-2013-002847</pub-id><pub-id pub-id-type="medline">23906948</pub-id></nlm-citation></ref><ref id="ref32"><label>32</label><nlm-citation citation-type="other"><person-group person-group-type="author"><name name-style="western"><surname>Zhuang</surname><given-names>N</given-names> </name><name name-style="western"><surname>Howells</surname><given-names>J</given-names> </name></person-group><article-title>A computational pipeline for stratifying autoimmune patients using binary antibody data</article-title><source>bioRxiv</source><comment>Preprint posted online on  Dec 19, 2026</comment><pub-id pub-id-type="doi">10.1101/2025.09.11.670596</pub-id></nlm-citation></ref><ref id="ref33"><label>33</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Lumumba</surname><given-names>V</given-names> </name><name name-style="western"><surname>Kiprotich</surname><given-names>D</given-names> </name><name name-style="western"><surname>Mpaine</surname><given-names>M</given-names> </name><name name-style="western"><surname>Makena</surname><given-names>N</given-names> </name><name name-style="western"><surname>Kavita</surname><given-names>M</given-names> </name></person-group><article-title>Comparative analysis of cross-validation techniques: LOOCV, k-folds cross-validation, and repeated k-folds cross-validation in machine learning models</article-title><source>Am J Theor Appl Stat</source><year>2024</year><month>10</month><volume>13</volume><issue>5</issue><fpage>127</fpage><lpage>137</lpage><pub-id pub-id-type="doi">10.11648/j.ajtas.20241305.13</pub-id></nlm-citation></ref><ref id="ref34"><label>34</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Salmi</surname><given-names>M</given-names> </name><name name-style="western"><surname>Atif</surname><given-names>D</given-names> </name><name name-style="western"><surname>Oliva</surname><given-names>D</given-names> </name><name name-style="western"><surname>Abraham</surname><given-names>A</given-names> </name><name name-style="western"><surname>Ventura</surname><given-names>S</given-names> </name></person-group><article-title>Handling imbalanced medical datasets: review of a decade of research</article-title><source>Artif Intell Rev</source><year>2024</year><volume>57</volume><issue>10</issue><fpage>273</fpage><pub-id pub-id-type="doi">10.1007/s10462-024-10884-2</pub-id></nlm-citation></ref><ref id="ref35"><label>35</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Hr</surname><given-names>S</given-names> </name><name name-style="western"><surname>B</surname><given-names>A</given-names> </name></person-group><article-title>Exploratory analysis of methods, techniques, and metrics to handle class imbalance problem</article-title><source>Procedia Comput Sci</source><year>2024</year><volume>235</volume><fpage>863</fpage><lpage>877</lpage><pub-id pub-id-type="doi">10.1016/j.procs.2024.04.082</pub-id></nlm-citation></ref><ref id="ref36"><label>36</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Chen</surname><given-names>W</given-names> </name><name name-style="western"><surname>Yang</surname><given-names>K</given-names> </name><name name-style="western"><surname>Yu</surname><given-names>Z</given-names> </name><name name-style="western"><surname>Shi</surname><given-names>Y</given-names> </name><name name-style="western"><surname>Chen</surname><given-names>CLP</given-names> </name></person-group><article-title>A survey on imbalanced learning: latest research, applications and future directions</article-title><source>Artif Intell Rev</source><year>2024</year><volume>57</volume><issue>6</issue><pub-id pub-id-type="doi">10.1007/s10462-024-10759-6</pub-id></nlm-citation></ref><ref id="ref37"><label>37</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Palanivinayagam</surname><given-names>A</given-names> </name><name name-style="western"><surname>Dama&#x0161;evi&#x010D;ius</surname><given-names>R</given-names> </name></person-group><article-title>Effective handling of missing values in datasets for classification using machine learning methods</article-title><source>Information</source><year>2023</year><month>02</month><volume>14</volume><issue>2</issue><fpage>92</fpage><pub-id pub-id-type="doi">10.3390/info14020092</pub-id></nlm-citation></ref><ref id="ref38"><label>38</label><nlm-citation citation-type="book"><person-group person-group-type="author"><name name-style="western"><surname>Grinsztajn</surname><given-names>L</given-names> </name><name name-style="western"><surname>Oyallon</surname><given-names>E</given-names> </name><name name-style="western"><surname>Varoquaux</surname><given-names>G</given-names> </name></person-group><article-title>Why do tree-based models still outperform deep learning on typical tabular data?</article-title><source>Proceedings of the 36th International Conference on Neural Information Processing Systems</source><year>2022</year><publisher-name>Curran Associates</publisher-name><pub-id pub-id-type="doi">10.52202/068431-0037</pub-id></nlm-citation></ref><ref id="ref39"><label>39</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Salvatier</surname><given-names>J</given-names> </name><name name-style="western"><surname>Wiecki</surname><given-names>TV</given-names> </name><name name-style="western"><surname>Fonnesbeck</surname><given-names>C</given-names> </name></person-group><article-title>Probabilistic programming in Python using PyMC3</article-title><source>PeerJ Comput Sci</source><year>2016</year><month>04</month><volume>2</volume><fpage>e55</fpage><pub-id pub-id-type="doi">10.7717/peerj-cs.55</pub-id></nlm-citation></ref><ref id="ref40"><label>40</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>L&#x00F6;tsch</surname><given-names>J</given-names> </name><name name-style="western"><surname>Ultsch</surname><given-names>A</given-names> </name></person-group><article-title>Recursive computed ABC (cABC) analysis as a precise method for reducing machine learning based feature sets to their minimum informative size</article-title><source>Sci Rep</source><year>2023</year><month>04</month><day>4</day><volume>13</volume><issue>1</issue><fpage>5470</fpage><pub-id pub-id-type="doi">10.1038/s41598-023-32396-9</pub-id><pub-id pub-id-type="medline">37016033</pub-id></nlm-citation></ref><ref id="ref41"><label>41</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Vabalas</surname><given-names>A</given-names> </name><name name-style="western"><surname>Gowen</surname><given-names>E</given-names> </name><name name-style="western"><surname>Poliakoff</surname><given-names>E</given-names> </name><name name-style="western"><surname>Casson</surname><given-names>AJ</given-names> </name></person-group><article-title>Machine learning algorithm validation with a limited sample size</article-title><source>PLOS ONE</source><year>2019</year><volume>14</volume><issue>11</issue><fpage>e0224365</fpage><pub-id pub-id-type="doi">10.1371/journal.pone.0224365</pub-id><pub-id pub-id-type="medline">31697686</pub-id></nlm-citation></ref><ref id="ref42"><label>42</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Parvandeh</surname><given-names>S</given-names> </name><name name-style="western"><surname>Yeh</surname><given-names>HW</given-names> </name><name name-style="western"><surname>Paulus</surname><given-names>MP</given-names> </name><name name-style="western"><surname>McKinney</surname><given-names>BA</given-names> </name></person-group><article-title>Consensus features nested cross-validation</article-title><source>Bioinformatics</source><year>2020</year><month>05</month><day>1</day><volume>36</volume><issue>10</issue><fpage>3093</fpage><lpage>3098</lpage><pub-id pub-id-type="doi">10.1093/bioinformatics/btaa046</pub-id><pub-id pub-id-type="medline">31985777</pub-id></nlm-citation></ref><ref id="ref43"><label>43</label><nlm-citation citation-type="journal"><person-group person-group-type="author"><name name-style="western"><surname>Filom</surname><given-names>K</given-names> </name><name name-style="western"><surname>Miroshnikov</surname><given-names>A</given-names> </name><name name-style="western"><surname>Kotsiopoulos</surname><given-names>K</given-names> </name><name name-style="western"><surname>Kannan</surname><given-names>AR</given-names> </name></person-group><article-title>On marginal feature attributions of tree-based models</article-title><source>Found Data Sci</source><year>2024</year><volume>6</volume><issue>4</issue><fpage>395</fpage><lpage>467</lpage><pub-id pub-id-type="doi">10.3934/fods.2024021</pub-id></nlm-citation></ref></ref-list><app-group><supplementary-material id="app1"><label>Multimedia Appendix 1</label><p>Distribution of disease classes and missing values in the dataset.</p><media xlink:href="ojphi_v18i1e84966_app1.docx" xlink:title="DOCX File, 119 KB"/></supplementary-material><supplementary-material id="app2"><label>Multimedia Appendix 2</label><p>Tables presenting statistical analysis, feature importance evaluation scores, and descriptions.</p><media xlink:href="ojphi_v18i1e84966_app2.docx" xlink:title="DOCX File, 68 KB"/></supplementary-material></app-group></back></article>