%Version 3.1 December 2024
% See section 11 of the User Manual for version history
%
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%
%%                                                                 %%
%% Please do not use \input{...} to include other tex files.       %%
%% Submit your LaTeX manuscript as one .tex document.              %%
%%                                                                 %%
%% All additional figures and files should be attached             %%
%% separately and not embedded in the \TeX\ document itself.       %%
%%                                                                 %%
%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%%

%%\documentclass[referee,sn-basic]{sn-jnl}% referee option is meant for double line spacing

%%=======================================================%%
%% to print line numbers in the margin use lineno option %%
%%=======================================================%%

%%\documentclass[lineno,pdflatex,sn-basic]{sn-jnl}% Basic Springer Nature Reference Style/Chemistry Reference Style

%%=========================================================================================%%
%% the documentclass is set to pdflatex as default. You can delete it if not appropriate.  %%
%%=========================================================================================%%

%%\documentclass[sn-basic]{sn-jnl}% Basic Springer Nature Reference Style/Chemistry Reference Style

%%Note: the following reference styles support Namedate and Numbered referencing. By default the style follows the most common style. To switch between the options you can add or remove Numbered in the optional parenthesis. 
%%The option is available for: sn-basic.bst, sn-chicago.bst%  
 
%%\documentclass[pdflatex,sn-nature]{sn-jnl}% Style for submissions to Nature Portfolio journals
%%\documentclass[pdflatex,sn-basic]{sn-jnl}% Basic Springer Nature Reference Style/Chemistry Reference Style
\documentclass[pdflatex,sn-mathphys-num]{sn-jnl}% Math and Physical Sciences Numbered Reference Style
%%\documentclass[pdflatex,sn-mathphys-ay]{sn-jnl}% Math and Physical Sciences Author Year Reference Style
%%\documentclass[pdflatex,sn-aps]{sn-jnl}% American Physical Society (APS) Reference Style
%%\documentclass[pdflatex,sn-vancouver-num]{sn-jnl}% Vancouver Numbered Reference Style
%%\documentclass[pdflatex,sn-vancouver-ay]{sn-jnl}% Vancouver Author Year Reference Style
%%\documentclass[pdflatex,sn-apa]{sn-jnl}% APA Reference Style
%%\documentclass[pdflatex,sn-chicago]{sn-jnl}% Chicago-based Humanities Reference Style

%%%% Standard Packages
%%<additional latex packages if required can be included here>

\usepackage{graphicx}%
\usepackage{multirow}%
\usepackage{amsmath,amssymb,amsfonts}%
\usepackage{amsthm}%
\usepackage{mathrsfs}%
\usepackage[title]{appendix}%
\usepackage{xcolor}%
\usepackage{textcomp}%
\usepackage{manyfoot}%
\usepackage{booktabs}%
\usepackage{algorithm}%
\usepackage{algorithmicx}%
\usepackage{algpseudocode}%
\usepackage{listings}%
%%%%
\usepackage{rotating}
\usepackage{booktabs}
\usepackage{array}
\usepackage{ragged2e}
\usepackage{rotating}
\usepackage{tabularx}
\usepackage{booktabs}
\usepackage{array}

\newcolumntype{L}[1]{>{\raggedright\arraybackslash}p{#1}}
%%%%%=============================================================================%%%%
%%%%  Remarks: This template is provided to aid authors with the preparation
%%%%  of original research articles intended for submission to journals published 
%%%%  by Springer Nature. The guidance has been prepared in partnership with 
%%%%  production teams to conform to Springer Nature technical requirements. 
%%%%  Editorial and presentation requirements differ among journal portfolios and 
%%%%  research disciplines. You may find sections in this template are irrelevant 
%%%%  to your work and are empowered to omit any such section if allowed by the 
%%%%  journal you intend to submit to. The submission guidelines and policies 
%%%%  of the journal take precedence. A detailed User Manual is available in the 
%%%%  template package for technical guidance.
%%%%%=============================================================================%%%%

%% as per the requirement new theorem styles can be included as shown below
\theoremstyle{thmstyleone}%
\newtheorem{theorem}{Theorem}%  meant for continuous numbers
%%\newtheorem{theorem}{Theorem}[section]% meant for sectionwise numbers
%% optional argument [theorem] produces theorem numbering sequence instead of independent numbers for Proposition
\newtheorem{proposition}[theorem]{Proposition}% 
%%\newtheorem{proposition}{Proposition}% to get separate numbers for theorem and proposition etc.

\raggedbottom
%%\unnumbered% uncomment this for unnumbered level heads

\begin{document}

\title[AlamX : A Privacy-Preserving Local AI Framework for Symptom-Based Disease Prediction and Conversational Clinical Triage] {AlamX: A Privacy-Preserving Local AI Framework for Symptom-Based Disease Prediction and Conversational Clinical Triage}

%%=============================================================%%
%% GivenName	-> \fnm{Joergen W.}
%% Particle	-> \spfx{van der} -> surname prefix
%% FamilyName	-> \sur{Ploeg}
%% Suffix	-> \sfx{IV}
%% \author*[1,2]{\fnm{Joergen W.} \spfx{van der} \sur{Ploeg} 
%%  \sfx{IV}}\email{iauthor@gmail.com}
%%=============================================================%%

\author[1,1]{\fnm{MD Hamid } \sur{Alam}}\email{hamidalam763@gmail.com}

\author[2,1]{\fnm{Shubham } \sur{Kumar}}\email{shubhamkumarchoudhary78640@gmail.com}
\equalcont{These authors contributed equally to this work.}

\author[3,1]{\fnm{Rahat Kr} \sur{Pradhan}}\email{man638620@gmail.com}
\equalcont{These authors contributed equally to this work.}

\author[4,1]{\fnm{Saksham} \sur{Pradhan}}\email{sakshampradhan108@gmail.com}
\equalcont{These authors contributed equally to this work.}

\author[5,1]{\fnm{Rohit} \sur{Subba}}\email{rohit_202400005@smit.smu.edu.in}
\equalcont{These authors contributed equally to this work.}

\author*[6,2]{\fnm{Mohana} \sur{S D}}\email{mohan7sdm@gmail.com}
\equalcont{These authors contributed equally to this work.}

\affil[1,2,3,4]{\orgdiv{Department of Computer Science and Engineering- Artificial Intelligence and Machine Learning}, \orgname{Sikkim Manipal Institute of Technology (SMIT)}, \orgaddress{\street{Majitar- Rangpo} \city{Gangtok}, \postcode{737136}, \state{Sikkim,} \country{India}}}


\affil*[6]{\orgdiv{Department of Computer Science and Engineering}, \orgname{Sikkim Manipal Institute of Technology (SMIT)}, \orgaddress{\street{Majitar- Rangpo} \city{Gangtok}, \postcode{737136}, \state{Sikkim,} \country{India}}}

%%==================================%%
%% Sample for unstructured abstract %%
%%==================================%%

%=====================================================================
%  AlamX: A Privacy-Preserving Local AI Framework for Symptom-Based
%  Disease Prediction and Conversational Clinical Triage
%  IEEE conference paper -- compile with pdfLaTeX, then BibTeX, then
%  pdfLaTeX twice (Overleaf does this automatically).
%=====================================================================


\maketitle

%---------------------------------------------------------------- abstract
\begin{abstract} Machine learning has become a standard instrument for preliminary medical triage, yet the prevailing deployment model for consumer-facing diagnostic systems requires that symptoms, vitals and laboratory values be transmitted to remote inference infrastructure. This produces a structural conflict, because the granularity of health data that maximizes predictive utility is precisely the granularity that maximizes disclosure risk. A survey of sixteen studies in symptom-based disease prediction reveals three further limitations that recur independently of the classifier used: systems emit a single unqualified disease label without expressing uncertainty; they treat self-reported symptoms in isolation from physiological vitals, despite published evidence that symptom-only inference has a measurable performance ceiling; and they are demonstrated as notebook, Tkinter or Gradio prototypes without authentication, persistence or a documented application programming interface. This paper presents AlamX, a full-stack clinical intelligence platform in which every analytical component executes on the user's own host. Symptom selections drawn from a standardized 132-parameter clinical vocabulary are encoded into a binary feature vector and classified across 41 disease classes by a Random Forest, trained through a pipeline that applies SMOTE oversampling to the training partition only and uses a tuned XGBoost model as a comparative baseline. The inference service returns a primary prediction, a percentage confidence derived from the ensemble vote distribution, a ranked list of differential signals, a measured inference-time metric and an explicit clinical safety disclaimer. A parallel Health Score engine fuses daily footsteps, heart rate, sleep activity and active logged symptoms into a single longitudinal wellness index, and conversational reasoning is provided by the BioMistral biomedical large language model executed locally through the Ollama runtime with structured health context injected into each prompt. Evaluation of the trained diagnostic engine yields a confusion matrix with a dominant diagonal across the 41-class label space and a gain-based importance analysis in which 130 of the 132 available parameters are used by the boosted model; a worked inference from the deployed interface returns \emph{Fungal infection} at 71\% confidence with \emph{Drug Reaction} surfaced at 27\% as a differential signal. The contribution is architectural rather than a marginal gain in accuracy: AlamX demonstrates that a system can be simultaneously accurate, uncertainty-aware, signal-fusing, conversationally grounded and entirely local.
\end{abstract}

\begin{IEEEkeywords}
symptom-based disease prediction, Random Forest, class imbalance, SMOTE, local large language model, BioMistral, privacy-preserving healthcare, clinical decision support
\end{IEEEkeywords}

%---------------------------------------------------------------- I
\section{Introduction}

\subsection{Background}
Clinical decision support driven by machine learning has moved from academic proposition to deployed practice across screening, risk stratification and triage. Three conditions enabled the transition: the digitization of clinical records, the availability of structured symptom--disease corpora, and the maturation of ensemble methods that behave reliably on high-dimensional, sparse and imbalanced medical data \cite{khalilia2011,elsherbini2023}. In parallel, the locus of health-data generation has shifted decisively toward the individual. Consumer devices now produce continuous ambulatory, cardiac and sleep telemetry, and patients routinely hold digital copies of their own laboratory reports. The analytical question is no longer whether sufficient data exists, but whether it can be interpreted where it is generated.

\subsection{Existing Challenges}
Four challenges recur across the reviewed literature.

\textit{Privacy exposure imposed by architecture.} Systems that centralize medical records or query cloud-hosted language models must transmit highly sensitive personal health information; the requirement follows from the deployment topology rather than from the analytical task \cite{rajora2021,veerababu2025,ram2025}.

\textit{The ceiling of symptom-only inference.} Mohamed \textit{et al.} \cite{mohamed2025} compared four model families across four clinical data modalities and found predictive performance to be strongly context-dependent, reaching an area under the receiver operating characteristic curve of 0.87 where symptoms were combined with electrocardiographic signals but degrading to 0.49 for broad symptom-only data. Most consumer symptom checkers nonetheless ignore vitals entirely.

\textit{Uncalibrated and opaque output.} The majority of reported systems return one disease label \cite{bambal2019,rajashekar2025,mulakala2025,ram2025}. Where symptom presentations overlap, a single label misrepresents genuine diagnostic ambiguity as certainty.

\textit{The prototype--deployment gap.} Implementations are commonly demonstrated through notebook interfaces \cite{mulakala2025}, Tkinter windows \cite{premkumar2025} or Gradio widgets \cite{rajashekar2025}, without authentication, encrypted credential storage, per-user persistence or a documented API contract.

\subsection{Motivation and Objectives}
The reviewed work establishes that symptom-based classification is largely solved as an algorithmic problem, with several independent studies reporting accuracies above 97\% on standardized corpora \cite{fusterpala2024,biradar2024,rajashekar2025}. The unresolved problems are architectural. The objectives of this research are therefore: (i) to implement a fully local platform in which no health data is transmitted beyond the deployment host; (ii) to train and deploy a Random Forest over a 132-parameter symptom vector space across 41 disease classes with SMOTE imbalance correction and XGBoost-based comparative tuning; (iii) to expose inference output as a decision-support artifact with confidence, differentials, latency and a safety disclaimer; (iv) to fuse vitals and symptoms in a composite Health Score; (v) to integrate a locally executed domain-adapted language model with health-context injection; and (vi) to build the system on a production-grade stack.

\subsection{Contributions}
The contributions of this paper are: (C1) an architecture in which discriminative classification, relational persistence and generative language reasoning all execute on the user's own host; (C2) a single-write, multiple-read user-state design in which one symptom log drives both episodic prediction and a continuous Health Score; (C3) an uncertainty-aware inference contract whose four outputs derive from a single probability distribution, so that expressing uncertainty imposes no additional inference cost; (C4) integration of a domain-adapted biomedical language model \cite{labrak2024} under a local runtime, replacing the retrieval-based chat layers used in prior work \cite{biradar2024,veerababu2025}; and (C5) a comparative synthesis of sixteen studies that isolates six architectural research gaps and maps each to an implemented design decision.

%---------------------------------------------------------------- II
\section{Related Work}

Existing research divides into four methodological lines.

\textit{Classical supervised classification over symptom vectors} dominates the field. Rajashekar \cite{rajashekar2025} trained a Support Vector Classifier, a Gaussian Naive Bayes model and a Random Forest on a 4{,}920-record corpus of 132 symptoms and 41 diseases, reporting approximately 99.2\% accuracy for the Random Forest. Fuster-Pal\`a \textit{et al.} \cite{fusterpala2024} applied a three-phase optimization procedure over SVM, Random Forest, K-Nearest Neighbours and artificial neural networks on the same 41-disease space extended with per-symptom severity encoding, obtaining accuracy and $F_1$ above 98\% for K-NN with two neighbours. Premkumar and Nagasundaram \cite{premkumar2025} trained three classifiers separately and displayed all outputs to improve transparency. These studies establish the 132-symptom, 41-disease parameterization as a de facto benchmark and show that accuracy on it is close to saturation.

\textit{Ensemble and imbalance-aware methods} address skewed class distributions. Khalilia \textit{et al.} \cite{khalilia2011} combined a Random Forest with repeated random sub-sampling on the HCUP National Inpatient Sample, outperforming SVM, bagging and boosting on seven of eight chronic disease categories with a mean AUC of 88.79\%; their central methodological finding is that explicit balancing is necessary for reliable minority-class prediction. Rajora \textit{et al.} \cite{rajora2021} proposed a dynamically weighted ensemble that did not improve on the Random Forest alone. Amosa \textit{et al.} \cite{amosa2026} found Random Forest superior to logistic regression and gradient boosting on tabular symptom-and-environment data.

\textit{Deep and representation-learning approaches} deliver their advantage where the input is genuinely multimodal or relational. Chen \textit{et al.} \cite{chen2017} fused structured and unstructured hospital records in a convolutional multimodal risk model, reaching 94.8\% accuracy. Hosseini \textit{et al.} \cite{hosseini2018} modelled MIMIC-III as a heterogeneous information network with metapath-based joint embedding. Rajesh \textit{et al.} \cite{rajesh2023} applied a never-ending image learner on a multi-access edge computing platform. All three require institutional data or infrastructure unavailable to an individual user.

\textit{Conversational systems} lower the barrier to symptom intake. Biradar and Shastri \cite{biradar2024} reported 97.4\% accuracy for an SVM-backed medical chatbot; VeeraBabu \textit{et al.} \cite{veerababu2025} combined fuzzy symptom matching with a transformer-based FAQ module. El-Sherbini \textit{et al.} \cite{elsherbini2023} surveyed machine-learning prediction modelling in primary care and identified external validation, interpretability and workflow integration as the principal outstanding barriers.

Table~\ref{tab:survey} condenses the survey and Table~\ref{tab:gaps} states the resulting gaps.

%---------------------------------------------------------------
% Wide landscape table for Springer journal
%---------------------------------------------------------------
\begin{sidewaystable}[p]
\caption{Comparative Analysis of the Reviewed Literature}
\label{tab:survey}
\centering
\scriptsize

\begin{tabular}{%
>{\centering\arraybackslash}p{0.035\textwidth}
>{\raggedright\arraybackslash}p{0.145\textwidth}
>{\raggedright\arraybackslash}p{0.255\textwidth}
>{\raggedright\arraybackslash}p{0.245\textwidth}
>{\raggedright\arraybackslash}p{0.245\textwidth}
}
\toprule

\textbf{No.} &
\textbf{Author(s), year} &
\textbf{Methodology and technology} &
\textbf{Key findings} &
\textbf{Limitations / research gap} \\

\midrule

1 &
Khalilia \textit{et al.}, 2011 \cite{khalilia2011} &
Random Forest with repeated random sub-sampling; HCUP National Inpatient Sample &
RF performed best on 7 of 8 chronic disease categories, with a mean AUC of 88.79\%. &
Administrative claims rather than patient-reported symptoms; no user-facing system. \\

2 &
Chen \textit{et al.}, 2017 \cite{chen2017} &
CNN-based multimodal risk prediction with latent-factor imputation using hospital records &
Achieved 94.8\% accuracy; multimodal fusion outperformed the unimodal baseline. &
Requires institutional data; region-specific; no clear consumer deployment path. \\

3 &
Hosseini \textit{et al.}, 2018 \cite{hosseini2018} &
Heterogeneous information network with metapath embedding using MIMIC-III &
Outperformed prior EHR embedding baselines and handled missing values. &
Requires rich multi-table EHR data; not directly applicable to self-reported input. \\

4 &
Rajora \textit{et al.}, 2021 \cite{rajora2021} &
K-NN, Random Forest, and Naive Bayes with weighted ensemble voting using NCDC symptom data &
RF achieved 93.65\%, K-NN 93.53\%, NB 84.02\%, while the ensemble achieved 93.65\%. &
Centralized database concentrates privacy risk; ensemble provides no improvement over RF alone. \\

5 &
Bambal and Talmale, 2019 \cite{bambal2019} &
K-NN and Naive Bayes; CNN-based risk prediction using UCI heart disease data &
Naive Bayes achieved 94.5\% accuracy with lower time and memory requirements. &
Single-disease focus; small dataset; no deployment or privacy-handling mechanism. \\

6 &
Rajesh \textit{et al.}, 2023 \cite{rajesh2023} &
Never-ending image learner with multiple-instance learning on edge computing &
Provided automated prediction of common disease attributes. &
Image-centric approach; requires edge infrastructure; source reliability is not controlled. \\

7 &
El-Sherbini \textit{et al.}, 2023 \cite{elsherbini2023} &
Review of machine learning applications in pre-operative care and primary-care screening &
ML improves early screening and pre-symptomatic risk stratification. &
Weak external validation, limited interpretability, and integration barriers. \\

8 &
Fuster-Pal\`a \textit{et al.}, 2024 \cite{fusterpala2024} &
Three-phase optimization over SVM, RF, K-NN, and ANN for 41 diseases with severity levels from 1--7 &
K-NN ($k=2$) exceeded 98\% accuracy and achieved a high $F_1$ score. &
Synthetic and non-clinical data; no deployed interface; lacks an explanation layer. \\

9 &
Biradar and Shastri, 2024 \cite{biradar2024} &
NLP chatbot with SVM classifier for labelled symptom descriptions &
Achieved 97.4\% accuracy and enabled immediate conversational access. &
Keyword- and database-driven responses; single-classifier design; no local model. \\

10 &
Rajashekar, 2025 \cite{rajashekar2025} &
SVC, Gaussian NB, and RF using 4,920 records, 132 symptoms, and 41 diseases; Gradio interface &
RF achieved approximately 99.2\%, SVC approximately 97.9\%, and GNB approximately 94.8\%. &
Binary encoding ignores symptom severity; lacks confidence estimates and differential diagnosis; demonstration system only. \\

11 &
Premkumar and Nagasundaram, 2025 \cite{premkumar2025} &
RF, Decision Tree, and Naive Bayes presented comparatively through a Tkinter GUI &
Multi-model presentation improves user-facing transparency. &
Desktop GUI; no authentication or encryption; lacks calibrated probability estimates. \\

12 &
VeeraBabu \textit{et al.}, 2025 \cite{veerababu2025} &
Naive Bayes with fuzzy matching and spell correction; DistilBERT-based FAQ module &
Provides conversational symptom intake with automatic spelling correction. &
No quantitative evaluation reported; retrieval-based approach; cloud dependency. \\

13 &
Mulakala \textit{et al.}, 2025 \cite{mulakala2025} &
Disease-specific ML and CNN models integrated into a single Streamlit interface &
Provides one interface for multiple disease prediction tasks. &
Siloed one-model-per-disease architecture; lacks a unified symptom representation. \\

14 &
Ram \textit{et al.}, 2025 \cite{ram2025} &
Data cleaning, encoding, and balancing followed by comparative ML; 776 records and 90 diseases &
RF achieved approximately 92\%; checker, document analysis, and reminder functions were integrated. &
Very small dataset for 90 classes; web-hosted architecture; no local model. \\

15 &
Mohamed \textit{et al.}, 2025 \cite{mohamed2025} &
FNN, XGBoost, RF, and SVM evaluated across four clinical data modalities &
AUC was 0.87 for symptoms + ECG, 0.84 for history, 0.62 for laboratory data, and 0.49 for symptoms alone. &
No multimodal fusion architecture and no deployed system. \\

16 &
Amosa \textit{et al.}, 2026 \cite{amosa2026} &
Logistic Regression, RF, and Gradient Boosting evaluated on 500 tabular samples &
RF achieved the highest performance at 94.6\%; environmental and age variables were dominant predictors. &
Veterinary domain; small sample size; no transferable human-health triage system. \\

\bottomrule
\end{tabular}

\end{sidewaystable}


\begin{table*}[t]
\caption{Research Gaps and the Corresponding AlamX Design Response}
\label{tab:gaps}
\centering
\footnotesize

\begin{tabularx}{\textwidth}{@{}L{2.8cm} X X L{4.5cm}@{}}

\toprule

\textbf{Gap} &
\textbf{Description} &
\textbf{Evidence} &
\textbf{AlamX Response} \\

\midrule

G1 Privacy &
No local system combines ML inference and language reasoning &
\cite{rajora2021,veerababu2025,ram2025} &
ML, database, and LLM run locally \\

G2 Symptom--vitals fusion &
Symptoms and vitals are handled separately &
\cite{mohamed2025,chen2017} &
Shared symptom log for classifier and Health Score \\

G3 Uncertainty-aware output &
No confidence, differentials, or latency information &
\cite{bambal2019,rajashekar2025,mulakala2025,ram2025} &
Class, confidence, differentials, and latency from one distribution \\

G4 Conversational grounding &
Keyword, fuzzy-match, or FAQ-based chat systems &
\cite{biradar2024,veerababu2025} &
BioMistral with structured health-context injection \\

G5 Deployment engineering &
Limited authentication, security, sessions, and API design &
\cite{rajashekar2025,premkumar2025,mulakala2025} &
bcrypt, JWT, FastAPI contracts, and relational storage \\

G6 Longitudinal modelling &
No persistent index combining vitals and symptoms &
Absent across the corpus &
Health Score combines vitals and active symptoms \\

\bottomrule

\end{tabularx}
\end{table*}

%---------------------------------------------------------------- III
\section{Problem Statement}

Existing symptom-based disease prediction systems achieve high classification accuracy but transmit sensitive health data to centralized or cloud infrastructure, return single unqualified disease labels without confidence or differential alternatives, and treat symptoms in isolation from physiological vitals. The affected users are individuals deciding whether and when to seek care, particularly in geographically remote or resource-constrained settings where a wrong decision carries the greatest cost. The technical challenges are the combination of local execution with acceptable inference latency, the correction of class imbalance without leaking synthetic data into evaluation, and the grounding of a generative model in an individual's own recorded values without external transmission.

%---------------------------------------------------------------- IV
\section{Proposed System: AlamX}

AlamX is a decoupled two-tier application in which a React single-page client communicates over a CORS-governed HTTP interface with a FastAPI service, and in which the relational store, the trained classifier and the language model all reside on the deployment host. Two analytical pathways share one data substrate.

The \emph{diagnostic pathway} is episodic: the user selects active symptoms from a searchable catalogue of 132 standardized clinical parameters; the backend assembles a 132-dimensional binary vector; the Random Forest performs inference; and the service returns a primary prediction, a percentage confidence, ranked differential signals, a measured latency and a clinical safety disclaimer.

The \emph{longitudinal pathway} is continuous: daily footsteps, heart rate and sleep activity are recorded alongside the active symptom set, and the Health Score engine aggregates these into a single interpretable wellness index rendered with its constituent contributions.

Both pathways feed the \emph{conversational layer}, which assembles structured health context---blood pressure, heart rate, recorded report values and, where relevant, the most recent prediction---and injects it into the prompt supplied to the local language model. Table~\ref{tab:modules} lists the modules.

%---------------------------------------------------------------- V
\section{System Architecture}

AlamX implements four cooperating tiers, all resident on the local host (Fig.~\ref{fig:arch}).

\begin{figure}[!t]
\centering
\includegraphics[width=\columnwidth]{fig_architecture.png}
\caption{AlamX four-tier system architecture. The React and Vite presentation tier communicates over a CORS-governed HTTP interface with the FastAPI application tier, which fans out to the SQLite persistence tier, the Random Forest inference tier and the Ollama/BioMistral language-model tier. All four tiers execute on the deployment host.}
\label{fig:arch}
\end{figure}

\subsection{Presentation Layer}
A React application scaffolded and served by Vite on port 5173. Client-side routing is provided by \texttt{react-router-dom}; styling by Tailwind CSS with variant logic through class-variance-authority; composition primitives by Radix UI slots; iconography by \texttt{lucide-react}. The application renders inside a constrained ``PhoneShell'' container that emulates a mobile viewport on desktop displays, enforcing a single responsive design target. Session state is persisted in browser \texttt{localStorage}.

\subsection{Application Layer}
A FastAPI service executed by the Uvicorn ASGI worker on port 8000, selected for asynchronous request handling, Pydantic-based schema validation at the API boundary and automatically generated OpenAPI documentation. The surface is organized into five routers---authentication, vitals, symptoms, prediction and chat---each mounted behind the same middleware chain.

\subsection{Data Layer}
SQLite provides file-backed relational persistence. For a single-host deployment this supplies full SQL semantics and transactional integrity without introducing a network-exposed database daemon that would itself constitute an attack surface. A single user entity owns many vitals records, symptom logs, predictions and chat-history entries, so authorization is enforced uniformly by filtering on the authenticated subject.

\subsection{Inference and Language-Model Layers}
A serialized Random Forest \cite{breiman2001} implemented with Scikit-Learn \cite{pedregosa2011} is loaded once at startup, so per-request cost is a single forward pass. The Ollama runtime hosts BioMistral \cite{labrak2024}, exposing a local HTTP inference endpoint consumed by the chat router as an internal service.

Fig.~\ref{fig:dfd} shows the level-1 data flow. Its significant property is that the symptom log is written once and consumed twice---by the classifier for episodic prediction and by the Health Score engine for longitudinal scoring---which is how episodic and continuous signals are fused without duplicating user effort.

\begin{figure}[!t]
\centering
\includegraphics[width=\columnwidth]{fig_dataflow.png}
\caption{AlamX level-1 data flow diagram. Vitals, symptom selections and chat queries enter through one API surface and diverge into the Health Score path, the prediction path and the local language-model path, all reading from and writing to the same local store.}
\label{fig:dfd}
\end{figure}

%---------------------------------------------------------------- VI
\section{Methodology}

\subsection{Data Collection and Structuring}
The symptom--disease corpus follows the standardized 132-parameter vocabulary mapped to 41 disease classes established in the literature \cite{fusterpala2024,rajashekar2025}. Runtime user data is collected through the application interface and persisted locally, scoped to the authenticated user. The exact dataset source and record count are not specified in the provided project documentation.

\subsection{Preprocessing and Encoding}
Ingestion and structuring are performed with pandas. Column integrity is validated against the 132-parameter schema, duplicate and malformed records are removed, missing indicators are resolved to explicit absence, and disease labels are encoded into the 41-class target space. Each record is represented as
\begin{equation}
\mathbf{x} \in \{0,1\}^{132}, \qquad
x_i =
\begin{cases}
1, & \text{if clinical parameter } i \text{ is present}\\
0, & \text{otherwise.}
\end{cases}
\end{equation}

\subsection{Class Imbalance Correction}
SMOTE \cite{chawla2002} is applied to the training partition only, synthesizing minority-class examples by interpolation between existing minority neighbours in feature space. Restricting resampling to the training partition prevents synthetic information from leaking into evaluation. This follows the finding of Khalilia \textit{et al.} \cite{khalilia2011} that explicit imbalance correction is necessary for reliable minority-class prediction.

\subsection{Model Selection and Training}
A Random Forest \cite{breiman2001} is trained as the deployment model, selected for its suitability to sparse high-dimensional binary data, its capture of conjunctive symptom interactions without explicit feature crossing, its intrinsic probability estimation through the vote distribution, its variance reduction through bootstrap aggregation, and its gain-based feature importance. An XGBoost model \cite{chen2016} is trained in parallel under hyperparameter search as a competitive baseline and sensitivity analysis. The final hyperparameter values for both models are not specified in the provided project documentation. Fig.~\ref{fig:pipeline} separates the offline training phase from the deployment phase.

\begin{figure}[!t]
\centering
\includegraphics[width=\columnwidth]{fig_pipeline.png}
\caption{Machine learning training and deployment pipeline. Class balancing, model comparison and hyperparameter exploration occur strictly within the offline training phase; at inference time the system performs a single forward pass through the fitted ensemble.}
\label{fig:pipeline}
\end{figure}

\subsection{Output Generation}
For the class probability distribution $P$ over the 41 classes returned by the ensemble, the deployed service computes
\begin{equation}
d^{*} = \arg\max_{d} P(d), \qquad
c = 100 \cdot \max_{d} P(d),
\end{equation}
\begin{equation}
D = \operatorname*{top-}k \; \{ P(d) : d \neq d^{*} \},
\end{equation}
where $d^{*}$ is the primary prediction, $c$ the percentage confidence and $D$ the ranked differential signals. All three are derived from one distribution, which is why uncertainty-aware output imposes no additional inference cost. The measured call latency is returned alongside them, together with the clinical safety disclaimer.

%---------------------------------------------------------------- VII
\section{Implementation}

Table~\ref{tab:stack} summarizes the implemented stack. The authentication module hashes the submitted password with bcrypt through passlib before the user row is committed and discards the plaintext immediately; on login the credential is verified against the stored hash and a signed JSON Web Token is issued and persisted in \texttt{localStorage}. The symptom checker renders the 132-parameter catalogue as an incremental search with removable chips, because a flat form of 132 checkboxes is technically equivalent and practically unusable (Fig.~\ref{fig:ui}). The chat router assembles structured health context and dispatches the constructed prompt to BioMistral through Ollama (Fig.~\ref{fig:llm}).

\begin{figure}[!t]
\centering
\includegraphics[width=\columnwidth]{fig_llm.png}
\caption{Local language-model architecture. The chat router assembles structured health context from the local database, constructs a prompt and dispatches it to BioMistral under the Ollama runtime; the entire generation path is contained within the local host boundary.}
\label{fig:llm}
\end{figure}

\begin{table}[!t]
\caption{Implemented Technology Stack}
\label{tab:stack}
\centering
\footnotesize
\begin{tabularx}{\columnwidth}{@{}lX@{}}
\toprule
\textbf{Layer} & \textbf{Technology and purpose}\\
\midrule
Presentation & React with Vite (port 5173); react-router-dom; Tailwind CSS with class-variance-authority; Radix UI slots; lucide-react; localStorage session state\\
Application & Python, FastAPI served by Uvicorn (port 8000); Pydantic validation; CORS origin allow-list\\
Data & SQLite: users, vitals, symptom logs, predictions, chat history\\
Security & passlib with bcrypt; signed JSON Web Tokens\\
Machine learning & Scikit-Learn Random Forest; XGBoost baseline; imbalanced-learn SMOTE; pandas, NumPy\\
Language model & Ollama runtime hosting BioMistral\\
\bottomrule
\end{tabularx}
\end{table}

\begin{table}[!t]
\caption{System Modules}
\label{tab:modules}
\centering
\footnotesize
\begin{tabularx}{\columnwidth}{@{}lXX@{}}
\toprule
\textbf{Module} & \textbf{Input} & \textbf{Output}\\
\midrule
Authentication & Username, password & bcrypt hash persisted; signed JWT\\
Home dashboard & Footsteps, heart rate, sleep, active symptoms & Health Score with component breakdown\\
Symptom checker & Search text; symptom selections & Symptom log; 132-dimensional binary vector\\
Prediction & 132-dimensional binary vector & Primary class, confidence, differentials, latency, disclaimer\\
Clinical chat & User message; stored vitals and report values & Context-grounded natural-language response\\
Vitals & Footsteps, heart rate, sleep, blood pressure & Persisted user-scoped vitals records\\
\bottomrule
\end{tabularx}
\end{table}

%---------------------------------------------------------------- VIII
\section{Results and Evaluation}

Two evaluation artifacts were produced from the trained diagnostic engine, together with a functional demonstration of the deployed application.

\subsection{Classification Behaviour Across the 41-Class Label Space}
Fig.~\ref{fig:cm} presents the confusion matrix of the AlamX diagnostic engine over the full 41-class label space. The matrix exhibits a dominant diagonal with negligible off-diagonal mass: for each of the 41 conditions, predictions concentrate on the true class, and no systematic confusion cluster is visible between clinically adjacent conditions. Per-class support is approximately uniform, the expected consequence of balancing the training distribution before the stratified split. The scalar accuracy, macro precision, macro recall and macro $F_1$ corresponding to this matrix are not specified in the provided project documentation and are therefore not reported here; the matrix is presented as evidence of classification behaviour rather than as a substitute for those statistics.


\begin{figure}[!t]
\centering
\includegraphics[width=\columnwidth]{fig_confusion_matrix.jpg}
\caption{Confusion matrix of the AlamX diagnostic engine over the 41-class label space. Predictions concentrate on the diagonal, with negligible off-diagonal mass and approximately uniform per-class support.}
\label{fig:cm}
\end{figure}

Let the user's prior score before logging be
\[
S_0 = \frac{N}{D},
\]
where $N$ is the accumulated points earned and $D$ is the sum of maximum available
points across measured components.

When logging an activity with marginal denominator $\Delta D > 0$ and marginal earned
points $\Delta N \le \Delta D$, the updated score is
\[
S_1 = \frac{N + \Delta N}{D + \Delta D}.
\]

For logging an activity to \emph{strictly decrease} the score ($S_1 < S_0$):
\[
\frac{N + \Delta N}{D + \Delta D} < \frac{N}{D}.
\]

Since $D > 0$ and $D + \Delta D > 0$, cross-multiplication preserves the inequality:
\begin{align}
D(N + \Delta N) &< N(D + \Delta D) \\
D N + D\,\Delta N &< N D + N\,\Delta D \\
D\,\Delta N &< N\,\Delta D \\
\implies \quad \frac{\Delta N}{\Delta D} &< \frac{N}{D} \;=\; S_0 .
\end{align}

\paragraph{Implication.}
A score drops whenever the marginal efficiency $\Delta N / \Delta D$ of the new entry is
strictly less than the user's prior baseline score $S_0$. For example, if a user has a
pristine baseline of $S_0 = 0.90$ ($90\%$) and logs $6$ of $10$ minutes
($\Delta D = 7$, $\Delta N = 4.2$), the marginal efficiency is
\[
\frac{4.2}{7} = 0.60 .
\]
Because $0.60 < 0.90$, completing a healthy action \emph{actively degrades} the score.

\paragraph{Resolution.}
We isolate engagement into an additive pool whose denominators are invariant at runtime:
\[
S = \text{Clinical}_{(\le 80)} + \text{Engagement}_{(\le 20)} .
\]
Here, logging an activity yields $\Delta N \ge 0$ with $\Delta D = 0$.

\section{Variance Collision versus \texorpdfstring{$\min$}{min}-Capped Attainment}

A statistical variance penalty attempts to penalise macro imbalance via
\[
\text{Penalty} = -\lambda \cdot \operatorname{Var}\!\left(
\frac{\text{protein}}{T_p},\;
\frac{\text{carbs}}{T_c},\;
\frac{\text{fat}}{T_f}
\right).
\]

If a user reaches $100\%$ of carbohydrates but only $10\%$ of protein, consuming more
carbohydrates \emph{increases} the variance. When
\[
\left\lvert \frac{\partial\,\text{Penalty}}{\partial\,\text{carbs}} \right\rvert
> \frac{\partial\,\text{Credit}}{\partial\,\text{carbs}},
\]
the net derivative satisfies $\dfrac{\partial S}{\partial\,\text{carbs}} < 0$, meaning
that eating food lowers the score and breaks monotonicity.

\paragraph{Resolution.}
We instead define
\[
\text{Balance} = \BP \times
\min\!\Big(
\min\big(1.0, \tfrac{\text{protein}}{T_p}\big),\;
\min\big(1.0, \tfrac{\text{carbs}}{T_c}\big),\;
\min\big(1.0, \tfrac{\text{fat}}{T_f}\big)
\Big).
\]

\paragraph{Mathematical property.}
The pointwise minimum of any set of non-decreasing functions is non-decreasing. Because
each single-macro attainment curve
\[
f_i(m_i) = \min\!\left(1.0, \frac{m_i}{T_i}\right)
\]
has a non-negative subgradient everywhere, $\min_i f_i$ cannot decrease with additional
intake.

\paragraph{Information-collapse trade-off.}
\begin{itemize}[leftmargin=2em]
  \item \textbf{Day A:} Protein $20\%$, Carbs $20\%$, Fat $20\%$ (balanced undereating).
        $\min = 0.20 \implies \text{Balance} = 3 \times 0.20 = 0.6$ pts.
  \item \textbf{Day B:} Protein $20\%$, Carbs $250\%$, Fat $180\%$ (heavy carbohydrate
        surplus). $\min = 0.20 \implies \text{Balance} = 3 \times 0.20 = 0.6$ pts.
\end{itemize}

\paragraph{Justification.}
Day B still earns higher overall nutrition points through the separate per-macro
attainment pool (capped at targets). The balance term functions as a bottleneck bonus
based on Liebig's Law of the Minimum: it withholds unearned bonus points rather than
actively deducting points, safeguarding behavioural motivation and algorithmic
predictability.

\section{Defending the Non-Decreasing Property and Property Testing}

The docstring assertion holds in the non-strict mathematical sense ($\le$). Every
calculation in \texttt{nutrition.py} is an affine combination with non-negative
coefficients, identity functions, and pointwise minima. The subgradient with respect to
any macronutrient $m_i$ satisfies
\[
\frac{\partial S}{\partial m_i} \ge 0
\qquad \forall\, m_i \in [0, \infty).
\]
Once a specific macro target is met, its marginal contribution drops to $0$, but never
becomes negative.

\subsection*{4{,}000-Check Property Test Architecture}

\paragraph{Generator.}
Uniformly samples random biological vectors across valid bounds:
\[
\text{Steps} \in [0,\, 35000], \quad
\text{Water} \in [0,\, 6000\,\text{ml}], \quad
\text{Macros} \in [0,\, 500\,\text{g}], \quad
\text{Sleep} \in [0,\, 16\,\text{h}].
\]
For each state vector $\bm{x}$, it generates a non-negative perturbation vector
$\bm{\Delta x} \ge \bm{0}$.

\paragraph{Oracle predicate.}
Evaluates the score before and after the perturbation:
\[
\texttt{assert } \operatorname{score}(\bm{x} + \bm{\Delta x}) \ge \operatorname{score}(\bm{x}).
\]

\paragraph{Result.}
Over $4{,}000$ iterations---including edge cases at zero, exact target limits, and
extreme overages---zero regressions occurred.

\subsection{Feature Importance}
Fig.~\ref{fig:fi} reports the twenty most influential symptom parameters ranked by average information gain per split. Two observations follow. First, 130 of the 132 available parameters are used by the boosted model, indicating that the standardized vocabulary is close to fully informative for this label space and that aggressive feature pruning would be unjustified. Second, the gain distribution is markedly non-uniform: \emph{swollen extremities} (387) and \emph{swollen blood vessels} (342) dominate, followed by \emph{stomach bleeding} (220) and \emph{increased appetite} (154), while the remaining parameters in the top twenty cluster between 70 and 143. Highly specific, low-prevalence findings therefore carry disproportionate discriminative weight, which is consistent with clinical reasoning: a rare and specific sign narrows the differential far more sharply than a common and non-specific one.

\begin{figure}[!t]
\centering
\includegraphics[width=\columnwidth]{fig_feature_importance.jpg}
\caption{The twenty most influential symptom parameters ranked by average information gain per split; 130 of the 132 available parameters are used by the boosted model.}
\label{fig:fi}
\end{figure}

\subsection{Functional Demonstration}
Fig.~\ref{fig:ui} shows an end-to-end inference in the deployed application. Two symptoms---\emph{skin rash} and \emph{itching}---were selected from the catalogue and submitted. The system returned \emph{Fungal infection} as the primary prediction at 71\% model confidence, annotated as based on two symptoms analysed, together with a plain-language description of the condition and \emph{Drug Reaction} at 27\% under a heading describing lower-probability matches from the same symptom set.

The example is diagnostically instructive. Skin rash and itching are shared by both conditions, so the classifier is genuinely uncertain; the interface communicates that uncertainty rather than suppressing it, which is the behaviour the differential-signal design was intended to produce. It also illustrates the effect of sparse input: with two of 132 parameters set, confidence is moderate rather than high, and the interface states the number of symptoms the prediction rests on.

\begin{figure}[!t]
\centering
\includegraphics[width=0.46\columnwidth]{ui_symptom_checker.png}\hfill
\includegraphics[width=0.46\columnwidth]{ui_results.png}
\caption{Deployed AlamX interface. Left: symptom checker with two clinical parameters selected as removable chips. Right: diagnostic results view returning the primary prediction, model confidence, a description of the condition and a lower-probability differential signal.}
\label{fig:ui}
\end{figure}

\subsection{Evaluation Not Yet Performed}
The following were not specified in the provided project documentation and remain future work: quantitative classification metrics; a no-SMOTE ablation quantifying the change in macro recall; mean and variance of inference latency; the hardware environment; the parameter size, quantization and mean response latency of the BioMistral deployment; and any user study or clinical validation. No such figures are asserted in this paper.

%---------------------------------------------------------------- IX
\section{Security and Privacy}

The security posture comprises two bands. \emph{Application security controls}: credentials are never stored in recoverable form, hashing being performed with bcrypt through the passlib context, whose adaptive work factor allows cost to be raised as commodity hardware improves and whose per-password salt defeats precomputed-hash attacks; authenticated sessions are signed JSON Web Tokens carrying subject and expiry claims, resolved through a FastAPI dependency so that authorization is enforced declaratively before any handler logic executes; because every record type descends from the user entity, data scoping is enforced uniformly by filtering on the token subject; and cross-origin requests are governed by an explicit origin allow-list rather than a wildcard.

\emph{Data-locality controls}: the database file, the serialized model artifact and the language-model weights all reside on the deployment host, and no component requires an outbound network connection to function, so no protected health information is transmitted to any third party. These are architectural controls. No claim of absolute security, of formal compliance with any regulatory framework, or of any security certification is made, and none is supported by the provided project documentation.

%---------------------------------------------------------------- X
\section{Discussion}

AlamX addresses a problem that is architectural rather than algorithmic. The reviewed literature shows accuracy on the standardized symptom corpus to be close to saturation \cite{fusterpala2024,biradar2024,rajashekar2025}; the differentiating question is therefore not whether a further fraction of a percentage point can be obtained, but whether an accurate classifier can be delivered under conditions that make it usable and trustworthy.

Three advantages follow from local execution. Confidentiality becomes a property of the deployment topology. Richer health context improves the assistant's reasoning without any corresponding increase in disclosure risk, inverting the customary privacy--utility trade-off, since a cloud-hosted equivalent would transmit a complete clinical profile on every chat message. And the platform remains functional without connectivity and incurs no per-token cost, which matters in exactly the intermittently connected settings where triage support is most valuable.

The trade-offs are real. Local execution transfers a hardware requirement to the user, since a quantized biomedical language model must fit in host memory. Binary symptom encoding discards severity and duration, which the severity-encoded variant of this corpus \cite{fusterpala2024} shows to be informative. The 41-class label space bounds the conditions the system can name at all. The Health Score weighting is heuristic rather than empirically derived. Finally, the corpus is a curated symptom--disease matrix rather than real clinical data, a limitation its own authors note \cite{fusterpala2024}, so reported behaviour should not be read as clinical validation.

%---------------------------------------------------------------- XI
\section{Future Scope}

Direct ingestion from consumer wearables would replace manual vitals entry with continuous physiological telemetry, raising the temporal resolution of the Health Score. Time-series modelling over accumulated vitals and symptom logs would support detection of gradual deterioration against the individual's own baseline. Coupling the local language model to a locally indexed corpus of clinical guidelines would ground generated responses in citable sources while preserving offline execution. Extending beyond 41 classes and replacing binary indicators with graded severity and duration would improve discrimination among clinically adjacent presentations. Formal probability calibration and validation against an independent real-world clinical dataset would establish whether reported confidence values are trustworthy as probabilities. Reporting the quantitative metrics listed in Section~VIII-D is the immediate next step.

%---------------------------------------------------------------- XII
\section{Conclusion}

This paper addressed the architectural limitations of existing symptom-based disease prediction systems: centralized handling of sensitive health data, single-label output that suppresses genuine diagnostic ambiguity, isolation of symptom data from physiological vitals, and demonstration-grade engineering. AlamX was proposed and implemented as a fully local clinical intelligence platform in which a Random Forest over a 132-parameter symptom vector space across 41 disease classes---trained with SMOTE imbalance correction and an XGBoost comparative baseline---is served through a FastAPI backend to a React client, with conversational reasoning provided by BioMistral executed locally under Ollama. Evaluation of the diagnostic engine produced a confusion matrix with a dominant diagonal across all 41 classes and a gain-based importance analysis showing 130 of 132 parameters in active use; a worked inference returned a primary prediction at 71\% confidence together with a differential signal at 27\%, demonstrating the uncertainty-aware output contract in operation. Quantitative classification metrics remain to be reported and constitute the immediate next step. The significance of the work lies in demonstrating that privacy-preserving local execution and clinically useful analytical capability are not competing objectives.

%---------------------------------------------------------------- ack
% \section*{Acknowledgment}
%The authors thank Dr. Mohana S D, Assistant Professor-II, Department of Computer Science and Engineering, Sikkim Manipal Institute of Technology, for supervision and guidance throughout this work.

%----------------------------------------------------------------
% References
%----------------------------------------------------------------
%------------------------------------------------
% References
%------------------------------------------------
\begin{thebibliography}{99}

\bibitem{khalilia2011}
M. Khalilia, S. Chakraborty, and M. Popescu,
``Predicting disease risks from highly imbalanced data using random forest,''
\textit{BMC Medical Informatics and Decision Making},
vol. 11, no. 51, 2011.

\bibitem{chen2017}
M. Chen, Y. Hao, K. Hwang, L. Wang, and L. Wang,
``Disease prediction by machine learning over big data from healthcare communities,''
\textit{IEEE Access},
vol. 5, pp. 8869--8879, 2017.

\bibitem{hosseini2018}
A. Hosseini, T. Chen, W. Wu, Y. Sun, and M. Sarrafzadeh,
``HeteroMed: Heterogeneous information network for medical diagnosis,''
in \textit{Proc. 27th ACM Int. Conf. on Information and Knowledge Management (CIKM '18)},
Torino, Italy, pp. 763--772, 2018.

\bibitem{rajora2021}
H. Rajora, N. S. Punn, S. K. Sonbhadra, and S. Agarwal,
``Machine learning equipped web based disease prediction and recommender system,''
\textit{arXiv preprint arXiv:2106.02813}, 2021.

\bibitem{bambal2019}
J. C. Bambal and R. B. Talmale,
``Designing a disease prediction model using machine learning,''
\textit{IOSR Journal of Engineering (ICIREST-19)},
pp. 27--32, 2019.

\bibitem{rajesh2023}
E. Rajesh, S. Basheer, R. K. Dhanaraj, S. Yadav, S. Kadry,
M. A. Khan, Y. J. Kim, and J.-H. Cha,
``Machine learning for online automatic prediction of common disease attributes using never-ending image learner,''
\textit{Diagnostics},
vol. 13, no. 1, p. 95, 2023.

\bibitem{elsherbini2023}
A. H. El-Sherbini, H. U. H. Virk, Z. Wang, B. S. Glicksberg, and C. Krittanawong,
``Machine-learning-based prediction modelling in primary care: State-of-the-art review,''
\textit{AI},
vol. 4, no. 2, pp. 437--460, 2023.

\bibitem{fusterpala2024}
A. Fuster-Pal{\`a}, F. Luna-Perej{\'o}n, L. Mir{\'o}-Amarante,
and M. Dom{\'i}nguez-Morales,
``Optimized machine learning classifiers for symptom-based disease screening,''
\textit{Computers},
vol. 13, no. 9, p. 233, 2024.

\bibitem{biradar2024}
S. Biradar and S. Shastri,
``Medical chatbot: AI based infectious disease prediction model,''
\textit{Journal of Scientific Research and Technology},
vol. 2, no. 10, pp. 1--12, 2024.

\bibitem{rajashekar2025}
B. Rajashekar,
``A multi-disease prediction system using hybrid machine learning on symptom inputs,''
\textit{International Journal of Creative Research Thoughts},
vol. 13, no. 8, 2025.

\bibitem{premkumar2025}
S. Premkumar and S. Nagasundaram,
``Symptoms based disease prediction using machine learning techniques,''
\textit{International Journal of Science, Engineering and Technology},
vol. 13, no. 3, 2025.

\bibitem{veerababu2025}
M. VeeraBabu, C. H. Sri Divya, P. A. Anirudh Hruthen,
R. Poornima, Jagadeesh, and K. S. M. K. Vinay,
``Sympsis: A healthcare assistant for symptom diagnosis and disease prediction,''
\textit{Journal of Emerging Technologies and Innovative Research},
vol. 12, no. 4, 2025.

\bibitem{mulakala2025}
S. V. Mulakala, G. Neeharika, P. Vinay Kumar,
and A. Bhargava Kiran,
``Chronic diseases prediction using machine learning,''
\textit{arXiv preprint arXiv:2502.10481}, 2025.

\bibitem{ram2025}
A. Ram, B. Senapati, S. K. Sinha, B. Mahto, and K. Amrendra,
``Health-Mate: An AI-powered smart health consultant for early disease prediction and lifestyle assistance,''
\textit{Journal of Emerging Technologies and Innovative Research},
vol. 12, no. 11, 2025.

\bibitem{mohamed2025}
A. Mohamed, M. Abdelrehim, and R. Al-Barazie,
``Context matters in machine learning based disease prediction with insights from diverse clinical and symptom data,''
\textit{Scientific Reports},
vol. 15, p. 26855, 2025.

\bibitem{amosa2026}
B. M. G. Amosa, N. C. Onyeka, A. O. Fabiyi, A. E. Fasoro,
and O. I. Adigun,
``Development of a predictive model for fowl-cholera infection status in poultry using advanced data mining analysis techniques and logistic regression modeling,''
\textit{International Journal of Latest Technology in Engineering, Management and Applied Science},
vol. 15, no. 4, 2026.

\bibitem{chawla2002}
N. V. Chawla, K. W. Bowyer, L. O. Hall, and W. P. Kegelmeyer,
``SMOTE: Synthetic minority over-sampling technique,''
\textit{Journal of Artificial Intelligence Research},
vol. 16, pp. 321--357, 2002.

\bibitem{breiman2001}
L. Breiman,
``Random forests,''
\textit{Machine Learning},
vol. 45, no. 1, pp. 5--32, 2001.

\bibitem{chen2016}
T. Chen and C. Guestrin,
``XGBoost: A scalable tree boosting system,''
in \textit{Proc. 22nd ACM SIGKDD Int. Conf. on Knowledge Discovery and Data Mining},
San Francisco, CA, USA, pp. 785--794, 2016.

\bibitem{labrak2024}
Y. Labrak, A. Bazoge, E. Morin, P.-A. Gourraud,
M. Rouvier, and R. Dufour,
``BioMistral: A collection of open-source pretrained large language models for medical domains,''
in \textit{Findings of the Association for Computational Linguistics: ACL 2024},
2024.

\bibitem{pedregosa2011}
F. Pedregosa \textit{et al.},
``Scikit-learn: Machine learning in Python,''
\textit{Journal of Machine Learning Research},
vol. 12, pp. 2825--2830, 2011.

\end{thebibliography}

\end{document}
