Publications
Refereed Journal Articles
- Sullivan, S. S., Hewner, S., Casucci, S., Bowen, E., Chandola, V., Sheehan, A. M., Anderson, A. J., Gabello, J., & Noyes, K. (2026). Balancing fidelity and flexibility: a case study presentation of an augmented dynamic adaptation process for socio-technical innovations in healthcare. Front Health Serv, 6, 1737047.
@article{Sullivan2026, author = {Sullivan, Suzanne S and Hewner, Sharon and Casucci, Sabrina and Bowen, Elizabeth and Chandola, Varun and Sheehan, Amy M and Anderson, Amanda J and Gabello, Jarod and Noyes, Katia}, journal = {Front Health Serv}, pages = {1737047}, title = {Balancing fidelity and flexibility: a case study presentation of an augmented dynamic adaptation process for socio-technical innovations in healthcare.}, volume = {6}, year = {2026} } - Castillo, R., Okumus, P., Khorasani, N. E., & Chandola, V. (2025). Shear Condition Classification of Cracked Reinforced Concrete Beams Using Machine Learning. Journal of Bridge Engineering, 30(7), 04025040.
@article{Castillo2025, author = {Castillo, Rodrigo and Okumus, Pinar and Khorasani, Negar Elhami and Chandola, Varun}, title = {Shear Condition Classification of Cracked Reinforced Concrete Beams Using Machine Learning}, journal = {Journal of Bridge Engineering}, volume = {30}, number = {7}, pages = {04025040}, year = {2025}, doi = {10.1061/JBENF2.BEENG-7290} }RC bridges represent about 40% of the US bridge inventory, with many of these bridges reaching or surpassing their design service life. As a result, there is a significant number of structures that require fast and accurate structural evaluation. Shear deficiencies can pose a higher safety risk than flexure deficiencies since shear failures are sudden. This study correlates shear crack width with shear condition and proposes a machine-learning framework to place RC beams into shear condition categories using quantitative estimates of shear, stiffness, and stirrup strain histories. The results of the proposed framework are compared with those from existing quantitative and qualitative assessment methodologies. The quantitative predictions of residual shear capacity and stiffness by the proposed framework are closer to experimental measurements than the ones by the existing methodologies. The qualitative condition classifications of the framework indicate less urgency for repair compared with the ones of the existing methodologies. The proposed framework enables the ranking of bridges within the same shear condition category due to its quantitative nature, and it has been implemented in a software application and can be used to set priorities for repair.
- Mahapatra, S., & Chandola, V. (2024). Learning manifolds from non-stationary streams. Journal of Big Data, 11(1), 42.
@article{Mahapatra2024, author = {Mahapatra, Suchismit and Chandola, Varun}, journal = {Journal of Big Data}, number = {1}, pages = {42}, title = {Learning manifolds from non-stationary streams}, volume = {11}, year = {2024} } - Desai, P., Juneja, N., Chandola, V., Zola, J., & Wodo, O. (2024). COMODO: Configurable morphology distance operator. Computational Materials Science, 244, 113208. https://www.sciencedirect.com/science/article/pii/S0927025624004294
@article{Desai2024, title = {COMODO: Configurable morphology distance operator}, journal = {Computational Materials Science}, volume = {244}, pages = {113208}, year = {2024}, issn = {0927-0256}, doi = {https://doi.org/10.1016/j.commatsci.2024.113208}, url = {https://www.sciencedirect.com/science/article/pii/S0927025624004294}, author = {Desai, Parth and Juneja, Namit and Chandola, Varun and Zola, Jaroslaw and Wodo, Olga}, keywords = {Morphology informatics, Distance operator, Graph, Configurable signature functions, Clustering}, file = {Desai2024.pdf} }Data-driven approaches have been recognized as a new paradigm for establishing and exploring process-morphology-property relationships. However, typical exploration methods deliver high-dimensional morphologies that pose the challenge of extracting the key features and patterns that could guide the processing and materials design. The high dimensionality also hampers the organization of the data and the associated data analytics. As a solution, the currently available approaches either take a simplified view of the morphology, e.g., focusing on pixels in the morphology images, or apply transformations that average out structural descriptors of morphologies. To address these shortcomings, we propose a new computationally efficient and configurable distance operator that takes an intermediate approach. Our main idea is to represent the morphology as a graph where graph connectivity reflects the relative arrangement of components (e.g., grains, droplets) in the morphology, and the label of the graph vertices captures the domain-specific information of each characteristic domain. Next, given the graph abstraction, the distance between morphologies is computed using vectorized graph-based representation. Because both morphology graph structure and associated signature functions have clear interpretations, our distance measure can be easily tailored to specific applications. Our results demonstrate the superior performance of the proposed approach on data from simulation and synthetic data, including in real-world applications like morphologies clustering.
- Guggilam, S., Chandola, V., & Patra, A. K. (2022). Tracking clusters and anomalies in evolving data streams. Statistical Analysis and Data Mining (SAM), 15(2), 156–178.
@article{Guggilam2022, author = {Guggilam, Sreelekha and Chandola, Varun and Patra, Abani K.}, year = {2022}, journal = {Statistical Analysis and Data Mining (SAM)}, volume = {15}, number = {2}, pages = {156--178}, title = {Tracking clusters and anomalies in evolving data streams}, doi = {10.1002/sam.11552}, file = {Guggilam2022.pdf} }Data-driven anomaly detection methods typically build a model for the normal behavior of the target system, and score each data instance with respect to this model. A threshold is invariably needed to identify data instances with high (or low) scores as anomalies. This presents a practical limitation on the applicability of such methods, since most methods are sensitive to the choice of the threshold, and it is challenging to set optimal thresholds. The issue is exacerbated in a streaming scenario, where the optimal thresholds vary with time. We present a probabilistic framework to explicitly model the normal and anomalous behaviors and probabilistically reason about the data. An extreme value theory based formulation is proposed to model the anomalous behavior as the extremes of the normal behavior. As a specific instantiation, a joint nonparametric clustering and anomaly detection algorithm (INCAD) is proposed that models the normal behavior as a Dirichlet process mixture model. Results on a variety of datasets, including streaming data, show that the proposed method provides effective and simultaneous clustering and anomaly detection without requiring strong initialization and threshold parameters.
- Lorenz, R., Chandola, V., Auerbach, S., Orom, H., Li, C.-S., & Chang, Y.-P. (2021). 160 Underlying Factors Contributing to Sleep Health Among Middle-aged and Older Adults. Sleep, 44(Supplement_2), A65–A65. https://doi.org/10.1093/sleep/zsab072.159
@article{Lorenz2021, author = {Lorenz, Rebecca and Chandola, Varun and Auerbach, Samantha and Orom, Heather and Li, Chin-Shang and Chang, Yu-Ping}, title = {{160 Underlying Factors Contributing to Sleep Health Among Middle-aged and Older Adults}}, journal = {Sleep}, volume = {44}, number = {Supplement_2}, pages = {A65-A65}, year = {2021}, month = may, issn = {0161-8105}, doi = {10.1093/sleep/zsab072.159}, url = {https://doi.org/10.1093/sleep/zsab072.159}, eprint = {https://academic.oup.com/sleep/article-pdf/44/Supplement\_2/A65/37655478/zsab072.159.pdf} }### Introduction Although poor sleep is not inherent with aging, an estimated 50-70 million adults in the US have insufficient sleep. Sleep duration is increasingly recognized as incomplete and insufficient. Instead, sleep health (SH), a multidimensional concept describing sleep/wake patterns that promote well-being has been shown to better reflect how sleep impacts the individual. Therefore, focusing on the underlying factors contributing to sleep health may provide the opportunity to develop interventions to improve sleep health in middle-age and older adults. ### Methods Data from the 2014 wave of the Health and Retirement Study (HRS) were used. Sample size was restricted to those who completed an additional questionnaire containing sleep variables. A derivation of the SH composite was constructed using eight selected sleep variables from the HRS data based on the five dimensions of sleep: Satisfaction, Alertness, Timing, Efficiency, and Duration. Total score ranged from 0-100, with higher scores indicating better SH. Weighting variables were based on complex sampling procedures and provided by HRS. Machine learning-based framework was used to identify determinants for predicting SH using twenty-six variables representing individual health and socio-demographics. Penalized linear regression with elastic net penalty was used to study the impact of individual predictors on SH. ### Results Our sample included 5,163 adults with a mean age of 67.8 years (SD=9.9; range 50-98 years). The majority were female (59%), white (78%), and married (61%). SH score ranged from 27-61 (mean=50; SD=6.7). Loneliness (coefficient=-1.92), depressive symptoms (coefficient=-1.28), and physical activity (coefficient=1.31) were identified as the strongest predictors of SH. Self-reported health status (coefficient=-1.11), daily pain (coefficient=-0.65), being middle-aged (coefficient=-0.26), and discrimination (coefficient=-0.23) were also significant predictors in this model. ### Conclusion Our study identified key predictors of SH among middle-aged and older adults using a novel approach of Machine Learning. Improving SH is a concrete target for health promotion through clinical interventions tailored towards increasing physical activity and reducing loneliness and depressive symptoms among middle-aged adults. ### Support (if any) This study was supported by National Heart, Lung, and Blood Institute (NHLBI) UB Clinical Scholar Program in Implementation Science to Achieve Triple Aims-NIH K12 Faculty Scholar Program in Implementation Science
- Zaidi, S. M. A., Chandola, V., & Yoo, E.-hye. (2021). DST-Predict: Predicting Individual Mobility Patterns From Mobile Phone GPS Data. IEEE Access, 9, 167592–167604.
@article{Zaidi2021, author = {Zaidi, Syed Mohammed Arshad and Chandola, Varun and Yoo, Eun-hye}, year = {2021}, journal = {IEEE Access}, volume = {9}, pages = {167592-167604}, title = {DST-Predict: Predicting Individual Mobility Patterns From Mobile Phone GPS Data}, doi = {10.1109/ACCESS.2021.3134586}, file = {Zaidi2021.pdf} }Predicting spatial behaviors of an individual (e.g., frequent visits to specific locations) is important to improve our understanding of the complexity of human mobility patterns, and to capture anomalous behaviors in an individual’s spatial movements, which can be particularly useful in situations such as those induced by the COVID-19 pandemic. We propose a system called Deep Spatio-Temporal Predictor (DST-Predict), that can predict the future visit frequency of an individual based on one’s past mobility behaviour patterns using GPS trace data collected from mobile phones. Predicting such spatial behavior is challenging, primarily because individuals’ patterns of location visits for each individual consists of both systematic and random components, which vary across the spatial and temporal scales of analysis. To address these issues, we propose a novel multi-view sequence-to-sequence model that uses Convolutional Long-short term memory (ConvLSTM) where the past history of frequent visit patterns features is used to predict individuals’ future visit patterns in a multi-step manner. Using the GPS survey data obtained from 1,464 participants in western New York, US, we demonstrated that the proposed system is capable of predicting individuals’ frequency of visits to common places in an urban setting, with high accuracy.
- Zaidi, S. M. A., Chandola, V., Ibrahim, M., Romanski, B., Mastrandrea, L. D., & Singh, T. (2021). Multi-step ahead predictive model for blood glucose concentrations of type-1 diabetic patients. Scientific Reports, 11(1), 24332.
@article{Zaidi2021a, author = {Zaidi, Syed Mohammed Arshad and Chandola, Varun and Ibrahim, Muhanned and Romanski, Bianca and Mastrandrea, Lucy D. and Singh, Tarunraj}, journal = {Scientific Reports}, number = {1}, pages = {24332}, title = {Multi-step ahead predictive model for blood glucose concentrations of type-1 diabetic patients}, volume = {11}, year = {2021} } - Schoeneman, F., Chandola, V., Napp, N., Wodo, O., & Zola, J. (2020). Learning Manifolds from Dynamic Process Data. Algorithms, 13(2).
@article{Schoeneman2020, author = {Schoeneman, Frank and Chandola, Varun and Napp, Nils and Wodo, Olga and Zola, Jaroslaw}, year = {2020}, journal = {Algorithms}, volume = {13}, number = {2}, title = {Learning Manifolds from Dynamic Process Data}, doi = {https://doi.org/10.3390/a13020030}, file = {Schoeneman2020.pdf} }Scientific data, generated by computational models or from experiments, are typically results of nonlinear interactions among several latent processes. Such datasets are typically high-dimensional and exhibit strong temporal correlations. Better understanding of the underlying processes requires mapping such data to a low-dimensional manifold where the dynamics of the latent processes are evident. While nonlinear spectral dimensionality reduction methods, e.g., Isomap, and their scalable variants, are conceptually fit candidates for obtaining such a mapping, the presence of the strong temporal correlation in the data can significantly impact these methods. In this paper, we first show why such methods fail when dealing with dynamic process data. A novel method, Entropy-Isomap, is proposed to handle this shortcoming. We demonstrate the effectiveness of the proposed method in the context of understanding the fabrication process of organic materials. The resulting low-dimensional representation correctly characterizes the process control variables and allows for informative visualization of the material morphology evolution.
- Wang, X., Bisantz, A. M., Bolton, M. L., Cavuoto, L., & Chandola, V. (2020). Explaining Supervised Learning Models: A Preliminary Study on Binary Classifiers. Ergonomics in Design, 28(3), 20–26.
@article{Wang2020, author = {Wang, Xiaomei and Bisantz, Ann M. and Bolton, Matthew L. and Cavuoto, Lora and Chandola, Varun}, year = {2020}, journal = {Ergonomics in Design}, volume = {28}, number = {3}, pages = {20-26}, title = {Explaining Supervised Learning Models: A Preliminary Study on Binary Classifiers}, doi = {10.1177/1064804620901641}, file = {Wang2020.pdf} }The reach of artificial intelligence continues to grow, particularly with the expansion of machine learning techniques that capitalize on increased computing power. Such systems could have tremendous benefits by providing predictions and suggestions. However, they are limited by the fact that they offer incomplete explanations of their predictions to human decision makers. The objective of this work was to summarize general information that could help users make judgments about whether a system is trustworthy and whether the system’s training “makes sense.” A preliminary study was summarized to show the importance of iterative design and testing for visualizing explanations.
- Xie, T., Chandola, V., & Kennedy, O. (2019). Query Log Compression for Workload Analytics. PVLDB.
@article{Xie2019, title = {Query Log Compression for Workload Analytics}, author = {Xie, Ting and Chandola, Varun and Kennedy, Oliver}, year = {2019}, journal = {pVLDB}, doi = {10.14778/3291264.3291265}, file = {Xie2019.pdf} }Analyzing database access logs is a key part of performance tuning, intrusion detection, benchmark development, and many other database administration tasks. Unfortunately, it is common for production databases to deal with millions or more queries each day, so these logs must be summarized before they can be used. Designing an appropriate summary encoding requires trading off between conciseness and infor- mation content. For example: simple workload sampling may miss rare, but high impact queries. In this paper, we present LogR, a lossy log compression scheme suitable for use in many automated log analytics tools, as well as for hu- man inspection. We formalize and analyze the space/fidelity trade-off in the context of a broader family of “pattern” and “pattern mixture” log encodings to which LogR belongs. We show through a series of experiments that LogR com- pressed encodings can be created efficiently, come with prov- able information-theoretic bounds on their accuracy, and outperform state-of-art log summarization strategies.
- Sullivan, S. S., Hewner, S., Chandola, V., & Westra, B. L. (2019). Mortality Risk in Homebound Older Adults Predicted From Routinely Collected Nursing Data. Nurs Res, 68(2), 156–166.
@article{Sullivan2019, author = {Sullivan, Suzanne S and Hewner, Sharon and Chandola, Varun and Westra, Bonnie L}, year = {2019}, journal = {Nurs Res}, volume = {68}, number = {2}, pages = {156--166}, title = {Mortality Risk in Homebound Older Adults Predicted From Routinely Collected Nursing Data.}, doi = {10.1097/NNR.0000000000000328}, file = {Sullivan2019.pdf} }BACKGROUND: Newer analytic approaches for developing predictive models provide a method of creating decision support to translate findings into practice. OBJECTIVES: The aim of this study was to develop and validate a clinically interpretable predictive model for 12-month mortality risk among community-dwelling older adults. This is done by using routinely collected nursing assessment data to aide homecare nurses in identifying older adults who are at risk for decline, providing an opportunity to develop care plans that support patient and family goals for care. METHODS: A retrospective secondary analysis of Medicare and Medicaid data of 635,590 Outcome and Assessment Information Set (OASIS-C) start-of-care assessments from January 1, 2012, to December 31, 2012, was linked to the Master Beneficiary Summary File (2012-2013) for date of death. The decision tree was benchmarked against gold standards for predictive modeling, logistic regression, and artificial neural network (ANN). The models underwent k-fold cross-validation and were compared using area under the curve (AUC) and other data science metrics, including Matthews correlation coefficient (MCC). RESULTS: Decision tree variables associated with 12-month mortality risk included OASIS items: age, (M1034) overall status, (M1800-M1890) activities of daily living total score, cancer, frailty, (M1410) oxygen, and (M2020) oral medication management. The final models had good discrimination: decision tree, AUC = .71, 95% confidence interval (CI) [.705, .712], sensitivity = .73, specificity = .58, MCC = .31; ANN, AUC = .74, 95% CI [.74, .74], sensitivity = .68, specificity = .68, MCC = .35; and logistic regression, AUC = .74, 95% CI [.735, .742], sensitivity = .64, specificity = .70, MCC = .35. DISCUSSION: The AUC and 95% CI for the decision tree are slightly less accurate than logistic regression and ANN; however, the decision tree was more accurate in detecting mortality. The OASIS data set was useful to predict 12-month mortality risk. The decision tree is an interpretable predictive model developed from routinely collected nursing data that may be incorporated into a decision support tool to identify older adults at risk for death.
- Jungquist, C. R., Chandola, V., Spulecki, C., Nguyen, K. V., Crescenzi, P., Tekeste, D., & Sayapaneni, P. R. (2019). Identifying Patients Experiencing Opioid-Induced Respiratory Depression During Recovery From Anesthesia: The Application of Electronic Monitoring Devices. Worldviews Evid Based Nurs, 16(3), 186–194.
@article{Jungquist2019, author = {Jungquist, Carla R and Chandola, Varun and Spulecki, Cheryl and Nguyen, Kenneth V and Crescenzi, Paul and Tekeste, Dejen and Sayapaneni, Phani Ram}, year = {2019}, journal = {Worldviews Evid Based Nurs}, volume = {16}, number = {3}, pages = {186--194}, title = {Identifying Patients Experiencing Opioid-Induced Respiratory Depression During Recovery From Anesthesia: The Application of Electronic Monitoring Devices.}, doi = {10.1111/wvn.12362}, file = {Jungquist2019.pdf} }BACKGROUND: Postsurgical patients experiencing opioid-related adverse drug events have 55% longer hospital stays, 47% higher costs associated with their care, 36% increased risk of 30-day readmission, and 3.4 times higher risk of inpatient mortality compared to those with no opioid-related adverse drug events. Most of the adverse events are preventable. GENERAL AIM: This study explored three types of electronic monitoring devices (pulse oximetry, capnography, and minute ventilation [MV]) to determine which were more effective at identifying the patient experiencing respiratory compromise and, further, to determine whether algorithms could be developed from the electronic monitoring data to aid in earlier detection of respiratory depression. MATERIALS AND METHODS: A study was performed in the postanesthesia care unit (PACU) in an inner city. Sixty patients were recruited in the preoperative admissions department on the day of their surgery. Forty-eight of the 60 patients wore three types of electronic monitoring devices while they were recovering from back, neck, hip, or knee surgery. Machine learning models were used for the analysis. RESULTS: Twenty-four of the 48 patients exhibited sustained signs of opioid-induced respiratory depression (OIRD). Although the SpO(2) values did not change, end-tidal CO(2) levels increased, and MV decreased, representing hypoventilation. A machine learning model was able to predict an OIRD event 10 min before the actual event occurred with 80% accuracy. LINKING EVIDENCE TO ACTION: Electronic monitoring devices are currently used as a tool to assess respiratory status using thresholds to distinguish when respiratory depression has occurred. This study introduces a potential paradigm shift from a reactive approach to a proactive approach that would identify a patient at high risk for OIRD. Capnography and MV were found to be effective tools in detecting respiratory compromise in the PACU.
- Luong, D. T. A., Singh, P., Ramezani, M., & Chandola, V. (2019). longSil: an Evaluation Metric to Assess Quality of Clustering Longitudinal Clinical Data. Journal of Healthcare Informatics Research, 3(4), 441–459.
@article{Luong2019, author = {Luong, Duc Thanh Anh and Singh, Prerna and Ramezani, Mahin and Chandola, Varun}, year = {2019}, journal = {Journal of Healthcare Informatics Research}, volume = {3}, number = {4}, pages = {441--459}, title = {longSil: an Evaluation Metric to Assess Quality of Clustering Longitudinal Clinical Data}, doi = {10.1007/s41666-019-00058-z}, file = {Luong2019.pdf} }Longitudinal disease subtyping is an important problem within the broader scope of computational phenotyping. In this article, we discuss several data-driven unsupervised disease subtyping methods to obtain disease subtypes from longitudinal clinical data. The methods are analyzed in the context of chronic kidney disease, one of the leading health problems, both in the USA and worldwide. To provide a quantitative comparison of the different methods, we propose a novel evaluation metric that measures the cluster tightness and degree of separation between the various clusters produced by each method. Comparative results for two significantly large clinical datasets are provided, along with key insights that are possible due to the proposed evaluation metric.
- Allen, M. R., Zaidi, S. M. A., Chandola, V., Morton, A. M., Brelsford, C. M., McManamay, R. A., KC, B., Sanyal, J., Stewart, R. N., & Bhaduri, B. L. (2018). A survey of analytical methods for inclusion in a new energy-water nexus knowledge discovery framework. Big Earth Data, 2(3), 197–227.
@article{Allen2018, author = {Allen, Melissa R. and Zaidi, Syed Mohammed Arshad and Chandola, Varun and Morton, April M. and Brelsford, Christa M. and McManamay, Ryan A. and KC, Binita and Sanyal, Jibonananda and Stewart, Robert N. and Bhaduri, Budhendra L.}, year = {2018}, journal = {Big Earth Data}, volume = {2}, number = {3}, pages = {197-227}, title = {A survey of analytical methods for inclusion in a new energy-water nexus knowledge discovery framework}, doi = {10.1080/20964471.2018.1524344}, file = {Allen2018.pdf} }The energy-water nexus, or the dependence of energy on water and water on energy, continues to receive attention as impacts on both energy and water supply and demand from growing populations and climate-related stresses are evaluated for future infrastructure planning. Changes in water and energy demand are related to changes in regional temperature, and precipitation extremes can affect water resources available for energy generation for those regional populations. Additionally, the vulnerabilities to the energy and water nexus are beyond the physical infrastructures themselves and extend into supporting and interdependent infrastructures. Evaluation of these vulnerabilities relies on the integration of the disparate and distributed data associated with each of the infrastructures, environments and populations served, and robust analytical methodologies of the data. A capability for the deployment of these methods on relevant data from multiple components on a single platform can provide actionable information for interested communities, not only for individual energy and water systems, but also for the system of systems that they comprise. Here, we survey the highest priority data needs and analytical methods for inclusion on such a platform.
- Zaidi, S. M. A., Chandola, V., Allen, M. R., Sanyal, J., Stewart, R. N., Bhaduri, B. L., & McManamay, R. A. (2018). Machine learning for energy-water nexus: challenges and opportunities. Big Earth Data, 2(3), 228–267.
@article{Zaidi2018, author = {Zaidi, Syed Mohammed Arshad and Chandola, Varun and Allen, Melissa R. and Sanyal, Jibonananda and Stewart, Robert N. and Bhaduri, Budhendra L. and McManamay, Ryan A.}, year = {2018}, journal = {Big Earth Data}, volume = {2}, number = {3}, pages = {228--267}, title = {Machine learning for energy-water nexus: challenges and opportunities}, doi = {10.1080/20964471.2018.1526057}, file = {Zaidi2018.pdf} }Modeling the interactions of water and energy systems is important to the enforcement of infrastructure security and system sustainability. To this end, recent technological advancement has allowed the production of large volumes of data associated with functioning of these sectors. We are beginning to see that statistical and machine learning techniques can help elucidate characteristic patterns across these systems from water availability, transport, and use to energy generation, fuel supply, and customer demand, and in the interdependencies among these systems that can leave these systems vulnerable to cascading impacts from single disruptions. In this paper, we discuss ways in which data and machine learning can be applied to the challenges facing the energy-water nexus along with the potential issues associated with the machine learning techniques themselves. We then survey machine learning techniques that have found application to date in energy-water nexus problems. We conclude by outlining future research directions and opportunities for collaboration among the energy-water nexus and machine learning communities that can lead to mutual synergistic advantage.
- Kul, G., Luong, D. T. A., Xie, T., Chandola, V., Kennedy, O., & Upadhyaya, S. (2018). Similarity Metrics for SQL Query Clustering. IEEE Trans. on Knowl. and Data Eng., 30(12), 2408–2420.
@article{Kul2018, author = {Kul, Gokhan and Luong, Duc Thanh Anh and Xie, Ting and Chandola, Varun and Kennedy, Oliver and Upadhyaya, Shambhu}, year = {2018}, journal = {IEEE Trans. on Knowl. and Data Eng.}, volume = {30}, number = {12}, pages = {2408–2420}, title = {Similarity Metrics for SQL Query Clustering}, doi = {10.1109/TKDE.2018.2831214}, file = {Kul2018.pdf} }Database access logs are the starting point for many forms of database administration, from database performance tuning, to security auditing, to benchmark design, and many more. Unfortunately, query logs are also large and unwieldy, and it can be difficult for an analyst to extract broad patterns from the set of queries found therein. Clustering is a natural first step towards understanding the massive query logs. However, many clustering methods rely on the notion of pairwise similarity, which is challenging to compute for SQL queries, especially when the underlying data and database schema is unavailable. We investigate the problem of computing similarity between queries, relying only on the query structure. We conduct a rigorous evaluation of three query similarity heuristics proposed in the literature applied to query clustering on multiple query log datasets, representing different types of query workloads. To improve the accuracy of the three heuristics, we propose a generic feature engineering strategy, using classical query rewrites to standardize query structure. The proposed strategy results in a significant improvement in the performance of all three similarity heuristics.
- Luong, D. T. A., Tran, D., Pace, W. D., Dickinson, M., Vassalotti, J., Carroll, J., Withiam-Leitch, M., Yang, M., Satchidanand, N., Staton, E., Kahn, L. S., Chandola, V., & Fox, C. H. (2017). Extracting Deep Phenotypes for Chronic Kidney Disease Using Electronic Health Records. EGEMS, 5(1), 9.
@article{Luong2017, author = {Luong, Duc Thanh Anh and Tran, Dinh and Pace, Wilson D and Dickinson, Miriam and Vassalotti, Joseph and Carroll, Jennifer and Withiam-Leitch, Matthew and Yang, Min and Satchidanand, Nikhil and Staton, Elizabeth and Kahn, Linda S and Chandola, Varun and Fox, Chester H}, year = {2017}, journal = {EGEMS}, volume = {5}, number = {1}, pages = {9}, title = {Extracting Deep Phenotypes for Chronic Kidney Disease Using Electronic Health Records.}, doi = {10.5334/egems.226}, file = {Luong2017.pdf} }INTRODUCTION: As chronic kidney disease (CKD) is among the most prevalent chronic diseases in the world with various rate of progression among patients, identifying its phenotypic subtypes is important for improving risk stratification and providing more targeted therapy and specific treatments for patients having different trajectories of the disease progression. PROBLEM DEFINITION AND DATA: The rapid growth and adoption of electronic health records (EHR) technology has created a unique opportunity to leverage the abundant clinical data, available as EHRs, to find meaningful phenotypic subtypes for CKD. In this study, we focus on extracting disease severity profiles for CKD while accounting for other confounding factors. PROBABILISTIC SUBTYPING MODEL: We employ a probabilistic model to identify precise phenotypes from EHR data of patients who have chronic kidney disease. Using this model, patient’s eGFR trajectory is decomposed as a combination of four different components including disease subtype effect, covariate effect, individual long-term effect and individual short-term effect. EXPERIMENTAL RESULTS: The discovered disease subtypes obtained by Probabilistic Subtyping Model for CKD are presented and their clinical relevance is analyzed. DISCUSSION: Several clinical health markers that were found associated with disease subtypes are presented with suggestion for further investigation on their use as risk predictors. Several assumptions in the study are also clarified and discussed. CONCLUSION: The large dataset of EHRs can be used to identify deep phenotypes retrospectively. Directions for further expansion of the model are also discussed.
- Chandola, V., Vatsavai, R., Kumar, D., & Ganguly, A. (2015). Analyzing Big Spatial and Big Spatiotemporal Data: A Case Study of Methods and Applications. Handbook of Statistics.
@article{Chandola2015, author = {Chandola, Varun and Vatsavai, Raju and Kumar, Devashish and Ganguly, Auroop}, year = {2015}, journal = {Handbook of Statistics}, title = {Analyzing Big Spatial and Big Spatiotemporal Data: A Case Study of Methods and Applications} }Spatial and spatiotemporal data mining is the process of discovering interesting and previously unknown, but potentially useful patterns from the data collected over time and space. However, explosive growth in the spatial and spatiotemporal data, and the emergence of social media and location sensing technologies, emphasizes the need for developing new and computationally efficient methods tailored for analyzing big data. In this chapter, we study approaches to handle big spatial and spatiotemporal data by closely looking at the computational and I/O requirements of several analysis algorithms for such data. We also study applications of such methods in domains where data is encountered at a massive scale.
- Chandola, V., Mithal, V., & Kumar, V. (2014). A reference based analysis framework for understanding anomaly detection techniques for symbolic sequences. Data Mining and Knowledge Discovery, 28(3), 702–735.
@article{Chandola2014, author = {Chandola, Varun and Mithal, Varun and Kumar, Vipin}, year = {2014}, journal = {Data Mining and Knowledge Discovery}, volume = {28}, number = {3}, pages = {702--735}, title = {A reference based analysis framework for understanding anomaly detection techniques for symbolic sequences}, doi = {10.1007/s10618-013-0315-0}, file = {Chandola2014.pdf} }Anomaly detection for symbolic sequence data is a highly important area of research and is relevant in many application domains. While several techniques have been proposed within different domains, understanding of their relative strengths and weaknesses is limited. The key factor for this is that the nature of sequence data varies significantly across domains, and hence while a technique might perform well in its original domain, its performance is not guaranteed in a different domain. In this paper, we aim at establishing this understanding for a wide variety of anomaly detection techniques for symbolic sequences. We present a comparative evaluation of a large number of anomaly detection techniques on a variety of publicly available as well as artificially generated data sets. Many of these are existing techniques while some are slight variants and/or adaptations of traditional anomaly detection techniques to sequence data. The analysis presented in this paper allows relative comparison of the different anomaly detection techniques and highlights their strengths and weaknesses. We extend the reference based analysis (RBA) framework, which was originally proposed to analyze multivariate categorical data, to analyze symbolic sequence data sets. We visualize the symbolic sequences using the characteristics provided by the RBA framework and use the visualization to understand various aspects of the sequence data. We then use the characterization done by RBA to understand the performance of the different techniques. Using the RBA framework, we propose two anomaly detection techniques for symbolic sequences, which show consistently superior performance over the existing techniques across the different data sets.
Refereed Conference Proceedings
- Guggilam, S., Chandola, V., & Patra, A. K. (2024). Tracking clusters and anomalies in evolving data streams. Statistical Analysis and Data Mining: The ASA Data Science Journal, 15(2), 156–178.
@inproceedings{Guggilam2024, author = {Guggilam, Sreelekha and Chandola, Varun and Patra, Abani K.}, year = {2024}, journal = {Statistical Analysis and Data Mining: The ASA Data Science Journal}, volume = {15}, number = {2}, pages = {156--178}, title = {Tracking clusters and anomalies in evolving data streams}, doi = {10.1002/sam.11552} }Data-driven anomaly detection methods typically build a model for the normal behavior of the target system, and score each data instance with respect to this model. A threshold is invariably needed to identify data instances with high (or low) scores as anomalies. This presents a practical limitation on the applicability of such methods, since most methods are sensitive to the choice of the threshold, and it is challenging to set optimal thresholds. The issue is exacerbated in a streaming scenario, where the optimal thresholds vary with time. We present a probabilistic framework to explicitly model the normal and anomalous behaviors and probabilistically reason about the data. An extreme value theory based formulation is proposed to model the anomalous behavior as the extremes of the normal behavior. As a specific instantiation, a joint nonparametric clustering and anomaly detection algorithm (INCAD) is proposed that models the normal behavior as a Dirichlet process mixture model. Results on a variety of datasets, including streaming data, show that the proposed method provides effective and simultaneous clustering and anomaly detection without requiring strong initialization and threshold parameters.
- Juneja, N., Chandola, V., Zola, J., Wodo, O., & Desai, P. (2024). Resource Efficient Bayesian Optimization. 2024 IEEE 17th International Conference on Cloud Computing (CLOUD), 12–19.
@inproceedings{10643903, author = {Juneja, Namit and Chandola, Varun and Zola, Jaroslaw and Wodo, Olga and Desai, Parth}, booktitle = {2024 IEEE 17th International Conference on Cloud Computing (CLOUD)}, title = {Resource Efficient Bayesian Optimization}, year = {2024}, volume = {}, number = {}, pages = {12-19}, keywords = {Training;Cloud computing;Costs;Computational modeling;Optimization methods;Machine learning;Bayes methods;Bayesian optimization;Resource-efficient op-timization;Expected Improvement;Gaussian processes;active learning}, doi = {10.1109/CLOUD62652.2024.00012} } - Juneja, N., Zola, J., Chandola, V., & Wodo, O. (2021). Graph-Based Strategy for Establishing Morphology Similarity. 33rd International Conference on Scientific and Statistical Database Management, 169–180. https://doi.org/10.1145/3468791.3468819
@inproceedings{Juneja2021, author = {Juneja, Namit and Zola, Jaroslaw and Chandola, Varun and Wodo, Olga}, title = {Graph-Based Strategy for Establishing Morphology Similarity}, year = {2021}, isbn = {9781450384131}, publisher = {Association for Computing Machinery}, address = {New York, NY, USA}, url = {https://doi.org/10.1145/3468791.3468819}, booktitle = {33rd International Conference on Scientific and Statistical Database Management}, pages = {169–180}, numpages = {12}, doi = {10.1145/3468791.3468819} }Analysis of morphological data is central to a broad class of scientific problems in materials science, astronomy, bio-medicine, and many others. Understanding relationships between morphologies is a core analytical task in such settings. In this paper, we propose a graph-based framework for measuring similarity between morphologies. Our framework delivers a novel representation of a morphology as an augmented graph that encodes application-specific knowledge through the use of configurable signature functions. It provides also an algorithm to compute the similarity between a pair of morphology graphs. We present experimental results in which the framework is applied to morphology data from high-fidelity numerical simulations that emerge in materials science. The results demonstrate that our proposed measure is superior in capturing the semantic similarity between morphologies, compared to the state-of-the-art methods such as FFT-based measures.
- Jiang, J., Hewner, S., & Chandola, V. (2021). Explainable Deep Learning for Readmission Prediction with Tree-GloVe Embedding. 2021 IEEE 9th International Conference on Healthcare Informatics (ICHI), 138–147.
@inproceedings{Jiang2021, author = {Jiang, Jialiang and Hewner, Sharon and Chandola, Varun}, booktitle = {2021 IEEE 9th International Conference on Healthcare Informatics (ICHI)}, title = {Explainable Deep Learning for Readmission Prediction with Tree-GloVe Embedding}, year = {2021}, pages = {138-147}, doi = {10.1109/ICHI52183.2021.00031} } - Böhlen, M., Jain, R., Sujarwo, W., & Chandola, V. (2021). From images in the wild to video-informed image classification. 2021 20th IEEE International Conference on Machine Learning and Applications (ICMLA), 656–661.
@inproceedings{Bohlen2021, author = {Böhlen, Marc and Jain, Raunaq and Sujarwo, Wawan and Chandola, Varun}, booktitle = {2021 20th IEEE International Conference on Machine Learning and Applications (ICMLA)}, title = {From images in the wild to video-informed image classification}, year = {2021}, pages = {656-661}, keywords = {Visualization;Conferences;Machine learning;Complexity theory;Image classification;Artificial intelligence in environmental studies;photography;video;video structure;classification;images in the wild;neural network-based image classification}, doi = {10.1109/ICMLA52953.2021.00109} } - Jiang, J., Chandola, V., & Hewner, S. (2019). Tree-based Regularization for Interpretable Readmission Prediction. Proceedings of AAAI Spring Symposium on Machine Learning and Knowledge Engineering (AAAI-MAKE).
@inproceedings{Jiang2019, author = {Jiang, Jialiang and Chandola, Varun and Hewner, Sharon}, year = {2019}, booktitle = {Proceedings of AAAI Spring Symposium on Machine Learning and Knowledge Engineering (AAAI-MAKE)}, title = {Tree-based Regularization for Interpretable Readmission Prediction}, file = {Jiang2019.pdf} }Preventable hospital readmissions have been identified as one of the primary targets for reducing costs and improv- ing healthcare delivery. However, most data driven studies for understanding readmissions have produced non-interpretable black boxes, which precludes them from being used effec- tively within the decision support systems in the hospitals. A novel strategy to improve the interpretability of a linear model by incorporating domain knowledge is proposed here. The central idea is to exploit the hierarchical relationships among the features (medical diagnosis codes, in this case) using a tree-structured sparsity-inducing regularization norm. The proposed method transforms the hierarchical relations among features into a graph and then applies graph-guided regular- ization during the model learning. Additionally, an evaluation metric is proposed to quantify the interpretability of a linear model with respect to the domain hierarchy. Results on two healthcare claims data sets are shown, where a model is learnt to predict a patient’s risk of readmission, based on the medi- cal history and other relevant features. Results show that the proposed method is able to learn a model which can predict readmission risk with accuracies that are comparable to ex- isting methods, but produces a highly interpretable output, which allows medical experts to draw clinically relevant in- sights and identify key factors associated with hospital read- missions. Some of these factors conform to existing beliefs, e.g., impact of surgical complications and infections during hospital stay. Other factors, such as the impact of mental dis- order and substance abuse on readmission, provide empirical evidence for several pre-existing but unverified hypotheses. The findings of this study will be instrumental in designing the next generation decision support systems for preventing readmissions.
- Guggilam, S., Zaidi, S. M. A., Chandola, V., & Patra, A. K. (2019). Integrated Clustering and Anomaly Detection (INCAD) for Streaming Data. Proceedings of International Conference on Computational Science, 45–59.
@inproceedings{Guggilam2019, author = {Guggilam, Sreelekha and Zaidi, Syed Mohammed Arshad and Chandola, Varun and Patra, Abani K.}, year = {2019}, booktitle = {Proceedings of International Conference on Computational Science}, pages = {45--59}, title = {Integrated Clustering and Anomaly Detection (INCAD) for Streaming Data}, doi = {10.1007/978-3-030-22747-0_4}, file = {Guggilam2019.pdf} }Most current clustering based anomaly detection methods use scoring schema and thresholds to classify anomalies. These methods are often tailored to target specific data sets with *known* number of clusters. The paper provides a streaming clustering and anomaly detection algorithm that does not require strict arbitrary thresholds on the anomaly scores or knowledge of the number of clusters while performing probabilistic anomaly detection and clustering simultaneously. This ensures that the cluster formation is not impacted by the presence of anomalous data, thereby leading to more reliable definition of “normal vs abnormal” behavior. The motivations behind developing the INCAD model [17] and the path that leads to the streaming model are discussed.
- Luong, D., & Chandola, V. (2019). Learning Deep Representations from Clinical Data for Chronic Kidney Disease. 2019 IEEE International Conference on Healthcare Informatics (ICHI), 1–10.
@inproceedings{Luong2019a, author = {Luong, Duc and Chandola, Varun}, year = {2019}, booktitle = {2019 IEEE International Conference on Healthcare Informatics (ICHI)}, pages = {1-10}, title = {Learning Deep Representations from Clinical Data for Chronic Kidney Disease}, doi = {10.1109/ICHI.2019.8904646}, file = {Luong2019a.pdf} }We study the behavior of a Time-Aware Long Short-Term Memory Autoencoder, a state-of-the-art method, in the context of learning latent representations from irregularly sampled patient data. We identify a key issue in the way such recurrent neural network models are being currently used and show that the solution of the issue leads to significant improvements in the learnt representations on both synthetic and real datasets. A detailed analysis of the improved methodology for representing patients suffering from Chronic Kidney Disease (CKD) using clinical data is provided. Experimental results show that the proposed T-LSTM model is able to capture the long-term trends in the data, while effectively handling the noise in the signal. Finally, we show that by using the latent representations of the CKD patients obtained from the T-LSTM autoencoder, one can identify unusual patient profiles from the target population.
- Sorkunlu, N., Luong, D. T. A., & Chandola, V. (2018). dynamicMF: A Matrix Factorization Approach to Monitor Resource Usage in High Performance Computing Systems. Proceedings of IEEE International Conference on Big Data.
@inproceedings{Sorkunlu2018, title = {dynamicMF: A Matrix Factorization Approach to Monitor Resource Usage in High Performance Computing Systems}, author = {Sorkunlu, Niyazi and Luong, Duc Thanh Anh and Chandola, Varun}, booktitle = {Proceedings of IEEE International Conference on Big Data}, year = {2018}, doi = {10.1109/BigData.2018.8622425} }High performance computing (HPC) facilities consist of a large number of interconnected computing units (or nodes) that execute highly complex scientific simulations to support scientific research. Monitoring such facilities, in real-time, is essential to ensure that the system operates at peak efficiency. Such systems are typically monitored using a variety of measurements and log data which capture the state of the various components within the system at regular intervals of time. As modern HPC systems grow in capacity and complexity, the data produced by current resource monitoring tools at a scale that is no longer feasible to be visually monitored by analysts. We propose a method that transforms the multidimensional output of resource monitoring tools to a low dimensional representation that facilitates the understanding of the behavior of a High Performance Computing (HPC) system. The proposed method automatically extracts the low-dimensional signal in the data which can be used to track the system efficiency and identify performance anomalies. The method models the resource usage data as a three dimensional tensor (capturing resource usage of all compute nodes for different resources over time). A dynamic matrix factorization algorithm, called dynamicMF, is proposed to extract a low-dimensional temporal signal for each node, which is subsequently fed into an anomaly detector. Results on resource usage data show anomalies identified which are correlated with anomalous events identified over the syslog messages.
- Mahapatra, S., & Chandola, V. (2017). S-Isomap++: Multi Manifold Learning from Streaming Data. IEEE International Conference on Big Data.
@inproceedings{Mahapatra2017, author = {Mahapatra, Suchismit and Chandola, Varun}, year = {2017}, booktitle = {IEEE International Conference on Big Data}, title = {S-Isomap++: Multi Manifold Learning from Streaming Data}, doi = {10.1109/BigData.2017.8257987}, file = {Mahapatra2017.pdf} }Manifold learning based methods have been widely used for non-linear dimensionality reduction (NLDR). However, in many practical settings, the need to process streaming data is a challenge for such methods, owing to the high computational complexity involved. Moreover, most methods operate under the assumption that the input data is sampled from a single manifold, embedded in a high dimensional space. We propose a method for streaming NLDR when the observed data is either sampled from multiple manifolds or irregularly sampled from a single manifold. We show that existing NLDR methods, such as Isomap, fail in such situations, primarily because they rely on smoothness and continuity of the underlying manifold, which is violated in the scenarios explored in this paper. However, the proposed algorithm is able to learn effectively in presence of multiple, and potentially intersecting, manifolds, while allowing for the input data to arrive as a massive stream.
- Kul, G., Luong, D., Xie, T., Coonan, P., Chandola, V., Kennedy, O., & Upadhyaya, S. (2016). Ettu: Analyzing Query Intents in Corporate Databases. Proceedings of the 25th International Conference Companion on World Wide Web, 463–466.
@inproceedings{10.1145/2872518.2888608, author = {Kul, Gokhan and Luong, Duc and Xie, Ting and Coonan, Patrick and Chandola, Varun and Kennedy, Oliver and Upadhyaya, Shambhu}, title = {Ettu: Analyzing Query Intents in Corporate Databases}, year = {2016}, booktitle = {Proceedings of the 25th International Conference Companion on World Wide Web}, pages = {463–466}, numpages = {4}, series = {WWW '16 Companion}, doi = {10.1145/2872518.2888608}, file = {Kul2016.pdf} }Insider threats to databases in the financial sector have become a very serious and pervasive security problem. This paper proposes a framework to analyze access patterns to databases by clustering SQL queries issued to the database. Our system Ettu works by grouping queries with other similarly structured queries. The small number of intent groups that result can then be efficiently labeled by human operators. We show how our system is designed and how the components of the system work. Our preliminary results show that our system accurately models user intent.
- Yang, Z., & Chandola, V. (2015). Surface Reconstruction from Intensity Image Using Illumination Model Based Morphable Modeling. Proceedings of the International Conference on Computer Vision Systems.
@inproceedings{Yang2015, author = {Yang, Zhi and Chandola, Varun}, year = {2015}, booktitle = {Proceedings of the International Conference on Computer Vision Systems}, title = {Surface Reconstruction from Intensity Image Using Illumination Model Based Morphable Modeling} }We present a new method for reconstructing depth of a known object from a single still image using deformed underneath sign matrix of a similar object. Existing Shape from Shading(SFS) methods try to establish a relationship between intensity values of a still image and surface normal of corresponding depth, but most of them resort to error minimization based approaches. Given the fact that these reconstruction approaches are fundamentally ill-posed, they have limited successes for surfaces like a human face. Photometric Stereo (PS) or Structure from Motion (SfM) based methods extend SFS by adding additional information/constraints about the target. Our goal is identical to SFS, however, we tackle the problem by building a relationship between gradient of depth and intensity value at the corresponding location of image of the same object. This formula is simplified and approximated for handing different materials, lighting conditions and, the underneath sign matrix is also obtained by resizing/deforming Region of Interest(ROI) with respect to its counterpart of a similar object. The target object is then reconstructed from its still image. In addition to the process, delicate details of the surface is also rebuilt using a Gabor Wavelet Network(GWN) on different ROIs. Finally, for merging the patches together, a Self-Organizing Maps(SOM) based method is used to retrieve and smooth boundary parts of ROIs. Compared with state of art SFS based methods, the proposed method yields promising results on both widely used benchmark datasets and images in the wild.
- Mahapatra, S., & Chandola, V. (2015). Modeling graphs using a mixture of Kronecker models. Proceedings of the International Conference on Big Data.
@inproceedings{Mahapatra2015, author = {Mahapatra, Suchismit and Chandola, Varun}, year = {2015}, booktitle = {Proceedings of the International Conference on Big Data}, title = {Modeling graphs using a mixture of Kronecker models} }Generative models for graphs are increasingly becoming a popular tool for researchers to generate realistic approximations of graphs. While in the past, focus was on generating graphs which follow general laws, such as the power law for degree distribution, current models have the ability to learn from observed graphs and generate synthetic approximations. The primary emphasis of existing models has been to closely match different properties of a single observed graph. Such models, though stochastic, tend to generate samples which do not have significant variance in terms of the various graph properties. We argue that in many cases real graphs are sampled drawn from a graph population (e.g., networks sampled at various time points, social networks for individual schools, healthcare networks for different geographic regions, etc.). Such populations typically exhibit significant variance. However, existing models are not designed to model this variance, which could lead to issues such as overfitting. We propose a graph generative model that focuses on matching the properties of real graphs and the natural variance expected for the corresponding population. The proposed model adopts a mixture-model strategy to expand the expressiveness of Kronecker product based graph models (KPGM), while building upon the two strengths of KPGM, viz., ability to model several key properties of graphs and to scale to massive graph sizes using its elegant fractal growth based formulation. The proposed model, called x-Kronecker Product Graph Model, or xKPGM, allows scalable learning from observed graphs and generates samples that match the mean and variance of several salient graph properties. We experimentally demonstrate the capability of the proposed model to capture the inherent variability in real world graphs on a variety of publicly available graph data sets.
- Chandola, V., Sukumar, S. R., & Schryver, J. C. (2013). Knowledge discovery from massive healthcare claims data. Proceedings of the 19th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, 1312–1320. https://doi.org/10.1145/2487575.2488205
@inproceedings{10.1145/2487575.2488205, author = {Chandola, Varun and Sukumar, Sreenivas R. and Schryver, Jack C.}, title = {Knowledge discovery from massive healthcare claims data}, year = {2013}, isbn = {9781450321747}, publisher = {Association for Computing Machinery}, address = {New York, NY, USA}, url = {https://doi.org/10.1145/2487575.2488205}, doi = {10.1145/2487575.2488205}, booktitle = {Proceedings of the 19th ACM SIGKDD International Conference on Knowledge Discovery and Data Mining}, pages = {1312–1320}, numpages = {9}, keywords = {fraud detection, healthcare analytics}, location = {Chicago, Illinois, USA}, series = {KDD '13}, file = {Chandola2013.pdf} }The role of big data in addressing the needs of the present healthcare system in US and rest of the world has been echoed by government, private, and academic sectors. There has been a growing emphasis to explore the promise of big data analytics in tapping the potential of the massive healthcare data emanating from private and government health insurance providers. While the domain implications of such collaboration are well known, this type of data has been explored to a limited extent in the data mining community. The objective of this paper is two fold: first, we introduce the emerging domain of b̈ig\"healthcare claims data to the KDD community, and second, we describe the success and challenges that we encountered in analyzing this data using state of art analytics for massive data. Specifically, we translate the problem of analyzing healthcare data into some of the most well-known analysis problems in the data mining community, social network analysis, text mining, and temporal analysis and higher order feature construction, and describe how advances within each of these areas can be leveraged to understand the domain of healthcare. Each case study illustrates a unique intersection of data mining and healthcare with a common objective of improving the cost-care ratio by mining for opportunities to improve healthcare operations and reducing what seems to fall under fraud, waste, and abuse.