@article{shokoohi2023fmrs, title = {Sparse Estimation in Finite Mixture of Accelerated Failure Time and Mixture of Regression Models with R Package fmrs}, author = {Farhad Shokoohi}, url = {NEJSDS39}, doi = {NEJSDS39}, year = {2023}, date = {2023-09-04}, urldate = {2023-09-04}, journal = {The New England Journal of Statistics in Data Science}, volume = {0}, number = {0}, issue = {0}, pages = {1-14}, abstract = {Variable selection in large-dimensional data has been extensively studied in different settings over the past decades. In a recent article, Shokoohi et. al. [DOI:10.1214/18-AOAS1198] proposed variable selection in the finite mixture of accelerated failure time regression models for studies on time-to-event data to capture heterogeneity in the population and account for censoring. We present the package fmrs to implement variable selection in such models. As a byproduct, variable selection in the finite mixture of regression models can be carried out using fmrs. Moreover, the package performs tuning parameter selection based on a component-wise BIC. The most commonly used penalties such as Lasso and Scad are embedded in fmrs. The C language is chosen to boost the optimization speed. We give an overview of the fmrs principles and the strategies taken for optimization. Hands-on illustrations are presented to help users get acquainted with fmrs. By analyzing a lung cancer dataset, fmrs estimated a two-component mixture showing a more aggressive form of the disease has a lower survival time.}, keywords = {Finite mixture models, Finite mixture of AFT model, Penalized regression, Survival models, Variable selection}, pubstate = {published}, tppubtype = {article} } @article{Shokoohi2023Biometrics, title = {Sparse estimation in semi-parametric finite mixture of varying coefficient regression models}, author = {Abbas Khalili, Farhad Shokoohi, Masoud Asgharian, Shili Lin}, url = {https://onlinelibrary.wiley.com/doi/10.1111/biom.13870}, doi = {10.1111/biom.13870}, year = {2023}, date = {2023-04-17}, urldate = {2023-04-17}, journal = {Biometrics }, volume = {0}, number = {0}, issue = {0}, pages = {1–13}, abstract = {Finite mixture of regressions (FMR) are commonly used to model heterogeneous effects of covariates on a response variable in settings where there are unknown underlying subpopulations. FMRs, however, cannot accommodate situations where covariates' effects also vary according to an “index” variable—known as finite mixture of varying coefficient regression (FM-VCR). Although complex, this situation occurs in real data applications: the osteocalcin (OCN) data analyzed in this manuscript presents a heterogeneous relationship where the effect of a genetic variant on OCN in each hidden subpopulation varies over time. Oftentimes, the number of covariates with varying coefficients also presents a challenge: in the OCN study, genetic variants on the same chromosome are considered jointly. The relative proportions of hidden subpopulations may also change over time. Nevertheless, existing methods cannot provide suitable solutions for accommodating all these features in real data applications. To fill this gap, we develop statistical methodologies based on regularized local-kernel likelihood for simultaneous parameter estimation and variable selection in sparse FM-VCR models. We study large-sample properties of the proposed methods. We then carry out a simulation study to evaluate the performance of various penalties adopted for our regularized approach and ascertain the ability of a BIC-type criterion for estimating the number of subpopulations. Finally, we applied the FM-VCR model to analyze the OCN data and identified several covariates, including genetic variants, that have age-dependent effects on OCN.}, keywords = {Finite mixture models, Functional Regression Model, High-dimensional data analysis, Likelihood based estimation, Regularization, Variable selection}, pubstate = {published}, tppubtype = {article} } @article{Shokoohi2020AoASDMCFB, title = {Identifying Differential Methylation in Cancer Epigenetics via a Bayesian Functional Regression Model}, author = {Farhad Shokoohi and David A. Stephens and Celia M.T. Greenwood}, url = {https://www.biorxiv.org/content/10.1101/2021.03.21.436232v1}, doi = {10.1101/2021.03.21.436232}, year = {2021}, date = {2021-12-31}, journal = {submitted }, abstract = {DNA methylation plays an essential role in regulating gene activity, modulating disease risk, and determining treatment response. Researchers can obtain insight into methylation patterns at a single nucleotide level utilizing next-generation sequencing technologies. However, complex features inherent in the data obtained via these technologies pose challenges beyond the typical big data problems. Identifying differentially methylated cytosines (DMC) or regions is one of such challenges. Current methodologies for identifying DMCs fall short in handling low read-depth data and missing values, capturing functional data patterns, granting multiple covariates (categorical, continuous, or combination), and multiple group comparisons. We have developed an efficient method to identify DMCs based on a Bayesian functional regression approach, termed DMCFB, that tackles these shortcomings. Through simulation studies, we establish that DMCFB outperforms current methods and results in better smoothing, and efficient imputation. We apply the proposed method to analyze a dataset containing patients with acute promyelocytic leukemia and control samples. With DMCFB, we discovered many new DMCs, and more importantly, exhibited enhanced consistency of differential methylation within islands and at their adjacent shores. Furthermore, we detected differential methylation at more of the binding sites of the fused gene involved in this cancer.}, keywords = {Bayesian computation, Bisulfite sequencing, Differentially methylated region, Functional Regression Model, Natural Cubic Splines, Next‐generation sequencing}, pubstate = {forthcoming}, tppubtype = {article} } @conference{Shokoohi2019CMStatistics, title = {New insight into the role of intangible heterogeneity of covariate effects in hidden subpopulation subject to censoring}, author = {Farhad Shokoohi}, url = {http://www.cmstatistics.org/CMStatistics2019/}, year = {2019}, date = {2019-12-14}, address = {The 12th International Conference of the ERCIM WG on Computational and Methodological Statistics (CMStatistics 2019) at the Senate House, University of London, 14-16 December 2019}, keywords = {Incomplete data, Mixture models, Regularization}, pubstate = {published}, tppubtype = {conference} } @article{Shokoohi2019biom, title = {A hidden Markov model for identifying differentially methylated sites in bisulfite sequencing data}, author = {Farhad Shokoohi and David A. Stephens and Guillaume Bourque and Tomi Pastinen and Celia M.T. Greenwood and Aurélie Labbe}, doi = {10.1111/biom.12965}, year = {2019}, date = {2019-06-01}, journal = {Biometrics}, volume = {75}, pages = {210-221}, address = {https://doi.org/10.1111/biom.12965}, abstract = {DNA methylation studies have enabled researchers to understand methylation patterns and their regulatory roles in biological processes and disease. However, only a limited number of statistical approaches have been developed to provide formal quantitative analysis. Specifically, a few available methods do identify differentially methylated CpG (DMC) sites or regions (DMR), but they suffer from limitations that arise mostly due to challenges inherent in bisulfite sequencing data. These challenges include: (1) that read‐depths vary considerably among genomic positions and are often low; (2) both methylation and autocorrelation patterns change as regions change; and (3) CpG sites are distributed unevenly. Furthermore, there are several methodological limitations: almost none of these tools is capable of comparing multiple groups and/or working with missing values, and only a few allow continuous or multiple covariates. The last of these is of great interest among researchers, as the goal is often to find which regions of the genome are associated with several exposures and traits. To tackle these issues, we have developed an efficient DMC identification method based on Hidden Markov Models (HMMs) called “DMCHMM” which is a three‐step approach (model selection, prediction, testing) aiming to address the aforementioned drawbacks. Our proposed method is different from other HMM methods since it profiles methylation of each sample separately, hence exploiting inter‐CpG autocorrelation within samples, and it is more flexible than previous approaches by allowing multiple hidden states. Using simulations, we show that DMCHMM has the best performance among several competing methods. An analysis of cell‐separated blood methylation profiles is also provided.}, keywords = {Blood cell-separated data, Differentially methylated region, Next‐generation sequencing, Read‐depth}, pubstate = {published}, tppubtype = {article} } @misc{Shokoohi2019LifeTimeDataSci, title = {Capturing heterogeneity of covariate effects in hidden subpopulations in the presence of censoring and large number of covariates}, author = {Farhad Shokoohi}, url = {http://lids2019.pitt.edu/}, year = {2019}, date = {2019-05-30}, address = {The 2nd Conference on Lifetime Data Science, May 29-31, 2019, University of Pittsburgh, Pennsylvania, USA}, note = {Invited Talk}, keywords = {Finite mixture of AFT model, Penalized regression, Right censoring}, pubstate = {published}, tppubtype = {presentation} } @misc{Shokoohi2019j, title = {New insight into the role of intangible heterogeneity of gene effects on survival time of patients with ovarian cancer}, author = {Farhad Shokoohi}, url = {http://gqinnovationcenter.com/index.aspx}, year = {2019}, date = {2019-04-25}, address = {Montreal Genomics at McGill University, Montreal, QC, Canada}, note = {Departmental Talk}, keywords = {Finite mixture of AFT model, Ovarian cancer, Penalized regression}, pubstate = {published}, tppubtype = {presentation} } @article{Shokoohi2019AnnApplStat, title = {Capturing heterogeneity of covariate effects in hidden subpopulations in the presence of censoring and large number of covariates}, author = {Farhad Shokoohi and Abbas Khalili and Masoud Asgharian and Shili Lin}, doi = {10.1214/18-AOAS1198}, year = {2019}, date = {2019-04-10}, journal = {The Annals of Applied Statistics}, volume = {13}, number = {1}, pages = {444-465}, abstract = {The advent of modern technology has led to a surge of high-dimensional data in biology and health sciences such as genomics, epigenomics and medicine. The high-grade serous ovarian cancer (HGS-OvCa) data reported by The Cancer Genome Atlas (TCGA) Research Network is one example. The TCGA and other research groups have analyzed several aspects of these data. Here we study the relationship between Disease Free Time (DFT) after surgery among ovarian cancer patients and their DNA methylation profiles of genomic features. Such studies pose additional challenges beyond the typical big data problem due to population substructure and censoring. Despite the availability of several methods for analyzing time-to-event data with a large number of covariates but a small sample size, there is no method available to date that accommodates the additional feature of heterogeneity. To this end, we propose a regularized framework based on the finite mixture of accelerated failure time model to capture intangible heterogeneity due to population substructure and to account for censoring simultaneously. We study the properties of the proposed framework both theoretically and numerically. Our data analysis indicates the existence of heterogeneity in the HGS-OvCa data, with one component of the mixture capturing a more aggressive form of the disease, and the second component capturing a less aggressive form. In particular, the second component portrays a significant positive relationship between methylation and DFT for BRCA1. By further unearthing the negative relationship between expression and methylation for this gene, one may provide a biologically reasonable explanation that sheds light on the relationship between DNA methylation, gene expression and mutation.}, keywords = {DNA methylation, Finite mixture of AFT model, Ovarian cancer, Penalized regression, Right censoring}, pubstate = {published}, tppubtype = {article} } @misc{Shokoohi2019Memphis, title = {New insight into the role of intangible heterogeneity of gene effects on survival time of patients with ovarian cancer}, author = {Farhad Shokoohi}, year = {2019}, date = {2019-03-13}, address = {Department of Mathematical Sciences, University of Memphis, TN, USA}, note = {Invited Talk}, keywords = {Finite mixture of AFT model, Ovarian cancer, Penalized regression}, pubstate = {published}, tppubtype = {presentation} } @misc{Shokoohi2019LasVegas, title = {New insight into the role of intangible heterogeneity of gene effects on survival time of patients with ovarian cancer}, author = {Farhad Shokoohi}, year = {2019}, date = {2019-03-04}, address = {Department of Mathematical Sciences, University of Nevada Las Vegas, NV, USA}, note = {Invited Talk}, keywords = {Ovarian cancer, Penalized regression}, pubstate = {published}, tppubtype = {presentation} } @misc{Shokoohi2019Truman, title = {New insight into the role of intangible heterogeneity of gene effects on survival time of patients with ovarian cancer}, author = {Farhad Shokoohi}, year = {2019}, date = {2019-02-08}, address = {Department of Statistics, Truman State University, MO, USA}, note = {Invited Talk}, keywords = {Finite mixture of AFT model, Ovarian cancer, Penalized regression}, pubstate = {published}, tppubtype = {presentation} } @misc{Shokoohi2019YorkUK, title = {Some new advances in high-dimensional data analysis}, author = {Farhad Shokoohi}, year = {2019}, date = {2019-02-01}, address = {Department of Mathematics, University of York, York, United Kingdom}, note = {Invited Talk}, keywords = {Finite mixture models, Penalized regression}, pubstate = {published}, tppubtype = {presentation} } @misc{Shokoohi2018Laval, title = {New insight into the role of intangible heterogeneity of gene effects on survival time of patients with ovarian cancer}, author = {Farhad Shokoohi}, year = {2018}, date = {2018-12-11}, address = {Laval University, Montreal, QC, Canada}, note = {Invited Talk}, keywords = {Finite mixture of AFT model, Likelihood based estimation, Ovarian cancer, Penalized regression}, pubstate = {published}, tppubtype = {presentation} } @article{Shokoohi2018AsutNewZealand, title = {Semi‐parametric small‐area estimation by combining time‐series and cross‐sectional data methods}, author = {Farhad Shokoohi and Mahmoud Torabi}, doi = {10.1111/anzs.12234}, year = {2018}, date = {2018-07-02}, journal = {Australia & New Zealand Journal of Statistics}, volume = {60}, number = {3}, pages = {323-342}, abstract = {In survey sampling, policymaking regarding the allocation of resources to subgroups (called small areas) or the determination of subgroups with specific properties in a population should be based on reliable estimates. Information, however, is often collected at a different scale than that of these subgroups; hence, the estimation can only be obtained on finer scale data. Parametric mixed models are commonly used in small‐area estimation. The relationship between predictors and response, however, may not be linear in some real situations. Recently, small‐area estimation using a generalised linear mixed model (GLMM) with a penalised spline (P‐spline) regression model, for the fixed part of the model, has been proposed to analyse cross‐sectional responses, both normal and non‐normal. However, there are many situations in which the responses in small areas are serially dependent over time. Such a situation is exemplified by a data set on the annual number of visits to physicians by patients seeking treatment for asthma, in different areas of Manitoba, Canada. In cases where covariates that can possibly predict physician visits by asthma patients (e.g. age and genetic and environmental factors) may not have a linear relationship with the response, new models for analysing such data sets are required. In the current work, using both time‐series and cross‐sectional data methods, we propose P‐spline regression models for small‐area estimation under GLMMs. Our proposed model covers both normal and non‐normal responses. In particular, the empirical best predictors of small‐area parameters and their corresponding prediction intervals are studied with the maximum likelihood estimation approach being used to estimate the model parameters. The performance of the proposed approach is evaluated using some simulations and also by analysing two real data sets (precipitation and asthma).}, keywords = {Data cloning, Exponential family, Maximum likelihood estimation, Penalized spline, Random effects}, pubstate = {published}, tppubtype = {article} } @misc{Shokoohi2017, title = {New insight into the role of intangible heterogeneity of gene effects on survival time of patients with ovarian cancer}, author = {Farhad Shokoohi}, year = {2017}, date = {2017-10-13}, address = {Concordia University, Montreal, QC, Canada}, note = {Departmental Talk}, keywords = {DNA methylation, Finite mixture of AFT model, Ovarian cancer, Penalized regression}, pubstate = {published}, tppubtype = {presentation} } @conference{Shokoohi2017manitobassc, title = {A Novel Hidden Markov Model Approach for Differentially Methylated CpG site Identification in DNA methylation Data}, author = {Farhad Shokoohi}, url = {https://ssc.ca/en/meeting/annual/2017}, year = {2017}, date = {2017-07-12}, address = {45th Annual Meeting of the Statistical Society of Canada. June 11-14, 2017, University of Manitoba, Winnipeg, MB, Canada}, note = {Contributed Talk}, keywords = {Bisulfite sequencing, Differentially methylated region, DNA methylation, Next‐generation sequencing}, pubstate = {published}, tppubtype = {conference} } @misc{Shokoohi2017b, title = {New insight into the role of heterogeneity in Ovarian Cancer Data}, author = {Farhad Shokoohi}, year = {2017}, date = {2017-03-06}, address = {Lady Davis Institute for Medical Research, Montreal, QC, Canada}, note = {Departmental Talk}, keywords = {DNA methylation, Finite mixture of AFT model, Ovarian cancer, Penalized regression}, pubstate = {published}, tppubtype = {presentation} } @misc{Shokoohi2016b, title = {Variable Selection in Finite Mixture of Survival Models for Biomedical Genomic Studies}, author = {Farhad Shokoohi}, year = {2016}, date = {2016-09-25}, address = {Shahid Beheshti University, Iran}, note = {Departmental Talk}, keywords = {DNA methylation, Finite mixture of AFT model, Ovarian cancer, Penalized regression}, pubstate = {published}, tppubtype = {presentation} } @misc{Shokoohi2016Modarres, title = {Variable Selection in Finite Mixture of Survival Models for Biomedical Genomic Studies}, author = {Farhad Shokoohi}, year = {2016}, date = {2016-09-24}, address = {Tarbiat Modarres University, Iran}, note = {Departmental Talk}, keywords = {Finite mixture of AFT model, Genomic profile, Ovarian cancer, Penalized regression, Variable selection}, pubstate = {published}, tppubtype = {presentation} } @conference{Shokoohi2016SSC, title = {A Novel Hidden Markov Model Approach to Analyze Sequencing-based DNA Methylation Data}, author = {David A. Stephens and Farhad Shokoohi and Aurélie Labbe}, url = {https://ssc.ca/en/meetings/2016-annual-meeting-st-catharines-ontario}, year = {2016}, date = {2016-05-30}, address = {44th Annual Meeting of the Statistical Society of Canada. May 29 – June 1, 2016, Brock University, St. Catharines ON, Canada}, note = {Invited Talk}, keywords = {Bayesian computation, Blood cell-separated data, Hidden Markov models, Next‐generation sequencing}, pubstate = {published}, tppubtype = {conference} } @conference{Shokoohi2016ssc44Brock, title = {Feature selection in high-dimensional heterogeneous time-to-event data; A study on Ovarian Cancer}, author = {Farhad Shokoohi}, url = {https://ssc.ca/en/meetings/2016-annual-meeting-st-catharines-ontario}, year = {2016}, date = {2016-05-30}, address = {44th Annual Meeting of the Statistical Society of Canada. May 29-June 1, 2016, Brock University, St. Catharines, ON, Canada}, note = {Contributed Talk}, keywords = {Finite mixture of AFT model, High-dimensional data analysis, Ovarian cancer, Variable selection}, pubstate = {published}, tppubtype = {conference} } @misc{Shokoohi2016d, title = {A Novel Hidden Markov Model Approach to Analyze Sequencing-based DNA Methylation Data}, author = {Farhad Shokoohi}, year = {2016}, date = {2016-04-29}, address = {Lady Davis Institute for Medical Research, Montreal, Canada}, note = {Departmental Talk}, keywords = {Blood cell-separated data, Differentially methylated region, DNA methylation, Hidden Markov models, Next‐generation sequencing}, pubstate = {published}, tppubtype = {presentation} } @misc{Shokoohi2016eboh, title = {Feature selection in high-dimensional heterogeneous time-to-event data; A study on Ovarian Cancer}, author = {Farhad Shokoohi}, year = {2016}, date = {2016-03-08}, address = {Dept. of Epidemiology, Biostatistics, Occupational Health and Public Health, McGill University, Montreal, QC, Canada}, note = {Departmental Talk}, keywords = {Finite mixture of AFT model, High-dimensional data analysis, Ovarian cancer, Right censoring, Variable selection}, pubstate = {published}, tppubtype = {presentation} } @misc{Shokoohi2016gerad, title = {Feature selection in high-dimensional heterogeneous time-to-event data; A study on Ovarian Cancer}, author = {Farhad Shokoohi}, year = {2016}, date = {2016-03-04}, address = {GERAD, Montreal, QC, Canada}, note = {Departmental Talk}, keywords = {Finite mixture of AFT model, High-dimensional data analysis, Ovarian cancer, Variable selection}, pubstate = {published}, tppubtype = {presentation} } @conference{Shokoohi2015haifax, title = {DNA Methylation Analysis; A Thorough Comparison of Available Analytic Tools}, author = {Farhad Shokoohi}, url = {https://ssc.ca/en/meeting/2015-annual-meeting-halifax-nova-scotia}, year = {2015}, date = {2015-06-14}, address = {43rd Annual Meeting of the Statistical Society of Canada. June 14-17, 2015, Dalhousie University, Halifax, NS, Canada}, note = {Contributed Talk}, keywords = {Bisulfite sequencing, DNA methylation, Hidden Markov models}, pubstate = {published}, tppubtype = {conference} } @article{Shokoohi2015CJS, title = {Non‐parametric generalized linear mixed models in small area estimation}, author = {Mahmoud Torabi and Farhad Shokoohi}, doi = {10.1002/cjs.11236}, year = {2015}, date = {2015-01-14}, journal = {The Canadian Journal of Statistics}, volume = {43}, pages = {82–96}, abstract = {Mixed models are commonly used for the analysis of small area estimation. In particular, small area estimation has been extensively studied under linear mixed models. Recently, small area estimation under the linear mixed model with penalized spline (P‐spline) regression model, for fixed part of the model, has been proposed. However, in practice there are many situations that we have counts or proportions in small areas; for example a dataset on the number of asthma physician visits in small areas in Manitoba. In particular, the covariates age, genetic, environmental factors, among other covariates seem to predict asthma physician visits, however, these relationships may not be linear (see Section 5). In this paper, small area estimation under generalized linear mixed models using P‐spline regression models is proposed to cover Normal and non‐Normal responses. In particular, the empirical best predictor of small area parameters with corresponding prediction intervals are studied. The performance of the proposed approach is evaluated through simulation studies and also by a real dataset.}, keywords = {Bayesian computation, Exponential family, Penalized spline, Prediction interval, Random effects, Small-area estimation}, pubstate = {published}, tppubtype = {article} } @conference{Shokoohi2014Toronto, title = {Variable Selection in Mixture of Survival Models}, author = {Farhad Shokoohi and Masoud Asgharian and Abbas Khalili and Shili Lin}, url = {https://ssc.ca/en/meetings/2014-annual-meeting-toronto}, year = {2014}, date = {2014-05-26}, address = {42nd Annual Meeting of the Statistical Society of Canada. May 25-28, 2014, University of Toronto, Toronto, ON, Canada}, note = {Contributed Talk}, keywords = {Finite mixture of AFT model, Survival models, Variable selection}, pubstate = {published}, tppubtype = {conference} } @misc{Shokoohi2014sherbrooke, title = {Variable Selection in Mixture Models with Observations Subject to Right Censoring}, author = {Farhad Shokoohi}, year = {2014}, date = {2014-03-27}, address = {Department of Mathematics, University of Sherbrooke, Sherbrooke, Canada}, note = {Departmental Talk}, keywords = {Finite mixture of AFT model, High-dimensional data analysis, Right censoring, Variable selection}, pubstate = {published}, tppubtype = {presentation} } @misc{Shokoohi2013mcgill, title = {Some Recent Developments in Likelihood-Based Small-Area Estimation}, author = {Farhad Shokoohi}, year = {2013}, date = {2013-10-04}, address = {Department of Mathematics and Statistics, McGill University, Montreal, QC, Canada}, note = {Departmental talk}, keywords = {Likelihood based estimation, Small-area estimation}, pubstate = {published}, tppubtype = {presentation} } @conference{Shokoohi2013JSM, title = {Non-parametric Small Area Estimation}, author = {Mahmoud Torabi and Farhad Shokoohi}, url = {https://ww2.amstat.org/meetings/jsm/2013/onlineprogram/AbstractDetails.cfm?abstractid=307865}, year = {2013}, date = {2013-08-08}, address = {The Joint Statistical Meeting 2013, Montreal, QC, Canada}, abstract = {Mixed models are commonly used for the analysis of small area estimation. In particular, small area estimation has been extensively studied under linear mixed models. Recently, small area estimation under the linear mixed model with penalized spline model, for fixed part of the model, was proposed. However, in practice there are many situations that we have counts or proportions in small area estimation as a response variable; for example a dataset on the number of incidences in small areas. In this talk, small area estimation under generalized linear mixed models using penalized spline mixed models is proposed. A likelihood-based approach is used to predict small area parameters and also to provide prediction intervals. The performance of the proposed approach is evaluated through simulation studies and also by real datasets.}, note = {Invited Talk}, keywords = {Asthma, Bayesian computation, Nonparametric models, Small-area estimation}, pubstate = {published}, tppubtype = {conference} } @conference{Shokoohi2013jsmb, title = {Likelihood Inference in Small-Area Estimation Using P-Spline and Time Series Models}, author = {Farhad Shokoohi}, url = {https://ww2.amstat.org/meetings/jsm/2013/onlineprogram/AbstractDetails.cfm?abstractid=309172}, year = {2013}, date = {2013-08-03}, address = {Joint Statistical Meeting 2013. August 3-8, 2013, at the Palais des congres de Montreal, Montreal, QC, Canada}, abstract = {Nonparametric methods, despite their advantages, have not been often used in Small Area Estimation (SAE) due to methodological difficulties. Recently, a nonparametric linear model using cross-sectional data was introduced in SAE. However, there are many real applications in SAE which are time-related as well. In this talk, we introduce non-parametric models for Normal and non-Normal data situations with incorporating cross-sectional and time-series data. Frequentist analysis of these models is computationally difficult. Recent method of data cloning has overcome computational difficulties of mixed models from frequentist perspective. We demonstrate how the data cloning approach can be used to perform frequentist analysis of these complex models for both continuous and discrete data. One advantage of the data cloning approach is that both prediction and prediction intervals are easily obtained. The performance of the proposed approach is evaluated through several simulation studies and also by a real application.}, note = {Contributed Talk}, keywords = {Likelihood based estimation, P-Splines, Small-area estimation, Time series}, pubstate = {published}, tppubtype = {conference} } @conference{Shokoohi2013ssc, title = {Bayesian Small-Area Estimation Using P-Spline and Time Series Models}, author = {Farhad Shokoohi}, url = {https://ssc.ca/en/meetings/2013-annual-meeting-edmonton-alberta}, year = {2013}, date = {2013-05-27}, address = {41st Annual Meeting of the Statistical Society of Canada. May 26-29, 2013, University of Alberta, Edmonton, Alberta}, note = {Contributed Talk}, keywords = {Bayesian computation, Small-area estimation, Time series}, pubstate = {published}, tppubtype = {conference} } @misc{Shokoohi2013manitoba, title = {Likelihood Inference in Small-Area Estimation by Combining Time Series and Cross-Sectional Data}, author = {Farhad Shokoohi}, year = {2013}, date = {2013-03-21}, address = {Department of Statistics, University of Manitoba}, note = {Departmental Talk}, keywords = {Autocorrelated errors, Likelihood based estimation, Small-area estimation, Time series}, pubstate = {published}, tppubtype = {presentation} } @article{Shokoohi2013JDS, title = {Bonus-malus system in Iran: An empirical evaluation}, author = {Rahim Mahmoudvand and Alireza Edalati and Farhad Shokoohi}, url = {http://www.jds-online.com/file_download/378/JDS-1098.pdf}, year = {2013}, date = {2013-01-01}, journal = {Journal of Data Science}, volume = {11}, number = {1}, pages = {29-41}, abstract = {The aim of this paper is to represent the Bonus-Malus System (BMS) of Iran, which is a mandatory scheme based on Insurance act number 56. We examine the current Iranian BMS, using various criteria such as elasticity and time of convergence to steady state with respect to the claim frequency as well as financial balance. We also find the closed form of stationary distribution of the Iranian BMS that plays a key role in study of BMSs. Moreover, we compare the results with the German and Japan BMS. Finally we give some hints that can be used to improve the performance of the current Iranian BMS.}, keywords = {Claim frequency, Iranian Bonus Malus system, Stationary distribution}, pubstate = {published}, tppubtype = {article} } @article{Shokoohi2012JMA, title = {Likelihood inference in small-area estimation by combining time-series and cross-sectional data}, author = {Mahmoud Torabi and Farhad Shokoohi}, doi = {10.1016/j.jmva.2012.05.016}, year = {2012}, date = {2012-10-01}, journal = {Journal of Multivariate Analysis}, volume = {111}, pages = {213-221}, abstract = {Using both time-series and cross-sectional data, a linear model incorporating autocorrelated random effects and sampling errors was previously proposed in small area estimation. However, in practice there are many situations that we have time-related counts or proportions in small area estimation; for example a monthly dataset on the number of incidences in small areas. The frequentist analysis of these complex models is computationally difficult. On the other hand, the advent of the Markov chain Monte Carlo algorithm has made the Bayesian analysis of complex models computationally convenient. Recent introduction of the method of data cloning has made frequentist analysis of mixed models also equally computationally convenient. We use data cloning to conduct frequentist analysis of small area estimation for Normal and non-Normal data situations with incorporating cross-sectional and time-series data. Another important feature of the proposed approach is to predict small area parameters by providing prediction intervals. The performance of the proposed approach is evaluated through several simulation studies and also by a real dataset.}, keywords = {Autocorrelated errors, Bayesian computation, Exponential family, Hierarchical model, Prediction interval, Random effects}, pubstate = {published}, tppubtype = {article} } @article{Shokoohi2012JSCS, title = {Hierarchical Bayes estimation in small-area estimation using cross-sectional and time-series data}, author = {Mahmoud Torabi and Farhad Shokoohi}, doi = {10.1080/00949655.2012.721365}, year = {2012}, date = {2012-09-11}, journal = {Journal of Statistical Computation and Simulation }, volume = {84}, pages = {605-613}, abstract = {Bayesian methods have been extensively used in small area estimation. A linear model incorporating autocorrelated random effects and sampling errors was previously proposed in small area estimation using both cross-sectional and time-series data in the Bayesian paradigm. There are, however, many situations that we have time-related counts or proportions in small area estimation; for example, monthly dataset on the number of incidence in small areas. This article considers hierarchical Bayes generalized linear models for a unified analysis of both discrete and continuous data with incorporating cross-sectional and time-series data. The performance of the proposed approach is evaluated through several simulation studies and also by a real dataset.}, keywords = {Bayesian computation, Hierarchical model, Random effects, Time series}, pubstate = {published}, tppubtype = {article} } @misc{Shokoohi2012sbu1, title = {Some Contribution to Small-Area Estimation}, author = {Farhad Shokoohi}, year = {2012}, date = {2012-09-07}, address = {Department of Statistics, Shahid Beheshti University}, note = {Departmental Talk}, keywords = {Asthma, Autocorrelated errors, Bayesian computation, Likelihood based estimation, Small-area estimation}, pubstate = {published}, tppubtype = {presentation} } @phdthesis{Shokoohi2012Thesis, title = {Likelihood Inference in small-area estimation by combining time-series and cross-sectional data}, author = {Farhad Shokoohi}, year = {2012}, date = {2012-09-04}, school = {Shahid Beheshti University}, abstract = {Recently and for the past decades, small-area estimation (SAE) methods have gain a lot of attention in many fields such as medical, economical, social and agricultural sciences. These methods have many applications in providing reliable statistics demanded for areas with small or no sample. There are many different models in order to deal with available data in small area context. The main problems with these methods lie in complexity of calculating the estimators. In some methods, such as likelihood estimation, there are high dimensional integrals which often leads to meaningless solutions. Also in most of these methods, there are no exact estimations for “MSE” and “MSPE” and usually the approximation using Taylor expansion is used for their estimators. More importantly the analytic proof of identifiability of the model and estimability of parameters is very hard and mostly impossible. In addition to tradi- tional methods which are useful in working with most of introduced models in small area context, Bayesian methods are used to handle count or proportional response variables. On the other hand, we know that Bayesian inference depends on the choice of prior. In this thesis, the fundamental problems with current methods are discussed and the new method of data cloning approach is reviewed. In order to cope with the problems, a new methodology is introduced which helps analyzing the models in SAE context such as Rao-Yu model and then the models with count and proportional data are generalized. Also, the nonparametric model introduced by Opsomer et all. (2008) is generalized in order to account for time series data. For application of the new introduced methodology the Asthma rate of Manitoba province in Canada, the Acid rate in North-Eastern lakes of USA and the unemployment rate in Iran are analyzed. }, keywords = {Asthma, Bayesian computation, Data cloning, Hierarchical model, Likelihood based estimation, Prediction interval, Random effects, Small-area estimation, Time series}, pubstate = {published}, tppubtype = {phdthesis} } @conference{Shokoohi2012Iran, title = {Small-Area Estimation Based on Data Cloning Approach}, author = {Farhad Shokoohi}, year = {2012}, date = {2012-08-28}, address = {11th Iranian Statistical Conference, University of Science and Technology, Tehran, Iran}, note = {Contributed Talk}, keywords = {Data cloning, Small-area estimation, Time series}, pubstate = {published}, tppubtype = {conference} } @misc{Shokoohi2007, title = {Bayesian Inference in Generalized Poisson Distribution}, author = {Farhad Shokoohi}, year = {2007}, date = {2007-09-01}, address = {Central Insurance of Iran}, note = {Invited Talk}, keywords = {Bayesian computation, Generalized Poisson distribution, Reference prior}, pubstate = {published}, tppubtype = {presentation} } @mastersthesis{Shokoohi2006Thesis, title = {Bayesian inference in generalized poisson distribution}, author = {Farhad Shokoohi}, year = {2006}, date = {2006-07-01}, school = {Shahid Beheshti University}, keywords = {Bayesian computation, Generalized Poisson distribution, Reference prior}, pubstate = {published}, tppubtype = {mastersthesis} } @conference{Shokoohi2002, title = {Record Statistics}, author = {Farhad Shokoohi}, year = {2002}, date = {2002-10-01}, address = {1st Statistics and Mathematics Student Conference, Razi University, Kermanshah, Iran}, note = {Contributed Talk}, keywords = {Order statistics, Record statistics, Sampling theory}, pubstate = {published}, tppubtype = {conference} }