@article{Tibshirani1996,
    abstract = {We propose a new method for estimation in linear models. The `lasso' minimizes the residual sum of squares subject to the sum of the absolute value of the coefficients being less than a constant. Because of the nature of this constraint it tends to produce some coefficients that are exactly 0 and hence gives interpretable models. Our simulation studies suggest that the lasso enjoys some of the favourable properties of both subset selection and ridge regression. It produces interpretable models like subset selection and exhibits the stability of ridge regression. There is also an interesting relationship with recent work in adaptive function estimation by Donoho and Johnstone. The lasso idea is quite general and can be applied in a variety of statistical models: extensions to generalized regression models and tree-based models are briefly described.},
    author = {Tibshirani, Robert},
    citeulike-article-id = {2453666},
    citeulike-linkout-0 = {http://dx.doi.org/10.2307/2346178},
    citeulike-linkout-1 = {http://www.jstor.org/stable/2346178},
    doi = {10.2307/2346178},
    issn = {00359246},
    journal = {Journal of the Royal Statistical Society. Series B (Methodological)},
    keywords = {inflation, methodology, problems, statistics, stepwise},
    number = {1},
    pages = {267--288},
    posted-at = {2010-05-26 10:32:01},
    priority = {2},
    publisher = {Blackwell Publishing for the Royal Statistical Society},
    title = {Regression Shrinkage and Selection via the Lasso},
    url = {http://dx.doi.org/10.2307/2346178},
    volume = {58},
    year = {1996}
}


@article{Steyerberg2001Feb,
	author = {Steyerberg, Ewout W. and Eijkemans, Marinus J. C. and Harrell, Frank E. and Habbema, J. Dik F.},
	title = {{Prognostic Modeling with Logistic Regression Analysis: In Search of a Sensible Strategy in Small Data Sets}},
	journal = {Med. Decis. Making},
	volume = {21},
	number = {1},
	pages = {45--56},
	year = {2001},
	month = feb,
	issn = {0272-989X},
	publisher = {SAGE Publications Inc STM},
	doi = {10.1177/0272989X0102100106},
	keywords = {regression analysis, logistic models, bias, variable selection, prediction},
	abstract = {{Clinical decision making often requires estimates of the likelihood of a dichotomous outcome in individual patients. When empirical data are available, these estimates may well be obtained from a logistic regression model. Several strategies may be followed in the development of such a model. In this study, the authors compare alternative strategies in 23 small subsamples from a large data set of patients with an acute myocardial infarction, where they developed predictive models for 30-day mortality. Evaluations were performed in an independent part of the data set. Specifically, the authors studied the effect of coding of covariables and stepwise selection on discriminative ability of the resulting model, and the effect of statistical “shrinkage” techniques on calibration. As expected, dichotomization of continuous covariables implied a loss of information. Remarkably, stepwise selection resulted in less discriminating models compared to full models including all available covariables, even when more than half of these were randomly associated with the outcome. Using qualitative information on the sign of the effect of predictors slightly improved the predictive ability. Calibration improved when shrinkage was applied on the standard maximum likelihood estimates of the regression coefficients. In conclusion, a sensible strategy in small data sets is to apply shrinkage methods in full models that include well-coded predictors that are selected based on external information.}}
}


@article{Wilkinson2016Mar,
	author = {Wilkinson, Mark D. and Dumontier, Michel and Aalbersberg, IJsbrand Jan and Appleton, Gabrielle and Axton, Myles and Baak, Arie and Blomberg, Niklas and Boiten, Jan-Willem and da Silva Santos, Luiz Bonino and Bourne, Philip E. and Bouwman, Jildau and Brookes, Anthony J. and Clark, Tim and Crosas, Merc{\ifmmode\grave{e}\else\`{e}\fi} and Dillo, Ingrid and Dumon, Olivier and Edmunds, Scott and Evelo, Chris T. and Finkers, Richard and Gonzalez-Beltran, Alejandra and Gray, Alasdair J. G. and Groth, Paul and Goble, Carole and Grethe, Jeffrey S. and Heringa, Jaap and {'}t Hoen, Peter A. C. and Hooft, Rob and Kuhn, Tobias and Kok, Ruben and Kok, Joost and Lusher, Scott J. and Martone, Maryann E. and Mons, Albert and Packer, Abel L. and Persson, Bengt and Rocca-Serra, Philippe and Roos, Marco and van Schaik, Rene and Sansone, Susanna-Assunta and Schultes, Erik and Sengstag, Thierry and Slater, Ted and Strawn, George and Swertz, Morris A. and Thompson, Mark and van der Lei, Johan and van Mulligen, Erik and Velterop, Jan and Waagmeester, Andra and Wittenburg, Peter and Wolstencroft, Katherine and Zhao, Jun and Mons, Barend},
	title = {{The FAIR Guiding Principles for scientific data management and stewardship}},
	journal = {Sci. Data},
	volume = {3},
	number = {160018},
	pages = {1--9},
	year = {2016},
	month = mar,
	issn = {2052-4463},
	publisher = {Nature Publishing Group},
	doi = {10.1038/sdata.2016.18},
	keywords = {Publication characteristics, Research data},
	abstract = {{There is an urgent need to improve the infrastructure supporting the reuse of scholarly data. A diverse set of stakeholders{\ifmmode---\else\textemdash\fi}representing academia, industry, funding agencies, and scholarly publishers{\ifmmode---\else\textemdash\fi}have come together to design and jointly endorse a concise and measureable set of principles that we refer to as the FAIR Data Principles. The intent is that these may act as a guideline for those wishing to enhance the reusability of their data holdings. Distinct from peer initiatives that focus on the human scholar, the FAIR Principles put specific emphasis on enhancing the ability of machines to automatically find and use the data, in addition to supporting its reuse by individuals. This Comment is the first formal publication of the FAIR Principles, and includes the rationale behind them, and some exemplar implementations in the community.}}
}

@article{Powers2020Oct,
	author = {Powers, David M. W.},
	title = {{Evaluation: from precision, recall and F-measure to ROC, informedness, markedness and correlation}},
	journal = {arXiv},
	year = {2020},
	month = oct,
	eprint = {2010.16061},
	doi = {10.48550/arXiv.2010.16061},
	keywords = {Machine Learning (cs.LG), Methodology (stat.ME), Machine Learning (stat.ML)},
	abstract = {{Commonly used evaluation measures including Recall, Precision, F-Measure and Rand Accuracy are biased and should not be used without clear understanding of the biases, and corresponding identification of chance or base case levels of the statistic. Using these measures a system that performs worse in the objective sense of Informedness, can appear to perform better under any of these commonly used measures. We discuss several concepts and measures that reflect the probability that prediction is informed versus chance. Informedness and introduce Markedness as a dual measure for the probability that prediction is marked versus chance. Finally we demonstrate elegant connections between the concepts of Informedness, Markedness, Correlation and Significance as well as their intuitive relationships with Recall and Precision, and outline the extension from the dichotomous case to the general multi-class case.}}
}

@article{Angell2019Nov,
	author = {Angell, Trevor E. and Maurer, Rie and Wang, Zhihong and Kim, Matthew I. and Alexander, Caroline A. and Barletta, Justine A. and Benson, Carol B. and Cibas, Edmund S. and Cho, Nancy L. and Doherty, Gerard M. and Doubilet, Peter M. and Frates, Mary C. and Gawande, Atul A. and Krane, Jeff F. and Marqusee, Ellen and Moore, Jr.,  Francis D. and Nehs, Matthew A. and Larsen, P. Reed and Alexander, Erik K.},
	title = {{A Cohort Analysis of Clinical and Ultrasound Variables Predicting Cancer Risk in 20,001 Consecutive Thyroid Nodules}},
	journal = {J. Clin. Endocrinol. Metab.},
	volume = {104},
	number = {11},
	pages = {5665--5672},
	year = {2019},
	month = nov,
	issn = {0021-972X},
	publisher = {Oxford Academic},
	doi = {10.1210/jc.2019-00664},
	abstract = {{ContextAssessing thyroid nodules for malignancy is complex. The impact of patient and nodule factors on cancer evaluation is uncertain.ObjectivesTo determine precise estimates of cancer risk associated with clinical and sonographic variables obtained during thyroid nodule assessment.DesignAnalysis of consecutive adult patients evaluated with ultrasound-guided fine-needle aspiration for a thyroid nodule {$\geq$}1 cm between 1995 and 2017. Demographics, nodule sonographic appearance, and pathologic findings were collected.Main Outcome MeasuresEstimated risk for thyroid nodule malignancy for patient and sonographic variables using mixed-effect logistic regression.ResultsIn 9967 patients [84{\%} women, median age 53 years (range 18 to 95)], thyroid cancer was confirmed in 1974 of 20,001 thyroid nodules (9.9{\%}). Significant ORs for malignancy were demonstrated for patient age {$<$}52 years [OR: 1.82, 95{\%} CI (1.63 to 2.05), P 75{\%} compared with predominantly solid, P {$<$} 0.0001 for both], and the presence of additional nodules {$\geq$}1 cm [OR: 0.69 (0.60 to 0.79) for two nodules, OR: 0.41 (0.34 to 0.49) for three nodules, and OR: 0.19 (0.16 to 0.22) for greater than or equal to four nodules compared with one nodule, P {$<$} 0.0001 for all]. A free online calculator was constructed to provide malignancy-risk estimates based on these variables.ConclusionsPatient and nodule characteristics enable more precise thyroid nodule risk assessment. These variables are obtained during routine initial thyroid nodule evaluation and provide new insights into individualized thyroid nodule care.}}
}

@article{Xi2022Jul,
	author = {Xi, Nan Miles and Wang, Lin and Yang, Chuanjia},
	title = {{Improving the diagnosis of thyroid cancer by machine learning and clinical data}},
	journal = {Sci. Rep.},
	volume = {12},
	number = {11143},
	pages = {1--11},
	year = {2022},
	month = jul,
	issn = {2045-2322},
	publisher = {Nature Publishing Group},
	doi = {10.1038/s41598-022-15342-z},
	keywords = {Cancer screening, Statistics, Tumour biomarkers},
	abstract = {{Thyroid cancer is a common endocrine carcinoma that occurs in the thyroid gland. Much effort has been invested in improving its diagnosis, and thyroidectomy remains the primary treatment method. A successful operation without unnecessary side injuries relies on an accurate preoperative diagnosis. Current human assessment of thyroid nodule malignancy is prone to errors and may not guarantee an accurate preoperative diagnosis. This study proposed a machine learning framework to predict thyroid nodule malignancy based on our collected novel clinical dataset. The ten-fold cross-validation, bootstrap analysis, and permutation predictor importance were applied to estimate and interpret the model performance under uncertainty. The comparison between model prediction and expert assessment shows the advantage of our framework over human judgment in predicting thyroid nodule malignancy. Our method is accurate, interpretable, and thus useable as additional evidence in the preoperative diagnosis of thyroid cancer.}}
}

@article{Triggiani2023Jan,
	author = {Triggiani, Vincenzo and Lisco, Giuseppe and Renzulli, Giuseppina and Frasoldati, Andrea and Guglielmi, Rinaldo and Garber, Jeffrey and Papini, Enrico},
	title = {{The TNAPP web-based algorithm improves thyroid nodule management in clinical practice: A retrospective validation study}},
	journal = {Front. Endocrinol.},
	volume = {13},
	pages = {1080159},
	year = {2023},
	month = jan,
	issn = {1664-2392},
	publisher = {Frontiers},
	doi = {10.3389/fendo.2022.1080159},
	keywords = {Thyroid Nodule, Thyroid carcinoma, Web-based algorithm, TNAPP, Fine-needle aspiration (FNA), Retrospective study},
	abstract = {{The detection of thyroid nodules has been increasing over time, resulting in extensive use of fine-needle aspiration (FNA) and cytology. Tailored methods are required to improve the management of thyroid nodules, including web-based tools. To assess the performance of the Thyroid Nodule App (TNAPP), a novel web-based readily modifiable, interactive algorithmic tool, in improving the management of thyroid nodules. One hundred twelve consecutive patients with 188 thyroid nodules who had FNA from January to December 2016 and thyroid surgery were retrospectively evaluated. Neck ultrasound images were collected from registry and re-examined to extract data to run TNAPP. Each nodule was evaluated for ultrasonographic risk and suitability for FNA. The sensitivity, specificity, positive and negative predictive values, and overall accuracy of TNAPP were calculated and compared to the diagnostic performance of other two algorithms (AACE/ACE/AME and ACR TI-RADS). TNAPP performed better in terms of sensitivity ({$>$}80{\%}) and negative predictive value (68{\%}) with an overall accuracy of 50.5{\%} which was similar to that found for the AACE/ACE/AME algorithm. The TNAPP displayed a slightly better performance compared to AACE/ACE/AME and ACR TI-RADS algorithms in discriminating unnecessary FNA for nodules with benign cytology (TNAPP 32{\%} vs. AACE/ACE/AME 31{\%} vs. ACR TI-RADS 29{\%}). The TNAPP reduced the number of missed diagnoses of thyroid nodules with suspicious and highly suspicious cytology (TNAPP 18{\%} vs. AACE/ACE/AME 26{\%} vs. ACR TI-RADS 20.5{\%}). Fourteen nodules that would not have been aspirated were malignant, 13 of which were microcarcinomas. The TNAPP algorithm is a reliable, easy to learn tool, that can be readily employed to improve the selection of thyroid nodules requiring cytological characterization. The rate of malignant nodules missed because of inaccurate characterization by TNAPP was low compared to the other two algorithms and almost all the cases were microcarcinomas. TNAPP{'}s use of size {$>$}20 mm as an independent determinant for considering or recommending FNA reduced its specificity. TNAPP performs well compared to AACE/ACE/AME and ACR-TIRADS. Additional retrospective and prospective studies are needed to confirm and guide the development of future iterations that incorporate different risk stratification systems and targets for diagnosing malignancy while reducing unnecessary FNA.}}
}

@article{Sands2011Feb,
	author = {Sands, Noah B. and Karls, Shawn and Amir, Alexander and Tamilia, Michael and Gologan, Olga and Rochon, Louise and Black, Martin J. and Hier, Michael P. and Payne, Richard J.},
	title = {{McGill Thyroid Nodule Score (MTNS): "rating the risk," a novel predictive scheme for cancer risk determination}},
	journal = {J. Otolaryngol. Head Neck Surg.},
	volume = {40},
	number = {Suppl},
	pages = {11--13},
	year = {2011},
	month = feb,
	issn = {1916-0216},
	publisher = {J Otolaryngol Head Neck Surg},
	eprint = {21453655},
	url = {https://pubmed.ncbi.nlm.nih.gov/21453655},
	keywords = {pmid:21453655, Comparative Study, Noah B Sands, Shawn Karls, Richard J Payne, Biopsy, Fine-Needle, Disease Progression, Female, Follow-Up Studies, Humans, Incidence, Male, Middle Aged, Positron-Emission Tomography, Quebec / epidemiology, Retrospective Studies, Risk Assessment / methods{$\ast$}, Risk Factors, Thyroid Neoplasms / diagnosis, Thyroid Neoplasms / epidemiology, Thyroid Nodule / diagnosis{$\ast$}, Thyroid Nodule / epidemiology, Thyroid Nodule / surgery, PubMed Abstract, NIH, NLM, NCBI, National Institutes of Health, National Center for Biotechnology Information, National Library of Medicine, MEDLINE},
	abstract = {{Our data suggest that a combined scoring system, the MTNS, can serve as an accurate predictor of the risk for thyroid cancer in a specific thyroid nodule. This will help physicians better formulate management decisions accordingly.}}
}

@article{Nixon2010Dec,
	author = {Nixon, Iain J. and Ganly, Ian and Hann, Lucy E. and Lin, Oscar and Yu, Changhong and Brandt, Suzanne and Shah, Jatin P. and Shaha, Ashok and Kattan, Michael W. and Patel, Snehal G.},
	title = {{Nomogram for predicting malignancy in thyroid nodules using clinical, biochemical, ultrasonographic, and cytologic features}},
	journal = {Surgery},
	volume = {148},
	number = {6},
	pages = {1120--1128},
	year = {2010},
	month = dec,
	issn = {0039-6060},
	publisher = {Mosby},
	doi = {10.1016/j.surg.2010.09.030},
	abstract = {{Background Thyroid nodules often discovered incidentally and present a management problem particularly when investigations suggest atypical or suspicious cells. Prediction of the risk of malignancy within such a thyroid nodule is based on clinical, biochemical, ultrasonographic, and cytologic features. Our aim was to create a nomogram to predict accurately the chance of malignancy within a thyroid nodule. Methods All patients with thyroid nodules who underwent ultrasonographic-guided fine needle aspiration and operative resection at our institution during 2007{\textendash}2008 were identified. Clinical records, biochemical profiles, pathology reports, ultrasonographic images, and cytology slides were reviewed. A multivariate logistic regression was used to quantify the value of the variables in estimating the risk of malignancy. Results The records of 158 patients with 190 nodules were reviewed. Eighteen nodules were excluded. The 8 variables with the greatest predictive value selected for the nomogram were biochemical (thyroid-stimulating hormone), ultrasonography (shape, echo texture, and vascularity), and cytology (nuclear grooves, pseudoinclusions, cellularity, and presence of colloid). The nomogram had an excellent predictive accuracy with a concordance index of 91{\%}. Conclusion We produced a nomogram that can quantify accurately the risk of malignancy in a thyroid nodule based on biochemical, ultrasonographic, and cytologic features.}}
}
