<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xml:lang="en"
    xmlns:mml="http://www.w3.org/1998/Math/MathML"
    xmlns:xlink="http://www.w3.org/1999/xlink"
    xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance">
    <processing-meta tagset-family="jats" base-tagset="publishing" mathml-version="2.0" table-model="xhtml"/>
    <front>
                        
                        <journal-meta>
            <issn>1732-3916</issn>
                                </journal-meta>
        <article-meta>
            <title-group>
                                    <article-title>Regression SVM for Incomplete Data</article-title>
                            </title-group>

                        <contrib-group>
                                                            <contrib contrib-type="author" corresp="yes">
                            <name>
                                <surname>Struski</surname>
                                <given-names>Łukasz</given-names>
                            </name>
                            <role>author</role>
                                                                                                                                    <xref ref-type="aff" rid="aff-1"/>
                                                                                        <xref ref-type="corresp" rid="cor-1"/>
                        </contrib>
                                            <contrib contrib-type="author" corresp="yes">
                            <name>
                                <surname>Śmieja</surname>
                                <given-names>Marek</given-names>
                            </name>
                            <role>author</role>
                                                                                                                                    <xref ref-type="aff" rid="aff-2"/>
                                                                                        <xref ref-type="corresp" rid="cor-2"/>
                        </contrib>
                                            <contrib contrib-type="author" corresp="no">
                            <name>
                                <surname>Zieliński </surname>
                                <given-names>Bartosz </given-names>
                            </name>
                            <role>author</role>
                                                                                                                                    <xref ref-type="aff" rid="aff-3"/>
                                                                                        <xref ref-type="corresp" rid="cor-3"/>
                        </contrib>
                                            <contrib contrib-type="author" corresp="yes">
                            <name>
                                <surname>Tabor</surname>
                                <given-names>Jacek</given-names>
                            </name>
                            <role>author</role>
                                                                                                                                    <xref ref-type="aff" rid="aff-4"/>
                                                                                        <xref ref-type="corresp" rid="cor-4"/>
                        </contrib>
                                                </contrib-group>

                                                                                        <aff id="aff-1">
                    <institution-wrap>
                        <institution>Uniwersytet Jagielloński w Krakowie, Polska, ul. Gołębia 24, 31-007 Kraków</institution>
                                            </institution-wrap>
                </aff>
                                                                                                                                                                                    <aff id="aff-4">
                    <institution-wrap>
                        <institution>Faculty of Mathematics and Computer Science, Jagiellonian University ul. Łojasiewicza 6, 30-348 Kraków, Poland</institution>
                                            </institution-wrap>
                </aff>
                            
            <author-notes>
                                    <corresp id="cor-1">Correspondence to: Łukasz Struski <email>lukasz.struski@im.uj.edu.pl</email></corresp>
                                    <corresp id="cor-2">Correspondence to: Marek Śmieja <email>marek.smiejag@ii.uj.edu.pl</email></corresp>
                                    <corresp id="cor-3">Correspondence to: Bartosz  Zieliński  <email></email></corresp>
                                    <corresp id="cor-4">Correspondence to: Jacek Tabor <email>jacek.tabor@uj.edu.pl</email></corresp>
                            </author-notes>

                            <pub-date date-type="pub" publication-format="electronic" iso-8601-date="2018-02-16">
                    <day>16</day>
                    <month>02</month>
                    <year>2018</year>
                </pub-date>
            
            <volume>Volume 26</volume>
            <issue>2017</issue>
                        <fpage>23</fpage>
                                    <lpage>35</lpage>
            
            <permissions>
                <copyright-statement>Copyright &#x00A9; 2018</copyright-statement>
                                    <copyright-year>2018</copyright-year>
                            </permissions>

            <funding-group specific-use="Crossref">
                <funding-statement></funding-statement>
            </funding-group>
        </article-meta>
    </front>
    <body>
        &lt;p style=&quot;text-align: left;&quot;&gt;The use of machine learning methods in the case of incomplete data is an important task in many scientific fields, like medicine, biology, or face recognition. Typically, missing values are substituted with artificial values that are estimated from the known samples, and the classical machine learning algorithms are applied. Although this methodology is very common, it produces less informative data, because artificially generated values are treated in the same way as the known ones. In this paper, we consider a probabilistic representation of missing data, where each vector is identified with a Gaussian probability density function, modeling the uncertainty of absent attributes. This representation allows to construct an analogue of RBF kernel for incomplete data. We show that such a kernel can be successfully used in regression SVM. Experimental results confirm that our approach capture relevant information that is not captured by traditional imputation methods.&lt;/p&gt;
    </body>
    <back>
                    <ref-list>
                                                                                <ref id="B1">
                            <label>1</label>
                            <article-title>[1] Little R.J., D’Agostino R., Cohen M.L., Dickersin K., Emerson S.S., Farrar J.T., Frangakis C., Hogan J.W., Molenberghs G., Murphy S.A., et al., The prevention and treatment of missing data in clinical trials. New England Journal of Medicine, 2012, 367 (14), pp. 1355–1360.</article-title>
                        </ref>
                                                                                                    <ref id="B2">
                            <label>2</label>
                            <article-title>[2] Acock, A.C., What to do about missing values., American Psychological Association, 2012.</article-title>
                        </ref>
                                                                                                    <ref id="B3">
                            <label>3</label>
                            <article-title>[3] Wagner A., Wright J., Ganesh A., Zhou Z., Mobahi H., Ma Y., Toward a practical face recognition system: Robust alignment and illumination by sparse representation. IEEE Transactions on Pattern Analysis and Machine Intelligence, 2012, 34 (2), pp. 372–386.</article-title>
                        </ref>
                                                                                                    <ref id="B4">
                            <label>4</label>
                            <article-title>[4] Garc´ıa-Laencina P.J., Sancho-G´omez J.L., Figueiras-Vidal A.R., Pattern classification with missing data: a review. Neural Computing and Applications, 2010, 19 (2), pp. 263–282.</article-title>
                        </ref>
                                                                                                    <ref id="B5">
                            <label>5</label>
                            <article-title>[5] McKnight P.E., McKnight K.M., Sidani S., Figueredo A.J., Missing data: A gentle introduction. Guilford Press, 2007.</article-title>
                        </ref>
                                                                                                    <ref id="B6">
                            <label>6</label>
                            <article-title>[6] Little R.J.A., Rubin D.B., Statistical analysis with missing data. John Wiley &amp;amp; Sons, 2014.</article-title>
                        </ref>
                                                                                                    <ref id="B7">
                            <label>7</label>
                            <article-title>[7] Schafer J.L., Analysis of incomplete multivariate data. CRC Press, 1997.</article-title>
                        </ref>
                                                                                                    <ref id="B8">
                            <label>8</label>
                            <article-title>[8] Ghahramani Z., Jordan M.I., Supervised learning from incomplete data via an EM approach. In: Advances in Neural Information Processing Systems, Citeseer, 1994, pp. 120–127.</article-title>
                        </ref>
                                                                                                    <ref id="B9">
                            <label>9</label>
                            <article-title>[9] Azur M.J., Stuart E.A., Frangakis C., Leaf P.J., Multiple imputation by chained equations: what is it and how does it work? International journal of methods in psychiatric research, 2011, 20 (1), pp. 40–49.</article-title>
                        </ref>
                                                                                                    <ref id="B10">
                            <label>10</label>
                            <article-title>[10] Williams D., Liao X., Xue Y., Carin L., Incomplete-data classification using logistic regression. In: Proceedings of the International Conference on Machine Learning, ACM, 2005, pp. 972–979.</article-title>
                        </ref>
                                                                                                    <ref id="B11">
                            <label>11</label>
                            <article-title>[11] Smola A.J., Vishwanathan S., Hofmann T., Kernel methods for missing variables. In: Proceedings of the International Conference on Artificial Intelligence and Statistics, Citeseer, 2005.</article-title>
                        </ref>
                                                                                                    <ref id="B12">
                            <label>12</label>
                            <article-title>[12] Williams D., Carin L., Analytical kernel matrix completion with incomplete multiview data. In: Proceedings of the ICML Workshop on Learning With Multiple Views, 2005.</article-title>
                        </ref>
                                                                                                    <ref id="B13">
                            <label>13</label>
                            <article-title>[13] Shivaswamy P.K., Bhattacharyya C., Smola A.J., Second order cone programming approaches for handling missing and uncertain data. Journal of Machine Learning Research, 2006, 7, pp. 1283–1314.</article-title>
                        </ref>
                                                                                                    <ref id="B14">
                            <label>14</label>
                            <article-title>[14] Chechik G., Heitz G., Elidan G., Abbeel P., Koller D., Max-margin classification of data with absent features. Journal of Machine Learning Research, 2008, 9, pp. 1–21.</article-title>
                        </ref>
                                                                                                    <ref id="B15">
                            <label>15</label>
                            <article-title>[15] Grangier D., Melvin I., Feature set embedding for incomplete data. In: Advances in Neural Information Processing Systems, 2010, pp. 793–801.</article-title>
                        </ref>
                                                                                                    <ref id="B16">
                            <label>16</label>
                            <article-title>[16] Struski L., ´Smieja M., Tabor J., Incomplete data representation for SVM classification. https://arxiv.org/abs/1612.01480, 2016. </article-title>
                        </ref>
                                                                                                    <ref id="B17">
                            <label>17</label>
                            <article-title>[17] Smola A.J., Sch¨olkopf B., A tutorial on support vector regression. Statistics and computing, 2004, 14 (3), pp. 199–222.</article-title>
                        </ref>
                                                                                                    <ref id="B18">
                            <label>18</label>
                            <article-title>[18] Asuncion A., Newman D.J., UCI Machine Learning Repository. http://archive.ics.uci.edu/ml/, 2007.</article-title>
                        </ref>
                                                                                                    <ref id="B19">
                            <label>19</label>
                            <article-title>[19] Batista G.E., Monard M.C., An analysis of four missing data treatment methods for supervised learning. Applied Artificial Intelligence, 2003, 17 (5-6), pp. 519– 533.</article-title>
                        </ref>
                                                                                                    <ref id="B20">
                            <label>20</label>
                            <article-title>[20] Li D., Deogun J., Spaulding W., Shuart B., Towards missing data imputation: a study of fuzzy k-means clustering method. In: International Conference on Rough Sets and Current Trends in Computing, Springer, 2004, pp. 573–579.</article-title>
                        </ref>
                                                                                                    <ref id="B21">
                            <label>21</label>
                            <article-title>[21] Honghai F., Guoshun C., Cheng Y., Bingru Y., Yumei C., A SVM regression based approach to filling in missing values. In: International Conference on Knowledge-Based and Intelligent Information and Engineering Systems, Springer, 2005, pp. 581–587.</article-title>
                        </ref>
                                                                                                    <ref id="B22">
                            <label>22</label>
                            <article-title>[22] Luengo J., Garc´ıa S., Herrera F., A study on the use of imputation methods for experimentation with radial basis function network classifiers handling missing attribute values: The good synergy between rbfns and eventcovering method. Neural Networks, 2010, 23 (3), pp. 406–418.</article-title>
                        </ref>
                                                                                                    <ref id="B23">
                            <label>23</label>
                            <article-title>[23] Alcal´a-Fdez J., Sanchez L., Garcia S., del Jesus M.J., Ventura S., Garrell J.M., Otero J., Romero C., Bacardit J., Rivas V.M., et al., Keel: a software tool to assess evolutionary algorithms for data mining problems. Soft Computing, 2009, 3 (3), pp. 307–318.</article-title>
                        </ref>
                                                                                                    <ref id="B24">
                            <label>24</label>
                            <article-title>[24] Demˇsar J., Statistical comparisons of classifiers over multiple data sets. Journal of Machine learning research, 2006, 7 (Jan), pp. 1–30.</article-title>
                        </ref>
                                                </ref-list>
            </back>
</article>
