<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.3 20210610//EN" "JATS-journalpublishing1-3.dtd">
<article article-type="research-article" dtd-version="1.3" xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xml:lang="ru"><front><journal-meta><journal-id journal-id-type="publisher-id">zldm</journal-id><journal-title-group><journal-title xml:lang="ru">Заводская лаборатория. Диагностика материалов</journal-title><trans-title-group xml:lang="en"><trans-title>Industrial laboratory. Diagnostics of materials</trans-title></trans-title-group></journal-title-group><issn pub-type="ppub">1028-6861</issn><issn pub-type="epub">2588-0187</issn><publisher><publisher-name>ООО «Издательство «ТЕСТ-ЗЛ»</publisher-name></publisher></journal-meta><article-meta><article-id pub-id-type="doi">10.26896/1028-6861-2023-89-7-71-77</article-id><article-id custom-type="elpub" pub-id-type="custom">zldm-1978</article-id><article-categories><subj-group subj-group-type="heading"><subject>Research Article</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="ru"><subject>МАТЕМАТИЧЕСКИЕ МЕТОДЫ ИССЛЕДОВАНИЯ</subject></subj-group><subj-group subj-group-type="section-heading" xml:lang="en"><subject>MATHEMATICAL METHODS OF INVESTIGATION</subject></subj-group></article-categories><title-group><article-title>Процедура проверки однородности выборок текстовых документов на основе непараметрических критериев</article-title><trans-title-group xml:lang="en"><trans-title>Procedure for checking the uniformity of samples of text documents based on nonparametric criteria</trans-title></trans-title-group></title-group><contrib-group><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Сафин</surname><given-names>Ш. И.</given-names></name><name name-style="western" xml:lang="en"><surname>Safin</surname><given-names>S. I.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Шахим Ильмирович Сафин</p><p>111250, Москва, ул. Красноказарменная, д. 14</p></bio><bio xml:lang="en"><p>Shahim I. Safin</p><p>14, Krasnokazarmennaya ul., Moscow, 111250</p></bio><xref ref-type="aff" rid="aff-1"/></contrib><contrib contrib-type="author" corresp="yes"><name-alternatives><name name-style="eastern" xml:lang="ru"><surname>Толчеев</surname><given-names>В. О.</given-names></name><name name-style="western" xml:lang="en"><surname>Tolcheev</surname><given-names>V. O.</given-names></name></name-alternatives><bio xml:lang="ru"><p>Владимир Олегович Толчеев</p><p>111250, Москва, ул. Красноказарменная, д. 14</p></bio><bio xml:lang="en"><p>Vladimir O. Tolcheev</p><p>14, Krasnokazarmennaya ul., Moscow, 111250</p></bio><email xlink:type="simple">tolcheevvo@mail.ru</email><xref ref-type="aff" rid="aff-1"/></contrib></contrib-group><aff-alternatives id="aff-1"><aff xml:lang="ru"><institution>Национальный исследовательский университет «Московский энергетический институт»</institution><country>Россия</country></aff><aff xml:lang="en"><institution>National Research University «Moscow Power Engineering Institute»</institution><country>Russian Federation</country></aff></aff-alternatives><pub-date pub-type="collection"><year>2023</year></pub-date><pub-date pub-type="epub"><day>26</day><month>07</month><year>2023</year></pub-date><volume>89</volume><issue>7</issue><fpage>71</fpage><lpage>77</lpage><permissions><copyright-statement>Copyright &amp;#x00A9; Сафин Ш.И., Толчеев В.О., 2023</copyright-statement><copyright-year>2023</copyright-year><copyright-holder xml:lang="ru">Сафин Ш.И., Толчеев В.О.</copyright-holder><copyright-holder xml:lang="en">Safin S.I., Tolcheev V.O.</copyright-holder><license xml:lang="ru" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>Данная работа распространяется под лицензией Creative Commons Attribution 4.0.</license-p></license><license xml:lang="en" license-type="creative-commons-attribution" xlink:href="https://creativecommons.org/licenses/by/4.0/" xlink:type="simple"><license-p>This work is licensed under a Creative Commons Attribution 4.0 License.</license-p></license></permissions><self-uri xlink:href="https://www.zldm.ru/jour/article/view/1978">https://www.zldm.ru/jour/article/view/1978</self-uri><abstract><p>При построении высокоточных классификаторов одной из важнейших задач является формирование достаточно больших репрезентативных и непротиворечивых выборок. В частности, при анализе и обработке текстовых документов объединяют наборы данных, полученных из различных информационных источников. В ряде случаев из-за нехватки профильных текстов на русском языке датасет расширяют за счет добавления переведенных англоязычных документов. В таких ситуациях целесообразно оценивать однородность-неоднородность объединяемых массивов. Однако подобная проверка осложняется тем, что документы представляют собой многомерные векторы, корректное сопоставление которых является весьма нетривиальной задачей. Недостаточная разработанность процедур проверки однородности выборок для многомерного случая приводит к тому, что на практике проблема возможных различий в данных игнорируется как несущественная. Как следствие, обучение классификаторов проводится по выборкам, представляющим собой смесь достаточно разнотипных текстов, и результирующее качество категоризации не улучшается (или даже ухудшается). Все это обуславливает актуальность разработки процедуры проверки однородности документальных выборок. Для этого авторы провели комплексное изучение проблемы сдвига в текстовых данных, выявили и проанализировали причины, которые определяют неоднородность документальных массивов. Исследуемые выборки состоят из библиографических описаний научных статьей (название, аннотация, ключевые слова). Авторы разработали процедуру оценки однородности двух выборок, имеющих приблизительно одинаковый объем и единый способ расчета весов терминов. Для сопоставления использовали центроиды, которые имеют размер общего словаря двух датасетов (в случае отсутствия некоторых терминов в соответствующие позиции центроидов проставляют нулевые значения). Представление выборок в виде «терминологических портретов» (центроидов) позволяет свести проверку однородности многомерных векторов-документов к хорошо изученной задаче анализа двух одномерных связанных выборок, для решения которой применяли непараметрические критерии (в частности, критерий знаков и критерий знаковых рангов Вилкоксона). Предложенная процедура проверки однородности выборок на основе непараметрических критериев проверена на трех коллекциях документов, полученных из русско- и англоязычных источников.</p></abstract><trans-abstract xml:lang="en"><p>One of the most important tasks in Text Mining is the formation of sufficiently large representative and consistent samples (datasets). Usually, datasets are obtained from various information sources. In some cases, due to the lack of specialized texts in Russian, the dataset is expanded by adding translated English-language documents. In such situations, it is advisable to evaluate the uniformity-heterogeneity of the combined arrays. However, such a verification is complicated by the fact that the documents are multidimensional vectors, the correct comparison of which is a very non-trivial task. Insufficient elaboration of procedures for checking the uniformity of samples for the multidimensional case leads to the fact the problem of possible differences in data is ignored that in practice as insignificant. As a result, classifiers are trained on samples that are a mixture of quite diverse texts, and the resulting quality of categorization does not improve (or even deteriorates). Thus, it seems relevant to develop a procedure for checking the uniformity of documentary samples. To do this, we provide a comprehensive study of the problem of shift in textual data, identified and analyzed the reasons that cause the heterogeneity of documentary arrays. In this study, the datasets consist of bibliographic descriptions of scientific articles (title, abstract, keywords). The authors develop a procedure for assessing the homogeneity of two samples having approximately the same volume and the same method for calculating the weights of terms. For comparison, centroids are used, which have the size of a common dictionary of two datasets (in the absence of some terms, zero values are put in the corresponding positions of the centroids). The representation of samples in the form of «terminological portraits» (centroids) allowed us to reduce the verification of the homogeneity of multidimensional document vectors to a well-studied problem of analyzing two one-dimensional connected samples, for which nonparametric criteria were used. The sign criterion and the Wilcoxon sign rank criterion were used in the study. The proposed procedure for checking the uniformity of samples was tested on three collections of documents obtained from Russian and English-language sources.</p></trans-abstract><kwd-group xml:lang="ru"><kwd>интеллектуальный анализ текстовых данных</kwd><kwd>однородность-неоднородность выборок</kwd><kwd>непараметрические критерии</kwd><kwd>сравнение центроидов</kwd></kwd-group><kwd-group xml:lang="en"><kwd>Text Mining</kwd><kwd>uniformity-heterogeneity of samples</kwd><kwd>nonparametric criteria</kwd><kwd>comparison of centroids</kwd></kwd-group></article-meta></front><back><ref-list><title>References</title><ref id="cit1"><label>1</label><citation-alternatives><mixed-citation xml:lang="ru">Орлов А. И. Прикладная статистика. — М.: Экзамен, 2006. — 671 с.</mixed-citation><mixed-citation xml:lang="en">Orlov A. I. Applied statistics. — Moscow: Ékzamen, 2006. — 671 p. [in Russian].</mixed-citation></citation-alternatives></ref><ref id="cit2"><label>2</label><citation-alternatives><mixed-citation xml:lang="ru">Бурков А. Инженерия машинного обучения. — М.: ДМК Пресс, 2022. — 306 с.</mixed-citation><mixed-citation xml:lang="en">Burkov A. Machine Learning Engineering. — Moscow: DMK Press, 2022. — 306 p. [in Russian].</mixed-citation></citation-alternatives></ref><ref id="cit3"><label>3</label><citation-alternatives><mixed-citation xml:lang="ru">Мулатов Н. И., Мохов А. С., Толчеев В. О. Способы построения текстовых коллекций для обучения классификаторов / Заводская лаборатория. Диагностика материалов. 2021. Т. 87. № 7. С. 76 – 84. DOI: 10.26896/1028-6861-2021-87-7-76-84</mixed-citation><mixed-citation xml:lang="en">Mulatov N. I., Mokhov A. S., Tolcheev V. O. Methods of constructing text collections for training classifiers / Zavod. Lab. Diagn. Mater. 2021. Vol. 87. N 7. P. 76 – 84 [in Russian]. DOI: 10.26896/1028-6861-2021-87-7-76-84</mixed-citation></citation-alternatives></ref><ref id="cit4"><label>4</label><citation-alternatives><mixed-citation xml:lang="ru">Кафтанников И. Л., Парасич А. В. Проблемы формирования обучающей выборки в задачах машинного обучения / Вестник ЮУрГУ. Серия Компьютерные технологии, управление, радиоэлектроника. 2016. Т. 16. № 3. С. 15 – 24.</mixed-citation><mixed-citation xml:lang="en">Kaftannikov I. L., Parasich A. V. Problems of training sample formation in machine learning tasks / Vestn. UUrGu. Ser. Komp’yut. Tekhnol. Upr. Radioélektr. 2016. Vol. 16 N 3. P. 15 – 24 [in Russian].</mixed-citation></citation-alternatives></ref><ref id="cit5"><label>5</label><citation-alternatives><mixed-citation xml:lang="ru">Холлендер М., Вульф Д. Непараметрические методы статистики. — М.: Финансы и статистика, 1983 — 518 с.</mixed-citation><mixed-citation xml:lang="en">Hollender M., Wolf D. Nonparametric methods of statistics. — Moscow: Finance and Statistics, 1983. — 518 p. [Russian translation].</mixed-citation></citation-alternatives></ref><ref id="cit6"><label>6</label><citation-alternatives><mixed-citation xml:lang="ru">Орлов А. И. Основные требования к математическим методам классификации / Заводская лаборатория. Диагностика материалов. 2020. Т. 86. ¹ 11. С. 67 – 78. DOI: 10.26896/1028-6861-2020-86-11-67-78</mixed-citation><mixed-citation xml:lang="en">Orlov A. I. Basic requirements for mathematical classification methods / Zavod. Lab. Diagn. Mater. 2020. Vol. 86. N 11. P. 67 – 78 [in Russian]. DOI: 10.26896/1028-6861-2020-86-11-67-78</mixed-citation></citation-alternatives></ref><ref id="cit7"><label>7</label><citation-alternatives><mixed-citation xml:lang="ru">Lipton Z., Wang Y-X., Smola A. Detecting and Correcting for Label Shift with Black Box Predictors / ArXiv: 1802.03916.2018.</mixed-citation><mixed-citation xml:lang="en">Lipton Z., Wang Y-X., Smola A. Detecting and Correcting for Label Shift with Black Box Predictors / ArXiv: 1802.03916.2018.</mixed-citation></citation-alternatives></ref><ref id="cit8"><label>8</label><citation-alternatives><mixed-citation xml:lang="ru">Dataset Shift in Machine Learning / J. Quinonero-Candela, M. Sugiyama, A. Schwaighofer, N. Lawrence, Eds. — The MIT Press, 2022. — 248 p.</mixed-citation><mixed-citation xml:lang="en">Dataset Shift in Machine Learning / J. Quinonero-Candela, M. Sugiyama, A. Schwaighofer, N. Lawrence, Eds. — The MIT Press, 2022. — 248 p.</mixed-citation></citation-alternatives></ref><ref id="cit9"><label>9</label><citation-alternatives><mixed-citation xml:lang="ru">Zhang K., Scholkopf B., Muandet K., Wang Z. Domain Adaptation under Target and Conditional Shift / Proceedings of the 30th International Conference on Machine Learning. 2013. Vol. 28. N 3. P. 819 – 827.</mixed-citation><mixed-citation xml:lang="en">Zhang K., Scholkopf B., Muandet K., Wang Z. Domain Adaptation under Target and Conditional Shift / Proceedings of the 30th International Conference on Machine Learning. 2013. Vol. 28. N 3. P. 819 – 827.</mixed-citation></citation-alternatives></ref><ref id="cit10"><label>10</label><citation-alternatives><mixed-citation xml:lang="ru">Subbaswamy A., Schulam P., Saria S. Preventing Failures Due to Dataset Shift: Learning Predictive Models that Transport / Proceedings of the 22nd International Conference on Artificial Intelligence and Statistics. 2019. Vol. 89. P. 3118 – 3127.</mixed-citation><mixed-citation xml:lang="en">Subbaswamy A., Schulam P., Saria S. Preventing Failures Due to Dataset Shift: Learning Predictive Models that Transport / Proceedings of the 22nd International Conference on Artificial Intelligence and Statistics. 2019. Vol. 89. P. 3118 – 3127.</mixed-citation></citation-alternatives></ref><ref id="cit11"><label>11</label><citation-alternatives><mixed-citation xml:lang="ru">Parker B., Khan L. Rapidly Labeling and Tracking Dynamically Evolving Concepts in Data Streams / IEEE 13th International Conference on Data Mining Workshops. 2013. P. 1161 – 1164.</mixed-citation><mixed-citation xml:lang="en">Parker B., Khan L. Rapidly Labeling and Tracking Dynamically Evolving Concepts in Data Streams / IEEE 13th International Conference on Data Mining Workshops. 2013. P. 1161 – 1164.</mixed-citation></citation-alternatives></ref><ref id="cit12"><label>12</label><citation-alternatives><mixed-citation xml:lang="ru">Ефимова И. В. Формирование однородных обучающих выборок для задач медицинской диагностики / Труды 57-й Международной научной конференции МФТИ. 2014. С. 91 – 92.</mixed-citation><mixed-citation xml:lang="en">Efimova I. V. Formation of homogeneous training samples for medical diagnostics tasks / Proceedings of the 57th International Scientific Conference of MIPT. 2014. P. 91 – 92 [in Russian].</mixed-citation></citation-alternatives></ref><ref id="cit13"><label>13</label><citation-alternatives><mixed-citation xml:lang="ru">Evangeline M., Shyamala K. Text Categorization Techniques: A Survey / International Conference on Innovative Practices in Technology and Management (ICIPTM). 2021. P. 137 – 142.</mixed-citation><mixed-citation xml:lang="en">Evangeline M., Shyamala K. Text Categorization Techniques: A Survey / International Conference on Innovative Practices in Technology and Management (ICIPTM). 2021. P. 137 – 142.</mixed-citation></citation-alternatives></ref><ref id="cit14"><label>14</label><citation-alternatives><mixed-citation xml:lang="ru">Kreutz C. K., Schenkel R. Scientific Paper Recommendation Systems: a Literature Review of recent Publications / ArXiv: 2201.00682.2022.</mixed-citation><mixed-citation xml:lang="en">Kreutz C. K., Schenkel R. Scientific Paper Recommendation Systems: a Literature Review of recent Publications / ArXiv: 2201.00682.2022.</mixed-citation></citation-alternatives></ref><ref id="cit15"><label>15</label><citation-alternatives><mixed-citation xml:lang="ru">Silambarasan M., Shathik J. Ensemble Text Classifier: A Document Classification Technique to Predict and Categorizes Regularised and Novel Classes Using Incremental Learning / International Journal of Applied Engineering Research. 2017. Vol. 12. N 22. P. 12454 – 12459.</mixed-citation><mixed-citation xml:lang="en">Silambarasan M., Shathik J. Ensemble Text Classifier: A Document Classification Technique to Predict and Categorizes Regularised and Novel Classes Using Incremental Learning / International Journal of Applied Engineering Research. 2017. Vol. 12. N 22. P. 12454 – 12459.</mixed-citation></citation-alternatives></ref><ref id="cit16"><label>16</label><citation-alternatives><mixed-citation xml:lang="ru">Understanding Dataset Shift and Potential Remedies. Technical Report. — Vector Institute, 2021. — 27 p.</mixed-citation><mixed-citation xml:lang="en">Understanding Dataset Shift and Potential Remedies. Technical Report. — Vector Institute, 2021. — 27 p.</mixed-citation></citation-alternatives></ref><ref id="cit17"><label>17</label><citation-alternatives><mixed-citation xml:lang="ru">Орлов А. И. Какие гипотезы можно проверять с помощью двухвыборочного критерия Вилкоксона / Заводская лаборатория. Диагностика материалов. 1999. Т. 65. № 1. С. 51 – 56.</mixed-citation><mixed-citation xml:lang="en">Orlov A. I. What hypotheses can be tested using the two-sample Wilcoxon criterion / Zavod. Lab. Diagn. Maters. 1999. Vol. 65. N 1. P. 51 – 56 [in Russian].</mixed-citation></citation-alternatives></ref><ref id="cit18"><label>18</label><citation-alternatives><mixed-citation xml:lang="ru">Орлов А. И. Модель анализа совпадений при расчете непараметрических ранговых статистик / Заводская лаборатория. Диагностика материалов. 2017. Т. 83. № 11. С. 66 – 72. DOI: 10.26896/1028-6861-2017-83-11-66-72</mixed-citation><mixed-citation xml:lang="en">Orlov A. I. Model of coincidence analysis in the calculation of nonparametric rank statistics / Zavod. Lab. Diagn. Mater. 2017. Vol. 83. N. 11. P. 66 – 72 [in Russian]. DOI: 10.26896/1028-6861-2017-83-11-66-72</mixed-citation></citation-alternatives></ref><ref id="cit19"><label>19</label><citation-alternatives><mixed-citation xml:lang="ru">Орлов А. И. Распределения реальных статистических данных не являются нормальными / Научный журнал КубГАУ. 2016. № 117. С. 71 – 90.</mixed-citation><mixed-citation xml:lang="en">Orlov A. I. Distributions of real statistical data are not normal / Scientific Journal of KubGAU. 2016. N. 117. P. 71 – 90 [in Russian].</mixed-citation></citation-alternatives></ref><ref id="cit20"><label>20</label><citation-alternatives><mixed-citation xml:lang="ru">Орлов А. И. Методы проверки однородности связанных выборок / Заводская лаборатория. Диагностика материалов. 2004. Т. 70. ¹ 7. С. 57 – 61.</mixed-citation><mixed-citation xml:lang="en">Orlov A. I. Methods of checking the homogeneity of related samples / Zavod. Lab. Diagn. Mater. 2004. Vol. 70. N. 7. P. 57 – 61 [in Russian].</mixed-citation></citation-alternatives></ref><ref id="cit21"><label>21</label><citation-alternatives><mixed-citation xml:lang="ru">Frias-Blanco I., Campo-Avila J., Ramos-Jimenez G., Morales-Bueno R., Ortiz-Diaz A., Caballero-Mota Y. Online and Non-Parametric Drift Detection Methods Based on Hoeffding’s Bounds / IEEE Transactions on Knowledge and Data Engineering. 2014. Vol. 27. N 3. P. 810 – 823.</mixed-citation><mixed-citation xml:lang="en">Frias-Blanco I., Campo-Avila J., Ramos-Jimenez G., Morales-Bueno R., Ortiz-Diaz A., Caballero-Mota Y. Online and Non-Parametric Drift Detection Methods Based on Hoeffding’s Bounds / IEEE Transactions on Knowledge and Data Engineering. 2014. Vol. 27. N 3. P. 810 – 823.</mixed-citation></citation-alternatives></ref><ref id="cit22"><label>22</label><citation-alternatives><mixed-citation xml:lang="ru">Digital Library Elibrary [cited February 3, 2023]. Available: https://eLibrary.ru</mixed-citation><mixed-citation xml:lang="en">Digital Library Elibrary [cited February 3, 2023]. Available: https://eLibrary.ru</mixed-citation></citation-alternatives></ref><ref id="cit23"><label>23</label><citation-alternatives><mixed-citation xml:lang="ru">Electronic archive of scientific articles of Cornell University with open access [cited February 3, 2023]. Available: https:// arxiv.org</mixed-citation><mixed-citation xml:lang="en">Electronic archive of scientific articles of Cornell University with open access [cited February 3, 2023]. Available: https://arxiv.org</mixed-citation></citation-alternatives></ref><ref id="cit24"><label>24</label><citation-alternatives><mixed-citation xml:lang="ru">Electronic Library of the Association for Computing Machinery ACM Digital Library [cited February 3, 2023]. Available: https://dl.acm.org</mixed-citation><mixed-citation xml:lang="en">Electronic Library of the Association for Computing Machinery ACM Digital Library [cited February 3, 2023]. Available: https://dl.acm.org</mixed-citation></citation-alternatives></ref></ref-list><fn-group><fn fn-type="conflict"><p>The authors declare that there are no conflicts of interest present.</p></fn></fn-group></back></article>
