<?xml version="1.0" encoding="UTF-8"?>
<!DOCTYPE article PUBLIC "-//NLM//DTD JATS (Z39.96) Journal Publishing DTD v1.2 20190208//EN" "https://jats.nlm.nih.gov/publishing/1.2/JATS-journalpublishing1.dtd">
<article xmlns:mml="http://www.w3.org/1998/Math/MathML" xmlns:xlink="http://www.w3.org/1999/xlink" xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance" xmlns:ali="http://www.niso.org/schemas/ali/1.0/" article-type="research-article" dtd-version="1.2" xml:lang="en">
<front> <journal-meta>
<journal-id journal-id-type="publisher-id">Informatization in the Digital Economy</journal-id>
<journal-title-group>
<journal-title xml:lang="en">Informatization in the Digital Economy</journal-title>
<trans-title-group xml:lang="ru">
<trans-title>Информатизация в цифровой экономике</trans-title>
</trans-title-group>
</journal-title-group>
<issn publication-format="print">2712-9306</issn>
<publisher>
<publisher-name xml:lang="en">BIBLIO-GLOBUS Publishing House</publisher-name>
</publisher>
</journal-meta><article-meta>
<article-id pub-id-type="publisher-id">126108</article-id>
<article-id pub-id-type="doi">10.18334/ide.7.2.126108</article-id>
<article-id custom-type="edn" pub-id-type="custom">EQIPNQ</article-id>
<article-categories>
<subj-group subj-group-type="toc-heading" xml:lang="en">
<subject>Articles</subject>
</subj-group>
<subj-group subj-group-type="toc-heading" xml:lang="ru">
<subject>Статьи</subject>
</subj-group>
<subj-group subj-group-type="article-type">
<subject>Research Article</subject>
</subj-group>
</article-categories>
<title-group>
<article-title xml:lang="en">An alternative approach to the industry classification of public companies based on text clustering: an empirical study based on NASDAQ data</article-title>
<trans-title-group xml:lang="ru">
<trans-title>Альтернативный подход к отраслевой классификации публичных компаний на основе текстовой кластеризации: эмпирическое исследование на данных NASDAQ</trans-title>
</trans-title-group>
</title-group>
<contrib-group>
<contrib contrib-type="author">
<contrib-id contrib-id-type="orcid">https://orcid.org/0000-0001-6860-727X</contrib-id><contrib-id contrib-id-type="spin">6526-2140</contrib-id>
<name-alternatives>
<name xml:lang="en">
<surname>Vetrova</surname>
<given-names>Maria Alexandrovna</given-names>
</name>
<name xml:lang="ru">
<surname>Ветрова</surname>
<given-names>Мария Александровна</given-names>
</name>
</name-alternatives>
<bio xml:lang="ru">
<p>доцент экономического факультета, кандидат экономических наук</p>
</bio>
<email>m.a.vetrova@spbu.ru</email>
<xref ref-type="aff" rid="aff1"/>
</contrib>

<contrib contrib-type="author">
<contrib-id contrib-id-type="orcid">https://orcid.org/0009-0005-9912-8841</contrib-id>
<name-alternatives>
<name xml:lang="en">
<surname>Kuporov</surname>
<given-names>Valerii Stanislavovich</given-names>
</name>
<name xml:lang="ru">
<surname>Купоров</surname>
<given-names>Валерий Станиславович</given-names>
</name>
</name-alternatives>
<bio xml:lang="ru">
<p>Финансовый аналитик</p>
</bio>
<email>valery.kuporov@gmail.com</email>
<xref ref-type="aff" rid="aff2"/>
</contrib>

<contrib contrib-type="author">
<contrib-id contrib-id-type="orcid">https://orcid.org/0009-0000-5861-5364</contrib-id>
<name-alternatives>
<name xml:lang="en">
<surname>Gorkavtsev</surname>
<given-names>Maksim Olegovich</given-names>
</name>
<name xml:lang="ru">
<surname>Горкавцев</surname>
<given-names>Максим Олегович</given-names>
</name>
</name-alternatives>
<bio xml:lang="ru">
<p>Консультант в отделе сопровождения сделок с капиталом</p>
</bio>
<email>gorkavcevmaksim@gmail.com</email>
<xref ref-type="aff" rid="aff3"/>
</contrib>
</contrib-group><aff-alternatives id="aff1">
<aff>
<institution xml:lang="en">St Petersburg State University</institution>
</aff>
<aff>
<institution xml:lang="ru">Санкт-Петербургский государственный университет</institution>
</aff>
</aff-alternatives>        
        <aff-alternatives id="aff2">
<aff>
<institution xml:lang="en">VERSUS</institution>
</aff>
<aff>
<institution xml:lang="ru">VERSUS</institution>
</aff>
</aff-alternatives>        
        <aff-alternatives id="aff3">
<aff>
<institution xml:lang="en">Trust Technologies - Consulting</institution>
</aff>
<aff>
<institution xml:lang="ru">Технологии Доверия - Консультирование</institution>
</aff>
</aff-alternatives>        
        
<pub-date date-type="pub" iso-8601-date="2026-06-30" publication-format="print">
<day>30</day>
<month>06</month>
<year>2026</year>
</pub-date>
<volume>7</volume>
<issue>2</issue>
<issue-title xml:lang="en">VOL 7, NO2 (2026)</issue-title>
<issue-title xml:lang="ru">ТОМ 7, №2 (2026)</issue-title>
<fpage></fpage>
<lpage></lpage>
<history>
<date date-type="received" iso-8601-date="2026-04-21">
<day>21</day>
<month>04</month>
<year>2026</year>
</date>
<date date-type="accepted" iso-8601-date="2026-05-20">
<day>20</day>
<month>05</month>
<year>2026</year>
</date>
</history>

<permissions>
<copyright-statement xml:lang="en">Copyright ©; 2026, Vetrova M.A., Kuporov V.S., Gorkavtsev M.O.</copyright-statement>
<copyright-statement xml:lang="ru">Copyright ©; 2026, Ветрова М.А., Купоров В.С., Горкавцев М.О.</copyright-statement>
<copyright-year>2026</copyright-year>
<copyright-holder xml:lang="en">Vetrova M.A., Kuporov V.S., Gorkavtsev M.O.</copyright-holder>
<copyright-holder xml:lang="ru">Ветрова М.А., Купоров В.С., Горкавцев М.О.</copyright-holder>
<ali:free_to_read xmlns:ali="http://www.niso.org/schemas/ali/1.0/" start_date="2026-06-30"/>
</permissions>



<self-uri xlink:href="https://1economic.ru/lib/126108">https://1economic.ru/lib/126108</self-uri>
<abstract xml:lang="en"><p>Expert industry classifiers (GICS in international markets, the All-Russian Classifier of Economic Activities in the Russian Federation) underlie most empirical research in finance and economics, but they poorly reflect hybrid business models emerging as a result of digital transformation. The article aims to develop and test a reproducible inductive approach to grouping companies based on their textual self–descriptions, optimizing the ratio of clustering quality and computational efficiency. A methodological combination was applied to a sample of 3,380 public companies traded on the NASDAQ stock exchange: Sentence-BERT embeddings (all-MiniLM-L6-v2 model) without additional training, dimensionality reduction using the principal component method and clustering based on the k-means algorithm. The stability of the results was confirmed in three independent ways: by two alternative configurations (without dimensionality reduction and through density clustering UMAP+HDBSCAN), formal metrics of agreement with GICS (adjusted Rand index 0.46, average purity 61.4%) and the financial profile of clusters according to seven indicators that were not involved in the cluster construction. 

Six meaningfully interpreted groups were obtained: two clusters (clinical biotechnologies and commercial banks) reproduce the industry structure with a purity of 97-98%, the other two reveal structural phenomena that are indistinguishable in GICS (shell companies in mergers and acquisitions and a consumer hybrid combining retail with B2C platforms). The proposed methodology is highly reproducible. It does not require additional model training, and it is applicable to refine the industry structure based on data from Russian companies using the All-Russian Classifier of Economic Activities as a reference classification.</p>
</abstract>
<trans-abstract xml:lang="ru"><p>Экспертные отраслевые классификаторы (GICS на международных рынках, ОКВЭД в Российской Федерации) лежат в основе большинства эмпирических исследований в финансах и экономике, однако плохо отражают гибридные бизнес-модели, возникающие в результате цифровой трансформации. Цель исследования – разработать и апробировать воспроизводимый индуктивный подход к группировке компаний на основе их текстовых самоописаний, оптимизирующий соотношение качества кластеризации и вычислительной экономичности. На выборке 3 380 публичных компаний, торгующихся на бирже NASDAQ, применена методологическая связка: эмбеддинги Sentence-BERT (модель all-MiniLM-L6-v2) без дообучения, снижение размерности методом главных компонент и кластеризация алгоритмом k-средних. Устойчивость результатов подтверждена тремя независимыми способами: двумя альтернативными конфигурациями (без снижения размерности и через плотностную кластеризацию UMAP+HDBSCAN), формальными метриками согласия с GICS (скорректированный индекс Рэнда 0,46, средняя чистота 61,4%) и финансовым профилем кластеров по семи показателям, не участвовавшим в построении кластеров. Получены шесть содержательно интерпретируемых групп: два кластера (клинические биотехнологии и коммерческие банки) воспроизводят отраслевую структуру с чистотой 97-98%, два других выявляют структурные феномены, неразличимые в GICS (компании-оболочки в рамках сделок слияния и поглощения и потребительский гибрид, объединяющий ритейл с B2C-платформами). Предложенная методология обладает высокой воспроизводимостью, не требует дообучения модели и применима для уточнения отраслевой структуры на данных российских компаний с использованием ОКВЭД в качестве референсной классификации.</p>
</trans-abstract>
<kwd-group xml:lang="en">
<kwd>industry classification</kwd>
<kwd>text clustering</kwd>
<kwd>Sentence-BERT</kwd>
<kwd>k-means method</kwd>
<kwd>GICS</kwd>
<kwd>digital transformation</kwd></kwd-group><kwd-group xml:lang="ru">
<kwd>отраслевая классификация</kwd>
<kwd>текстовая кластеризация</kwd>
<kwd>Sentence-BERT</kwd>
<kwd>метод k-средних</kwd>
<kwd>GICS</kwd>
<kwd>цифровая трансформация</kwd></kwd-group>
</article-meta>
</front>
<back> <ref-list>
<ref id="B1">
<label>1.</label>
<mixed-citation>1. Горда А.С. Цифровая трансформация бизнес-моделей предприятий в сфере розничной торговли // Омский научный вестник. Серия Общество. История. Современность. – 2025. – № 4. – c. 120-126. – doi: 10.25206/2542-0488-2025-10-4-120-126.</mixed-citation>
</ref>
<ref id="B2">
<label>2.</label>
<mixed-citation>2. Козырь Н.С., Коваленко В.С. Метрика отраслевой классификации в Российской Федерации и за рубежом // Экономический анализ: теория и практика. – 2017. – № 10(469). – c. 1914-1927. – doi: 10.24891/ea.16.10.1914.</mixed-citation>
</ref>
<ref id="B3">
<label>3.</label>
<mixed-citation>3. Макеева Е.Ю., Аршавский И.В. Применение нейронных сетей и семантического анализа для прогнозирования банкротства // Корпоративные финансы. – 2014. – № 4(32). – c. 130-141. – doi: 10.17323/j.jcfr.2073-0438.8.4.2014.130-141.</mixed-citation>
</ref>
<ref id="B4">
<label>4.</label>
<mixed-citation>4. Михненко П.А. Трансформация деловой лексики годовых отчетов крупнейших российских компаний: Data Mining // Управленец. – 2022. – № 5. – c. 17-33. – doi: 10.29141/2218-5003-2022-13-5-2.</mixed-citation>
</ref>
<ref id="B5">
<label>5.</label>
<mixed-citation>5. Федорова Е.А., Сальникова П.А. Влияние раскрытия информации об экологических инициативах на цены акций публичных компаний России // Экономический журнал. – 2024. – № 2. – c. 223-247. – doi: 10.17323/1813-8691-2024-28-2-223-247.</mixed-citation>
</ref>
<ref id="B6">
<label>6.</label>
<mixed-citation>6. Bai H., Xing F. Z., Cambria E., Huang W.-B. Business Taxonomy Construction Using Concept-Level Hierarchical Clustering // Proceedings of the First Workshop on Financial Technology and Natural Language Processing. Macao, China, 2019. – p. 1-7.– doi: 10.48550/arXiv.1906.09694.</mixed-citation>
</ref>
<ref id="B7">
<label>7.</label>
<mixed-citation>7. Devlin J., Chang M.-W., Lee K., Toutanova K. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding // Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies. Minneapolis, Minnesota, 2019. – p. 4171-4186.– doi: 10.18653/v1/N19-1423.</mixed-citation>
</ref>
<ref id="B8">
<label>8.</label>
<mixed-citation>8. Hoberg G., Phillips G. Product Market Synergies and Competition in Mergers and Acquisitions: A Text-Based Analysis // Review of Financial Studies. – 2010. – № 10. – p. 3773-3811. – doi: 10.1093/rfs/hhq053.</mixed-citation>
</ref>
<ref id="B9">
<label>9.</label>
<mixed-citation>9. Hoberg G., Phillips G. Text-Based Network Industries and Endogenous Product Differentiation // Journal of Political Economy. – 2016. – № 5. – p. 1423-1465. – doi: 10.1086/688176.</mixed-citation>
</ref>
<ref id="B10">
<label>10.</label>
<mixed-citation>10. Jagrič T., Herman A. AI Model for Industry Classification Based on Website Data // Information (Switzerland). – 2024. – № 2. – p. 89. – doi: 10.3390/info15020089.</mixed-citation>
</ref>
<ref id="B11">
<label>11.</label>
<mixed-citation>11. Kim D., Kang H.G., Bae K., Jeon S. An artificial intelligence-enabled industry classification and its interpretation // Internet Research: Electronic Networking Applications and Policy. – 2022. – № 2. – p. 406-424. – doi: 10.1108/INTR-05-2020-0299.</mixed-citation>
</ref>
<ref id="B12">
<label>12.</label>
<mixed-citation>12. McInnes L., Healy J., Melville J. UMAP: Uniform Manifold Approximation and Projection for Dimension Reduction. ArXiv preprint. [Электронный ресурс]. URL: https://arxiv.org/abs/1802.03426.</mixed-citation>
</ref>
<ref id="B13">
<label>13.</label>
<mixed-citation>13. Ortakci Ya., Borhan B. Optimizing SBERT for long text clustering: two novel approaches with empirical insights // The Journal of Supercomputing. – 2025. – № 8. – p. 950. – doi: 10.1007/s11227-025-07414-4.</mixed-citation>
</ref>
<ref id="B14">
<label>14.</label>
<mixed-citation>14. Papenkov M., Meredith C., Noel C., Padalkar J., Hendrickson T., Nitiutomo D., Farrell T. Multi-Industry Simplex: A Probabilistic Extension of GICS. arXiv preprint. [Электронный ресурс]. URL: https://arxiv.org/abs/2310.04280.</mixed-citation>
</ref>
<ref id="B15">
<label>15.</label>
<mixed-citation>15. Reimers N., Gurevych I. Sentence-BERT: Sentence Embeddings using Siamese BERT-Networks // Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing. Hong Kong, China, 2019. – p. 3982-3992.– doi: 10.18653/v1/D19-1410.</mixed-citation>
</ref>
<ref id="B16">
<label>16.</label>
<mixed-citation>16. Vamvourellis D., Tóth M., Bhagat S., Desai D., Mehta D., Pasquali S. Company Similarity using Large Language Models. ArXiv preprint. [Электронный ресурс]. URL: https://arxiv.org/abs/2308.08031.</mixed-citation>
</ref>
<ref id="B17">
<label>17.</label>
<mixed-citation>17. Financial statement data for publicly traded companies. Yahoo Finance. [Электронный ресурс]. URL: https://finance.yahoo.com/ (дата обращения: 16.05.2026).</mixed-citation>
</ref>
<ref id="B18">
<label>18.</label>
<mixed-citation>18. Yang H., Lee H. J., Cho S., Cho E. Automatic Classification of Securities using Hierarchical Clustering of the 10-Ks // 2016 IEEE International Conference on Big Data (Big Data). Washington, DC, 2016. – p. 3936-3943.– doi: 10.1109/BigData.2016.7841069.</mixed-citation>
</ref>
</ref-list>
</back>
</article>