@inproceedings{10752,
  abstract     = {The digitalization of almost all aspects of our everyday lives has led to unprecedented amounts of data being freely available on the Internet. In particular social media platforms provide rich sources of user-generated data, though typically in unstructured form, and with high diversity, such as written in many different languages. Automatically identifying meaningful information in such big data resources and extracting it efficiently is one of the ongoing challenges of our time. A common step for this is sentiment analysis, which forms the foundation for tasks such as opinion mining or trend prediction. Unfortunately, publicly available tools for this task are almost exclusively available for English-language texts. Consequently, a large fraction of the Internet users, who do not communicate in English, are ignored in automatized studies, a phenomenon called rare-language discrimination.In this work we propose a technique to overcome this problem by a truly multi-lingual model, which can be trained automatically without linguistic knowledge or even the ability to read the many target languages. The main step is to combine self-annotation, specifically the use of emoticons as a proxy for labels, with multi-lingual sentence representations.To evaluate our method we curated several large datasets from data obtained via the free Twitter streaming API. The results show that our proposed multi-lingual training is able to achieve sentiment predictions at the same quality level for rare languages as for frequent ones, and in particular clearly better than what mono-lingual training achieves on the same data. },
  author       = {Lampert, Jasmin and Lampert, Christoph},
  booktitle    = {2021 IEEE International Conference on Big Data},
  isbn         = {9781665439022},
  location     = {Orlando, FL, United States},
  pages        = {5185--5192},
  publisher    = {IEEE},
  title        = {{Overcoming rare-language discrimination in multi-lingual sentiment analysis}},
  doi          = {10.1109/bigdata52589.2021.9672003},
  year         = {2022},
}