2026
|
Samet Tenekeci; Efe Sezgin; Selma Tekir A Contrastive Learning Framework for Efficient Viral Escape Prediction Journal Article IEEE Transactions on Computational Biology and Bioinformatics, 23 (3), pp. 1277–1289, 2026, ISSN: 2998-4165. Links | BibTeX @article{Tenekeci2026,
title = {A Contrastive Learning Framework for Efficient Viral Escape Prediction},
author = {Samet Tenekeci and Efe Sezgin and Selma Tekir},
url = {http://dx.doi.org/10.1109/TCBBIO.2026.3674782},
doi = {10.1109/tcbbio.2026.3674782},
issn = {2998-4165},
year = {2026},
date = {2026-01-01},
journal = {IEEE Transactions on Computational Biology and Bioinformatics},
volume = {23},
number = {3},
pages = {1277\textendash1289},
publisher = {Institute of Electrical and Electronics Engineers (IEEE)},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
|
Mustafa İlter; Yasemin Özcan Gönülal; Buket Erşahin; Doğan Evecen; Sezen Karabulut; Selma Tekir; Emre Onuç; İbrahim Berci Cross-individual sentiment analysis for historical text processing: detecting author-based relationships in Ottoman-Turkish memoirs Journal Article Digital Scholarship in the Humanities, 41 (3), pp. 1383-1400, 2026, ISSN: 2055-7671. Abstract | Links | BibTeX @article{10.1093/llc/fqag063,
title = {Cross-individual sentiment analysis for historical text processing: detecting author-based relationships in Ottoman-Turkish memoirs},
author = {Mustafa \.{I}lter and Yasemin \"{O}zcan G\"{o}n\"{u}lal and Buket Er\c{s}ahin and Do\u{g}an Evecen and Sezen Karabulut and Selma Tekir and Emre Onu\c{c} and \.{I}brahim Berci},
url = {https://doi.org/10.1093/llc/fqag063},
doi = {10.1093/llc/fqag063},
issn = {2055-7671},
year = {2026},
date = {2026-01-01},
journal = {Digital Scholarship in the Humanities},
volume = {41},
number = {3},
pages = {1383-1400},
abstract = {Sentiment analysis in digital humanities can reveal interpersonal relationships from large amounts of data across different texts at the same time. This study aims to automatically detect authors’ sentiments toward individuals mentioned in Late Ottoman\textemdashEarly Turkish Republic period memoirs. We focused on two staged pipeline which allows understanding authors’ relationships with other individual personalities. We first fine-tuned BERTurk model for Named Entity Recognition (NER) task to detect individuals in memoirs. Secondly, while the literature acknowledges the challenges of sentiment analysis in historical and literary texts, we further endeavor to detect not only the general sentiments of the given text but also authors’ sentiments toward mentioned individuals. To address this, we experimentally explored possible ways to identify authors’ sentiments toward individuals, namely cross-individual sentiment analysis (CISA), by fine-tuning encoder-based PLMs. Our general sentiment analysis model achieved an F1 score of 0.9262, and our CISA pipeline achieved 0.8705. The framework from NER to sentiment analysis revealed promising results for such tasks, as shown in excerpts from \.{I}brahim Temo’s memoir, subsequently offering the field of digital humanities a framework to analyse interpersonal relationships within large corpora of historical texts.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Sentiment analysis in digital humanities can reveal interpersonal relationships from large amounts of data across different texts at the same time. This study aims to automatically detect authors’ sentiments toward individuals mentioned in Late Ottoman—Early Turkish Republic period memoirs. We focused on two staged pipeline which allows understanding authors’ relationships with other individual personalities. We first fine-tuned BERTurk model for Named Entity Recognition (NER) task to detect individuals in memoirs. Secondly, while the literature acknowledges the challenges of sentiment analysis in historical and literary texts, we further endeavor to detect not only the general sentiments of the given text but also authors’ sentiments toward mentioned individuals. To address this, we experimentally explored possible ways to identify authors’ sentiments toward individuals, namely cross-individual sentiment analysis (CISA), by fine-tuning encoder-based PLMs. Our general sentiment analysis model achieved an F1 score of 0.9262, and our CISA pipeline achieved 0.8705. The framework from NER to sentiment analysis revealed promising results for such tasks, as shown in excerpts from İbrahim Temo’s memoir, subsequently offering the field of digital humanities a framework to analyse interpersonal relationships within large corpora of historical texts. |
2025
|
Ege Yiğit Çelik; Selma Tekir CiteBART: Learning to Generate Citations for Local Citation Recommendation Inproceedings Christodoulopoulos, Christos; Chakraborty, Tanmoy; Rose, Carolyn; Peng, Violet (Ed.): Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing, pp. 1703–1719, Association for Computational Linguistics, Suzhou, China, 2025, ISBN: 979-8-89176-332-6. Abstract | Links | BibTeX @inproceedings{celik-tekir-2025-citebart,
title = {CiteBART: Learning to Generate Citations for Local Citation Recommendation},
author = {Ege Yi{\u{g}}it {\c{C}}elik and Selma Tekir},
editor = {Christos Christodoulopoulos and Tanmoy Chakraborty and Carolyn Rose and Violet Peng},
url = {https://aclanthology.org/2025.emnlp-main.89/},
doi = {10.18653/v1/2025.emnlp-main.89},
isbn = {979-8-89176-332-6},
year = {2025},
date = {2025-01-01},
booktitle = {Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing},
pages = {1703--1719},
publisher = {Association for Computational Linguistics},
address = {Suzhou, China},
abstract = {Local citation recommendation (LCR) suggests a set of papers for a citation placeholder within a given context. This paper introduces CiteBART, citation-specific pre-training within an encoder-decoder architecture, where author-date citation tokens are masked to learn to reconstruct them to fulfill LCR. The global version (CiteBART-Global) extends the local context with the citing paper's title and abstract to enrich the learning signal. CiteBART-Global achieves state-of-the-art performance on LCR benchmarks except for the FullTextPeerRead dataset, which is quite small to see the advantage of generative pre-training. The effect is significant in the larger benchmarks, e.g., Refseer and ArXiv., with the Refseer pre-trained model emerging as the best-performing model. We perform comprehensive experiments, including an ablation study, a qualitative analysis, and a taxonomy of hallucinations with detailed statistics. Our analyses confirm that CiteBART-Global has a cross-dataset generalization capability; the macro hallucination rate (MaHR) at the top-3 predictions is 4%, and when the ground-truth is in the top-k prediction list, the hallucination tendency in the other predictions drops significantly. We publicly share our code, base datasets, global datasets, and pre-trained models to support reproducibility.},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
Local citation recommendation (LCR) suggests a set of papers for a citation placeholder within a given context. This paper introduces CiteBART, citation-specific pre-training within an encoder-decoder architecture, where author-date citation tokens are masked to learn to reconstruct them to fulfill LCR. The global version (CiteBART-Global) extends the local context with the citing paper's title and abstract to enrich the learning signal. CiteBART-Global achieves state-of-the-art performance on LCR benchmarks except for the FullTextPeerRead dataset, which is quite small to see the advantage of generative pre-training. The effect is significant in the larger benchmarks, e.g., Refseer and ArXiv., with the Refseer pre-trained model emerging as the best-performing model. We perform comprehensive experiments, including an ablation study, a qualitative analysis, and a taxonomy of hallucinations with detailed statistics. Our analyses confirm that CiteBART-Global has a cross-dataset generalization capability; the macro hallucination rate (MaHR) at the top-3 predictions is 4%, and when the ground-truth is in the top-k prediction list, the hallucination tendency in the other predictions drops significantly. We publicly share our code, base datasets, global datasets, and pre-trained models to support reproducibility. |
Ali Acar; Selma Tekir Recognition of Counterfactual Statements in Turkish Journal Article ACM Trans. Asian Low-Resour. Lang. Inf. Process., 24 (1), 2025, ISSN: 2375-4699. Abstract | Links | BibTeX @article{acar2025,
title = {Recognition of Counterfactual Statements in Turkish},
author = {Ali Acar and Selma Tekir},
url = {https://doi.org/10.1145/3706105},
doi = {10.1145/3706105},
issn = {2375-4699},
year = {2025},
date = {2025-01-01},
journal = {ACM Trans. Asian Low-Resour. Lang. Inf. Process.},
volume = {24},
number = {1},
publisher = {Association for Computing Machinery},
address = {New York, NY, USA},
abstract = {Counterfactual statements are examples of causal reasoning as they describe events that did not happen and, optionally, those events’ consequences if they happened. SemEval-2020 introduces the counterfactual detection (CFD) task and shares an English dataset. Since then, a set of datasets has been released in English, German, and Japanese as part of Amazon product reviews. This work releases the first Turkish corpus of counterfactuals (TRCD). The data collection process is driven by a clue phrase list of counterfactuals, mainly in the form of verb inflections in Turkish. We use clue phrase-based filtering to collect sentences from the Turkish National Corpus (TNC). On the other hand, half of the collection is subject to random word filtering to avoid selection bias due to clue phrases. After the human annotation process with an Inter Annotator Agreement of 0.65, we have 5000 sentences, of which 12.8% contain counterfactual statements. Furthermore, we provide a comprehensive baseline of transformer-based models by testing the effect of clue phrases, cross-lingual performance comparisons using the available CFD datasets, and zero-shot cross-lingual classification experiments using fine-tuning on the different combinations of the existing datasets. The results confirm that TRCD is compatible with the other CFD datasets. Moreover, fine-tuning a Turkish-specific model (BERTurk) performs better than the multilingual alternatives (mBERT and XLM-R). BERTurk is more robust to clue phrase masking. This result emphasizes the importance of a language-specific tokenizer for contextual understanding, especially for low-resource languages. Finally, our qualitative analysis gives insights into errors by different models.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Counterfactual statements are examples of causal reasoning as they describe events that did not happen and, optionally, those events’ consequences if they happened. SemEval-2020 introduces the counterfactual detection (CFD) task and shares an English dataset. Since then, a set of datasets has been released in English, German, and Japanese as part of Amazon product reviews. This work releases the first Turkish corpus of counterfactuals (TRCD). The data collection process is driven by a clue phrase list of counterfactuals, mainly in the form of verb inflections in Turkish. We use clue phrase-based filtering to collect sentences from the Turkish National Corpus (TNC). On the other hand, half of the collection is subject to random word filtering to avoid selection bias due to clue phrases. After the human annotation process with an Inter Annotator Agreement of 0.65, we have 5000 sentences, of which 12.8% contain counterfactual statements. Furthermore, we provide a comprehensive baseline of transformer-based models by testing the effect of clue phrases, cross-lingual performance comparisons using the available CFD datasets, and zero-shot cross-lingual classification experiments using fine-tuning on the different combinations of the existing datasets. The results confirm that TRCD is compatible with the other CFD datasets. Moreover, fine-tuning a Turkish-specific model (BERTurk) performs better than the multilingual alternatives (mBERT and XLM-R). BERTurk is more robust to clue phrase masking. This result emphasizes the importance of a language-specific tokenizer for contextual understanding, especially for low-resource languages. Finally, our qualitative analysis gives insights into errors by different models. |
2024
|
Mahmut Agral; Selma Tekir Improvements on a Multi-task BERT Model Inproceedings 32nd Signal Processing and Communications Applications Conference,
SIU 2024, Mersin, Turkiye, May 15-18, 2024, pp. 1–4, IEEE, 2024. Links | BibTeX @inproceedings{DBLP:conf/siu/AgralT24,
title = {Improvements on a Multi-task BERT Model},
author = {Mahmut Agral and Selma Tekir},
url = {https://doi.org/10.1109/SIU61531.2024.10600801},
doi = {10.1109/SIU61531.2024.10600801},
year = {2024},
date = {2024-01-01},
booktitle = {32nd Signal Processing and Communications Applications Conference,
SIU 2024, Mersin, Turkiye, May 15-18, 2024},
pages = {1--4},
publisher = {IEEE},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
|
Okan Çiftçi; Fatih Soygazi; Selma Tekir Enrichment of Turkish question answering systems using knowledge graphs Journal Article Turkish J. Electr. Eng. Comput. Sci., 32 (4), pp. 516–533, 2024. Links | BibTeX @article{DBLP:journals/elektrik/CiftciST24,
title = {Enrichment of Turkish question answering systems using knowledge graphs},
author = {Okan \c{C}ift\c{c}i and Fatih Soygazi and Selma Tekir},
url = {https://doi.org/10.55730/1300-0632.4085},
doi = {10.55730/1300-0632.4085},
year = {2024},
date = {2024-01-01},
journal = {Turkish J. Electr. Eng. Comput. Sci.},
volume = {32},
number = {4},
pages = {516--533},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
|
Samet Tenekeci; Selma Tekir Identifying promoter and enhancer sequences by graph convolutional
networks Journal Article Comput. Biol. Chem., 110 , pp. 108040, 2024. Links | BibTeX @article{DBLP:journals/candc/TenekeciT24,
title = {Identifying promoter and enhancer sequences by graph convolutional
networks},
author = {Samet Tenekeci and Selma Tekir},
url = {https://doi.org/10.1016/j.compbiolchem.2024.108040},
doi = {10.1016/J.COMPBIOLCHEM.2024.108040},
year = {2024},
date = {2024-01-01},
journal = {Comput. Biol. Chem.},
volume = {110},
pages = {108040},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
|
2023
|
Selma Tekir; Aybüke Güzel; Samet Tenekeci; Bekir Haman Quote Detection: A New Task and Dataset for NLP Inproceedings Proceedings of the 7th Joint SIGHUM Workshop on Computational Linguistics for Cultural Heritage, Social Sciences, Humanities and Literature, pp. 21–27, Association for Computational Linguistics, 2023. Links | BibTeX @inproceedings{Tekir2023,
title = {Quote Detection: A New Task and Dataset for NLP},
author = {Selma Tekir and Ayb\"{u}ke G\"{u}zel and Samet Tenekeci and Bekir Haman},
url = {http://dx.doi.org/10.18653/v1/2023.latechclfl-1.3},
doi = {10.18653/v1/2023.latechclfl-1.3},
year = {2023},
date = {2023-01-01},
booktitle = {Proceedings of the 7th Joint SIGHUM Workshop on Computational Linguistics for Cultural Heritage, Social Sciences, Humanities and Literature},
pages = {21\textendash27},
publisher = {Association for Computational Linguistics},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
|
2022
|
Ege Yiğit Çelik; Zeynel Orulluoğlu; Rıdvan Mertoğlu; Selma Tekir Asking the Right Questions to Solve Algebraic Word Problems Journal Article Turkish Journal of Electrical Engineering & Computer Sciences, 30 (7), pp. 2672-2687, 2022. Links | BibTeX @article{celik2022,
title = {Asking the Right Questions to Solve Algebraic Word Problems},
author = {Ege Yi\u{g}it \c{C}elik and Zeynel Orulluo\u{g}lu and Rıdvan Merto\u{g}lu and Selma Tekir},
doi = {10.55730/1300-0632.3962},
year = {2022},
date = {2022-11-28},
journal = {Turkish Journal of Electrical Engineering & Computer Sciences},
volume = {30},
number = {7},
pages = {2672-2687},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
|
2021
|
Erhan Sezerer; Samet Tenekeci; Ali Acar; Bora Baloğlu; Selma Tekir Author Reputation Measurement on Question and Answer Sites by the Classification of Author-Generated Content Journal Article International Journal of Software Engineering and Knowledge Engineering, 31 (10), pp. 1421-1445, 2021. Links | BibTeX @article{doi:10.1142/S0218194021500479,
title = {Author Reputation Measurement on Question and Answer Sites by the Classification of Author-Generated Content},
author = { Erhan Sezerer and Samet Tenekeci and Ali Acar and Bora Balo\u{g}lu and Selma Tekir},
url = {https://doi.org/10.1142/S0218194021500479},
doi = {10.1142/S0218194021500479},
year = {2021},
date = {2021-01-01},
journal = {International Journal of Software Engineering and Knowledge Engineering},
volume = {31},
number = {10},
pages = {1421-1445},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
|
Erhan Sezerer; Selma Tekir Incorporating Concreteness in Multi-Modal Language Models with Curriculum Learning Journal Article Applied Sciences, 11 (17), 2021, ISSN: 2076-3417. Links | BibTeX @article{app11178241,
title = {Incorporating Concreteness in Multi-Modal Language Models with Curriculum Learning},
author = { Erhan Sezerer and Selma Tekir},
url = {https://www.mdpi.com/2076-3417/11/17/8241},
doi = {10.3390/app11178241},
issn = {2076-3417},
year = {2021},
date = {2021-01-01},
journal = {Applied Sciences},
volume = {11},
number = {17},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
|
İskender Ülgen Oğul; Selma Tekir Performance Evaluation of BERT Vectors on Natural Language Inference Models Inproceedings 2021 29th Signal Processing and Communications Applications Conference (SIU), pp. 1-4, 2021. Links | BibTeX @inproceedings{9478044,
title = {Performance Evaluation of BERT Vectors on Natural Language Inference Models},
author = { \.{I}skender \"{U}lgen O\u{g}ul and Selma Tekir},
doi = {10.1109/SIU53274.2021.9478044},
year = {2021},
date = {2021-01-01},
booktitle = {2021 29th Signal Processing and Communications Applications Conference (SIU)},
pages = {1-4},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
|
2020
|
Elgun Jabrayilzade; Selma Tekir Lgpsolver-solving logic grid puzzles automatically Inproceedings Findings of the Association for Computational Linguistics: EMNLP 2020, pp. 1118–1123, 2020. BibTeX @inproceedings{jabrayilzade2020lgpsolver,
title = {Lgpsolver-solving logic grid puzzles automatically},
author = { Elgun Jabrayilzade and Selma Tekir},
year = {2020},
date = {2020-01-01},
booktitle = {Findings of the Association for Computational Linguistics: EMNLP 2020},
pages = {1118--1123},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
|
Selma Tekir; Yalin Bastanlar Deep learning: Exemplar studies in natural language processing and computer vision Journal Article Data Mining-Methods, Applications and Systems, 2020. BibTeX @article{tekir2020deep,
title = {Deep learning: Exemplar studies in natural language processing and computer vision},
author = { Selma Tekir and Yalin Bastanlar},
year = {2020},
date = {2020-01-01},
journal = {Data Mining-Methods, Applications and Systems},
publisher = {IntechOpen},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
|
Elgün Jabrayilzade; Algin Poyraz Arslan; Hasan Para; Ozan Polatbilek; Erhan Sezerer; Selma Tekir Çok-etiketli film türü siniflandirmasi için Türkçe konu modellemesi veri kümesi Inproceedings 2020 28th Signal Processing and Communications Applications Conference, SIU 2020-Proceedings, Institute of Electrical and Electronics Engineers 2020. BibTeX @inproceedings{jabrayilzade2020ccok,
title = {\c{C}ok-etiketli film t\"{u}r\"{u} siniflandirmasi i\c{c}in T\"{u}rk\c{c}e konu modellemesi veri k\"{u}mesi},
author = { Elg\"{u}n Jabrayilzade and Algin Poyraz Arslan and Hasan Para and Ozan Polatbilek and Erhan Sezerer and Selma Tekir},
year = {2020},
date = {2020-01-01},
booktitle = {2020 28th Signal Processing and Communications Applications Conference, SIU 2020-Proceedings},
organization = {Institute of Electrical and Electronics Engineers},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
|
Damla Yaşar; Selma Tekir Estimating spatiotemporal focus of documents using entropy with PMI Journal Article Turkish Journal of Electrical Engineering and Computer Sciences, 28 (2), pp. 1070–1085, 2020. BibTeX @article{yacsar2020estimating,
title = {Estimating spatiotemporal focus of documents using entropy with PMI},
author = { Damla Ya\c{s}ar and Selma Tekir},
year = {2020},
date = {2020-01-01},
journal = {Turkish Journal of Electrical Engineering and Computer Sciences},
volume = {28},
number = {2},
pages = {1070--1085},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
|