@article{10.1093/llc/fqag063,
title = {Cross-individual sentiment analysis for historical text processing: detecting author-based relationships in Ottoman-Turkish memoirs},
author = {Mustafa \.{I}lter and Yasemin \"{O}zcan G\"{o}n\"{u}lal and Buket Er\c{s}ahin and Do\u{g}an Evecen and Sezen Karabulut and Selma Tekir and Emre Onu\c{c} and \.{I}brahim Berci},
url = {https://doi.org/10.1093/llc/fqag063},
doi = {10.1093/llc/fqag063},
issn = {2055-7671},
year = {2026},
date = {2026-01-01},
journal = {Digital Scholarship in the Humanities},
volume = {41},
number = {3},
pages = {1383-1400},
abstract = {Sentiment analysis in digital humanities can reveal interpersonal relationships from large amounts of data across different texts at the same time. This study aims to automatically detect authors’ sentiments toward individuals mentioned in Late Ottoman\textemdashEarly Turkish Republic period memoirs. We focused on two staged pipeline which allows understanding authors’ relationships with other individual personalities. We first fine-tuned BERTurk model for Named Entity Recognition (NER) task to detect individuals in memoirs. Secondly, while the literature acknowledges the challenges of sentiment analysis in historical and literary texts, we further endeavor to detect not only the general sentiments of the given text but also authors’ sentiments toward mentioned individuals. To address this, we experimentally explored possible ways to identify authors’ sentiments toward individuals, namely cross-individual sentiment analysis (CISA), by fine-tuning encoder-based PLMs. Our general sentiment analysis model achieved an F1 score of 0.9262, and our CISA pipeline achieved 0.8705. The framework from NER to sentiment analysis revealed promising results for such tasks, as shown in excerpts from \.{I}brahim Temo’s memoir, subsequently offering the field of digital humanities a framework to analyse interpersonal relationships within large corpora of historical texts.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Sentiment analysis in digital humanities can reveal interpersonal relationships from large amounts of data across different texts at the same time. This study aims to automatically detect authors’ sentiments toward individuals mentioned in Late Ottoman—Early Turkish Republic period memoirs. We focused on two staged pipeline which allows understanding authors’ relationships with other individual personalities. We first fine-tuned BERTurk model for Named Entity Recognition (NER) task to detect individuals in memoirs. Secondly, while the literature acknowledges the challenges of sentiment analysis in historical and literary texts, we further endeavor to detect not only the general sentiments of the given text but also authors’ sentiments toward mentioned individuals. To address this, we experimentally explored possible ways to identify authors’ sentiments toward individuals, namely cross-individual sentiment analysis (CISA), by fine-tuning encoder-based PLMs. Our general sentiment analysis model achieved an F1 score of 0.9262, and our CISA pipeline achieved 0.8705. The framework from NER to sentiment analysis revealed promising results for such tasks, as shown in excerpts from İbrahim Temo’s memoir, subsequently offering the field of digital humanities a framework to analyse interpersonal relationships within large corpora of historical texts.
@article{\.{I}\"{U}30.0c,
title = {TurkMedNLI: a Turkish medical natural language inference dataset through large language model based translation},
author = {O\u{g}ul \.{I}\"{U} and Soygazi F and Belgin Ergen\c{c} Bostano\u{g}lu},
doi = {https://doi.org/10.7717/peerj-cs.2662},
year = {2025},
date = {2025-01-30},
journal = {PeerJ Computer Science},
volume = {11},
pages = {e2662},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
@article{tenekeci2025,
title = {Automating software size measurement from python code using language models},
author = {Samet Tenekeci and H\"{u}seyin \"{U}nl\"{u} and Bedir Arda G\"{u}l and Damla Kele\c{s} and Murat K\"{u}\c{c}\"{u}k and Onur Demir\"{o}rs},
url = {https://doi.org/10.1007/s10515-025-00571-z},
doi = {10.1007/s10515-025-00571-z},
year = {2025},
date = {2025-01-01},
journal = {Automated Software Engineering},
volume = {33},
number = {1},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
@article{unlu2025,
title = {Automating software size measurement with language models: Insights from industrial case studies},
author = {H\"{u}seyin \"{U}nl\"{u} and Samet Tenekeci and Dhia Eddine Kennouche and Onur Demir\"{o}rs},
url = {https://doi.org/10.1016/j.jss.2025.112638},
doi = {10.1016/j.jss.2025.112638},
year = {2025},
date = {2025-01-01},
journal = {Journal of Systems and Software},
volume = {231},
pages = {112638},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Christodoulopoulos, Christos; Chakraborty, Tanmoy; Rose, Carolyn; Peng, Violet (Ed.): Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing, pp. 1703–1719, Association for Computational Linguistics, Suzhou, China, 2025, ISBN: 979-8-89176-332-6.
@inproceedings{celik-tekir-2025-citebart,
title = {CiteBART: Learning to Generate Citations for Local Citation Recommendation},
author = {Ege Yi{\u{g}}it {\c{C}}elik and Selma Tekir},
editor = {Christos Christodoulopoulos and Tanmoy Chakraborty and Carolyn Rose and Violet Peng},
url = {https://aclanthology.org/2025.emnlp-main.89/},
doi = {10.18653/v1/2025.emnlp-main.89},
isbn = {979-8-89176-332-6},
year = {2025},
date = {2025-01-01},
booktitle = {Proceedings of the 2025 Conference on Empirical Methods in Natural Language Processing},
pages = {1703--1719},
publisher = {Association for Computational Linguistics},
address = {Suzhou, China},
abstract = {Local citation recommendation (LCR) suggests a set of papers for a citation placeholder within a given context. This paper introduces CiteBART, citation-specific pre-training within an encoder-decoder architecture, where author-date citation tokens are masked to learn to reconstruct them to fulfill LCR. The global version (CiteBART-Global) extends the local context with the citing paper's title and abstract to enrich the learning signal. CiteBART-Global achieves state-of-the-art performance on LCR benchmarks except for the FullTextPeerRead dataset, which is quite small to see the advantage of generative pre-training. The effect is significant in the larger benchmarks, e.g., Refseer and ArXiv., with the Refseer pre-trained model emerging as the best-performing model. We perform comprehensive experiments, including an ablation study, a qualitative analysis, and a taxonomy of hallucinations with detailed statistics. Our analyses confirm that CiteBART-Global has a cross-dataset generalization capability; the macro hallucination rate (MaHR) at the top-3 predictions is 4%, and when the ground-truth is in the top-k prediction list, the hallucination tendency in the other predictions drops significantly. We publicly share our code, base datasets, global datasets, and pre-trained models to support reproducibility.},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
Local citation recommendation (LCR) suggests a set of papers for a citation placeholder within a given context. This paper introduces CiteBART, citation-specific pre-training within an encoder-decoder architecture, where author-date citation tokens are masked to learn to reconstruct them to fulfill LCR. The global version (CiteBART-Global) extends the local context with the citing paper's title and abstract to enrich the learning signal. CiteBART-Global achieves state-of-the-art performance on LCR benchmarks except for the FullTextPeerRead dataset, which is quite small to see the advantage of generative pre-training. The effect is significant in the larger benchmarks, e.g., Refseer and ArXiv., with the Refseer pre-trained model emerging as the best-performing model. We perform comprehensive experiments, including an ablation study, a qualitative analysis, and a taxonomy of hallucinations with detailed statistics. Our analyses confirm that CiteBART-Global has a cross-dataset generalization capability; the macro hallucination rate (MaHR) at the top-3 predictions is 4%, and when the ground-truth is in the top-k prediction list, the hallucination tendency in the other predictions drops significantly. We publicly share our code, base datasets, global datasets, and pre-trained models to support reproducibility.
@article{acar2025,
title = {Recognition of Counterfactual Statements in Turkish},
author = {Ali Acar and Selma Tekir},
url = {https://doi.org/10.1145/3706105},
doi = {10.1145/3706105},
issn = {2375-4699},
year = {2025},
date = {2025-01-01},
journal = {ACM Trans. Asian Low-Resour. Lang. Inf. Process.},
volume = {24},
number = {1},
publisher = {Association for Computing Machinery},
address = {New York, NY, USA},
abstract = {Counterfactual statements are examples of causal reasoning as they describe events that did not happen and, optionally, those events’ consequences if they happened. SemEval-2020 introduces the counterfactual detection (CFD) task and shares an English dataset. Since then, a set of datasets has been released in English, German, and Japanese as part of Amazon product reviews. This work releases the first Turkish corpus of counterfactuals (TRCD). The data collection process is driven by a clue phrase list of counterfactuals, mainly in the form of verb inflections in Turkish. We use clue phrase-based filtering to collect sentences from the Turkish National Corpus (TNC). On the other hand, half of the collection is subject to random word filtering to avoid selection bias due to clue phrases. After the human annotation process with an Inter Annotator Agreement of 0.65, we have 5000 sentences, of which 12.8% contain counterfactual statements. Furthermore, we provide a comprehensive baseline of transformer-based models by testing the effect of clue phrases, cross-lingual performance comparisons using the available CFD datasets, and zero-shot cross-lingual classification experiments using fine-tuning on the different combinations of the existing datasets. The results confirm that TRCD is compatible with the other CFD datasets. Moreover, fine-tuning a Turkish-specific model (BERTurk) performs better than the multilingual alternatives (mBERT and XLM-R). BERTurk is more robust to clue phrase masking. This result emphasizes the importance of a language-specific tokenizer for contextual understanding, especially for low-resource languages. Finally, our qualitative analysis gives insights into errors by different models.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Counterfactual statements are examples of causal reasoning as they describe events that did not happen and, optionally, those events’ consequences if they happened. SemEval-2020 introduces the counterfactual detection (CFD) task and shares an English dataset. Since then, a set of datasets has been released in English, German, and Japanese as part of Amazon product reviews. This work releases the first Turkish corpus of counterfactuals (TRCD). The data collection process is driven by a clue phrase list of counterfactuals, mainly in the form of verb inflections in Turkish. We use clue phrase-based filtering to collect sentences from the Turkish National Corpus (TNC). On the other hand, half of the collection is subject to random word filtering to avoid selection bias due to clue phrases. After the human annotation process with an Inter Annotator Agreement of 0.65, we have 5000 sentences, of which 12.8% contain counterfactual statements. Furthermore, we provide a comprehensive baseline of transformer-based models by testing the effect of clue phrases, cross-lingual performance comparisons using the available CFD datasets, and zero-shot cross-lingual classification experiments using fine-tuning on the different combinations of the existing datasets. The results confirm that TRCD is compatible with the other CFD datasets. Moreover, fine-tuning a Turkish-specific model (BERTurk) performs better than the multilingual alternatives (mBERT and XLM-R). BERTurk is more robust to clue phrase masking. This result emphasizes the importance of a language-specific tokenizer for contextual understanding, especially for low-resource languages. Finally, our qualitative analysis gives insights into errors by different models.
@article{10.1093/bioinformatics/btaf468,
title = {PRISM: privacy-preserving rare disease analysis using fully homomorphic encryption},
author = {G\"{u}liz Akkaya and Nesli Erdo\u{g}mu\c{s} and Mete Akg\"{u}n},
url = {https://doi.org/10.1093/bioinformatics/btaf468},
doi = {10.1093/bioinformatics/btaf468},
issn = {1367-4811},
year = {2025},
date = {2025-01-01},
journal = {Bioinformatics},
volume = {41},
number = {10},
pages = {btaf468},
abstract = {Rare diseases affect millions of people worldwide, yet their genomic foundations remain poorly understood due to limited patient data and strict privacy regulations, such as the General Data Protection Regulation (GDPR) (https://gdpr.eu/tag/gdpr/) in March 2025. These restrictions can hinder the collaborative analysis of genomic data necessary for uncovering disease-causing variants.We present PRISM, a novel privacy-preserving framework based on fully homomorphic encryption (FHE) that facilitates rare disease variant analysis across multiple institutions without exposing sensitive genomic information. To address the challenges of centralized trust, PRISM is built upon a Threshold FHE scheme. This approach decentralizes key management across participating institutions and ensures no single entity can unilaterally decrypt sensitive data. Our method filters disease-causing variants under recessive, dominant, and de novo inheritance models entirely on encrypted data. We propose two algorithmic variants: a multiplication-intensive (MUL-IN) approach and an addition-intensive (ADD-IN) approach. The ADD-IN algorithms minimize the number of costly multiplication operations, enabling up to a 17× improvement in runtime for recessive/dominant filtering and 22× for de novo filtering, compared to MUL-IN methods. While ADD-IN produces larger ciphertexts, efficient parallelization via SIMD and multithreading allows it to handle millions of variants in reasonable time. To the best of our knowledge, this is the first study that utilizes FHE for privacy-preserving rare disease analysis across multiple inheritance models, demonstrating its practicality and scalability in a single-cloud setting.The source code and the data used in this work can be found in https://github.com/mdppml/PRISM.git.},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Rare diseases affect millions of people worldwide, yet their genomic foundations remain poorly understood due to limited patient data and strict privacy regulations, such as the General Data Protection Regulation (GDPR) (https://gdpr.eu/tag/gdpr/) in March 2025. These restrictions can hinder the collaborative analysis of genomic data necessary for uncovering disease-causing variants.We present PRISM, a novel privacy-preserving framework based on fully homomorphic encryption (FHE) that facilitates rare disease variant analysis across multiple institutions without exposing sensitive genomic information. To address the challenges of centralized trust, PRISM is built upon a Threshold FHE scheme. This approach decentralizes key management across participating institutions and ensures no single entity can unilaterally decrypt sensitive data. Our method filters disease-causing variants under recessive, dominant, and de novo inheritance models entirely on encrypted data. We propose two algorithmic variants: a multiplication-intensive (MUL-IN) approach and an addition-intensive (ADD-IN) approach. The ADD-IN algorithms minimize the number of costly multiplication operations, enabling up to a 17× improvement in runtime for recessive/dominant filtering and 22× for de novo filtering, compared to MUL-IN methods. While ADD-IN produces larger ciphertexts, efficient parallelization via SIMD and multithreading allows it to handle millions of variants in reasonable time. To the best of our knowledge, this is the first study that utilizes FHE for privacy-preserving rare disease analysis across multiple inheritance models, demonstrating its practicality and scalability in a single-cloud setting.The source code and the data used in this work can be found in https://github.com/mdppml/PRISM.git.
@article{kaya2024compiler,
title = {Compiler-Managed Replication of CUDA Kernels for Reliable Execution of GPGPU Applications},
author = { Erc\"{u}ment Kaya and I{\c{s}}il \"{O}z},
year = {2024},
date = {2024-01-01},
journal = {Journal of Circuits, Systems and Computers},
publisher = {World Scientific},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
@article{Tekin2024,
title = {Edge Deletion Based Subgraph Hiding},
author = {Leyla Tekin and Belgin Ergen\c{c} Bostano\u{g}lu},
doi = {10.37394/23209.2024.21.32},
year = {2024},
date = {2024-01-01},
journal = {WSEAS Transactions on Information Science and Applications},
volume = {21},
pages = {2224-3402},
organization = {WSEAS Transactions on Information Science and Applications},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
@inproceedings{nl2024,
title = {Predicting Software Functional Size Using Natural Language Processing: An Exploratory Case Study},
author = {H\"{u}seyin \"{U}nl\"{u} and Samet Tenekeci and Can \c{C}iftci and \.{I}brahim Baran Oral and Tunahan Atalay and Tuna Hacalo\u{g}lu and Burcu Musao\u{g}lu and Onur Demir\"{o}rs},
url = {http://dx.doi.org/10.1109/SEAA64295.2024.00036},
doi = {10.1109/seaa64295.2024.00036},
year = {2024},
date = {2024-01-01},
booktitle = {2024 50th Euromicro Conference on Software Engineering and Advanced Applications (SEAA)},
pages = {188\textendash193},
publisher = {IEEE},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
@inproceedings{yilmaz2023open,
title = {Open-Source Visual Target-Tracking System Both on Simulation Environment and Real Unmanned Aerial Vehicles},
author = { Celil Y{i}lmaz and Abdulkadir Ozgun and Berat A Erol and Abdurrahman Gumus},
year = {2023},
date = {2023-01-01},
booktitle = {International Congress of Electrical and Computer
Engineering},
pages = {147--159},
organization = {Springer},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
@inproceedings{aksoy2023real,
title = {Real Time Computer Vision Based Robotic Arm Controller with ROS and Gazebo Simulation Environment},
author = { Egemen Aksoy and Arif Dorukan {\c{C}}a\k{i}r and Berat A Erol and Abdurrahman Gumus},
year = {2023},
date = {2023-01-01},
booktitle = {2023 14th International Conference on Electrical and
Electronics Engineering (ELECO)},
pages = {1--5},
organization = {IEEE},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
@inproceedings{soygazi2023analysis,
title = {An Analysis of Large Language Models and LangChain in Mathematics
Education},
author = { Fatih Soygazi and Damla Oguz},
year = {2023},
date = {2023-01-01},
booktitle = {Proceedings of the 2023 7th International Conference on
Advances in Artificial Intelligence},
pages = {92--97},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
@article{gyires2023teaching,
title = {Teaching Accelerated Computing and Deep Learning at a Large-Scale
with the NVIDIA Deep Learning Institute},
author = { B\'{a}lint Gyires-T\'{o}th and I{\c{s}}{i}l \"{O}z and Joe Bungo},
year = {2023},
date = {2023-01-01},
journal = {Journal of Computational Science},
volume = {14},
number = {1},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
@article{gyires2023teachingb,
title = {Teaching Accelerated Computing and Deep Learning at a Large-Scale
with the NVIDIA Deep Learning Institute},
author = { B\'{a}lint Gyires-T\'{o}th and I{\c{s}}{i}l \"{O}z and Joe Bungo},
year = {2023},
date = {2023-01-01},
journal = {Journal of Computational Science},
volume = {14},
number = {1},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
@inproceedings{soylu2023size,
title = {Size Measurement and Effort Estimation in Microservice-based
Projects: Results from Pakistan.},
author = { G\"{o}rkem Kilin{\c{c}} Soylu and H\"{u}seyin \"{U}nl\"{u} and Isra Shafique Ahmad and Onur Demir\"{o}rs},
year = {2023},
date = {2023-01-01},
booktitle = {IWSM-Mensura},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
@article{yilmaz2023bim,
title = {BIM-CAREM: Assessing the BIM capabilities of design, construction
and facilities management processes in the construction industry},
author = { Gokcen Yilmaz and Asli Akcamete and Onur Demirors},
year = {2023},
date = {2023-01-01},
journal = {Computers in Industry},
volume = {147},
pages = {103861},
publisher = {Elsevier},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Hüseyin Ünlü; Ozan Raşit Yürüm; Özden Özcan-Top; Onur Demirörs
How Software Practitioners Perceive Work-Related Barriers and
Benefits Based on their Educational Background: Insights from a Survey
Study Journal Article
@article{unlu2023software,
title = {How Software Practitioners Perceive Work-Related Barriers and
Benefits Based on their Educational Background: Insights from a Survey
Study},
author = { H\"{u}seyin \"{U}nl\"{u} and Ozan Ra{\c{s}}it Y\"{u}r\"{u}m and \"{O}zden \"{O}zcan-Top and Onur Demir\"{o}rs},
year = {2023},
date = {2023-01-01},
journal = {IEEE Software},
publisher = {IEEE},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
@inproceedings{unlu2023exploratory,
title = {An exploratory case study on effort estimation in microservices},
author = { H\"{u}seyin \"{U}nl\"{u} and Tuna Hacalo{\u{g}}lu and Neslihan K\"{u}{\c{c}}\"{u}kate{\c{s}} \"{O}m\"{u}ral and Neslihan {\c{C}}ali{\c{s}}kanel and Onur Leblebici and Onur Demir\"{o}rs},
year = {2023},
date = {2023-01-01},
booktitle = {2023 49th Euromicro Conference on Software Engineering and
Advanced Applications (SEAA)},
pages = {215--218},
organization = {IEEE},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
How Software Practitioners Perceive Work-Related Barriers and
Benefits Based on their Educational Background: Insights from a Survey
Study Journal Article
@article{unlu2023readiness,
title = {Readiness and maturity models for Industry 4.0: A systematic
literature review},
author = { H\"{u}seyin \"{U}nl\"{u} and Onur Demir\"{o}rs and Vahid Garousi},
year = {2023},
date = {2023-01-01},
journal = {Journal of Software: Evolution and Process},
pages = {e2641},
publisher = {Wiley Online Library},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
Proceedings of the 7th Joint SIGHUM Workshop on Computational Linguistics for Cultural Heritage, Social Sciences, Humanities and Literature, pp. 21–27, Association for Computational Linguistics, 2023.
@inproceedings{Tekir2023,
title = {Quote Detection: A New Task and Dataset for NLP},
author = {Selma Tekir and Ayb\"{u}ke G\"{u}zel and Samet Tenekeci and Bekir Haman},
url = {http://dx.doi.org/10.18653/v1/2023.latechclfl-1.3},
doi = {10.18653/v1/2023.latechclfl-1.3},
year = {2023},
date = {2023-01-01},
booktitle = {Proceedings of the 7th Joint SIGHUM Workshop on Computational Linguistics for Cultural Heritage, Social Sciences, Humanities and Literature},
pages = {21\textendash27},
publisher = {Association for Computational Linguistics},
keywords = {},
pubstate = {published},
tppubtype = {inproceedings}
}
@article{celik2022,
title = {Asking the Right Questions to Solve Algebraic Word Problems},
author = {Ege Yi\u{g}it \c{C}elik and Zeynel Orulluo\u{g}lu and Rıdvan Merto\u{g}lu and Selma Tekir},
doi = {10.55730/1300-0632.3962},
year = {2022},
date = {2022-11-28},
journal = {Turkish Journal of Electrical Engineering & Computer Sciences},
volume = {30},
number = {7},
pages = {2672-2687},
keywords = {},
pubstate = {published},
tppubtype = {article}
}
@article{caner2022performance,
title = {Performance analysis and feature selection for network-based intrusion detectionwith deep learning},
author = { Serhat Caner and Nesli Erdo\u{g}mu\c{s} and Yusuf Murat Erten},
year = {2022},
date = {2022-01-01},
journal = {Turkish Journal of Electrical Engineering and Computer Sciences},
volume = {30},
number = {3},
pages = {629--643},
keywords = {},
pubstate = {published},
tppubtype = {article}
}