
@article{kellyDetectingTextReuse2025,
	title = {Detecting {Text} {Reuse} in {Historical} {Arabic} {Texts}: {Challenges} and {Strategies}},
	volume = {3},
	issn = {2773-2363},
	shorttitle = {Detecting {Text} {Reuse} in {Historical} {Arabic} {Texts}},
	url = {https://brill.com/view/journals/jdir/3/2/article-p362_2.xml},
	doi = {10.1163/27732363-bja00013},
	abstract = {Abstract
            
              Text reuse – including quotation, paraphrase, and allusion – has been a defining feature of Arabic and Islamicate literature since at least the 2nd/8th century. From verbatim transmissions of the Prophet Muḥammad’s sayings to the frequent citation of poetry and historical reports, the recycling of Arabic prose and verse is a widespread literary phenomenon. With the rapid digitization of Arabic texts over the past two decades, this corpus is an ideal candidate for text reuse detection (
              TRD
              ). However, existing
              TRD
              tools – primarily developed for Latin-script languages and distinct textual traditions – yield only partial and often inadequate results when applied to Arabic.
            
            
              This paper introduces alNaql, a new
              TRD
              software specifically designed for Arabic. By incorporating algorithms tailored to Arabic’s unique morphological and syntactic properties, alNaql identifies significantly more reuse instances than currently available tools. We compare its performance to that of passim – which has previously been used by researchers in the Islamicate digital humanities – and TextPAIR, known for its success in European corpora. Our results demonstrate that alNaql not only captures all reuse instances found by passim, the vast majority detected by TextPAIR, but also reveals many matches missed by both. These gains are especially notable in the detection of morphologically varied and syntactically reordered reuse.
            
            This article shows that alNaql represents a major advance for the study of Arabic texts, offering researchers a more nuanced and comprehensive tool to investigate intertextuality and literary transmission across genres. Though it is computationally expensive, alNaql presents new opportunities for digital research into Arabic-Islamicate literature and lays the groundwork for a more detailed corpus-wide reuse analysis.},
	number = {2},
	urldate = {2026-01-10},
	journal = {Journal of Digital Islamicate Research},
	author = {Kelly, Tynan},
	month = nov,
	year = {2025},
	pages = {362--395},
}

@article{jurczykTextMiningTafsir2025,
	title = {Text {Mining} {Tafsir}: {Compilation} and {Preliminary} {Explorations} of a {Curated} {Corpus} of 80 {Qurʾanic} {Commentaries}},
	volume = {3},
	copyright = {Creative Commons Namensnennung - Weitergabe unter gleichen Bedingungen 4.0 Internationale Lizenz},
	url = {https://brill.com/view/journals/jdir/3/1/article-p97_4.xml},
	doi = {10.1163/27732363-bja00010},
	number = {1},
	journal = {Journal of Digital Islamicate Research},
	publisher = {Brill},
	author = {Jurczyk, Thomas and Seidel, Roman and Bernhard, Adrian and Scheffler, Tatjana and Buessow, Johann},
	year = {2025},
	note = {Place: Leiden, The Netherlands},
	pages = {97 -- 167},
	file = {PDF:/home/thomas/Zotero/storage/96BIVBAT/Jurczyk et al. - 2025 - Text Mining Tafsir Compilation and Preliminary Explorations of a Curated Corpus of 80 Qurʾanic Comm.pdf:application/pdf},
}

@article{mosaSynergizingStructureSemantics2025,
	title = {Synergizing structure and semantics: a knowledge graph-transformer framework for narrator disambiguation in hadith networks},
	volume = {40},
	copyright = {https://academic.oup.com/pages/standard-publication-reuse-rights},
	issn = {2055-7671, 2055-768X},
	shorttitle = {Synergizing structure and semantics},
	url = {https://academic.oup.com/dsh/article/40/4/1085/8253513},
	doi = {10.1093/llc/fqaf088},
	abstract = {Abstract
            Historical transmission chains (isnads) are fundamental to verifying authenticity in Hadith literature, yet narrator identity resolution is a persistent challenge due to onomastic ambiguity and complex naming conventions. While traditional methods lack scalability and modern language models overlook crucial network structures, this study bridges the gap by synergizing structural and semantic information. We introduce a novel hybrid framework that integrates a Knowledge Graph (KG) representing the narrator network topology with a Transformer-based model for deep contextual understanding. Our approach first leverages the KG to generate a high-probability set of candidate identities, then employs a hybrid scoring model to evaluate them based on both global network prominence and local semantic compatibility. Evaluated on the AR-Sanad 280K-v2 benchmark, our method establishes a new state-of-the-art, achieving 97.8\% accuracy and significantly outperforming existing baselines. This work provides a scalable, high-fidelity solution for narrator disambiguation, advancing computational methods in Hadith studies and historical identity resolution.},
	language = {en},
	number = {4},
	urldate = {2025-11-19},
	journal = {Digital Scholarship in the Humanities},
	author = {Mosa, Mohamed Atef},
	month = dec,
	year = {2025},
	pages = {1085--1100},
}

@article{bernhardDevelopmentQuranicWAY2024,
	title = {Development of the {Quranic} {WAY} {Metaphor} in {Tafsir}: {A} {Corpus} {Analysis} {Approach}},
	author = {Bernhard, Adrian},
	month = jan,
	year = {2024},
	note = {Num Pages: 133},
	file = {Full Text:/home/thomas/Zotero/storage/PRYM8KGV/Bernhard, Adrian - 108014250077 - MA (1).pdf:application/pdf},
}

@misc{KITABTextReuse,
	title = {{KITAB}: {Text} {Reuse}},
	url = {https://kitab-project.org/methods/text-reuse},
}

@article{al-kabiExtendedTopicalClassification2015,
	title = {Extended {Topical} {Classification} of {Hadith} {Arabic} {Text}},
	volume = {3},
	journal = {International Journal on Islamic Applications in Computer Science And Technology},
	author = {Al-Kabi, Mohammed and Wahsheh, Heider and Alsmadi, Izzat and Al-Akhras, Abdallah},
	month = sep,
	year = {2015},
	pages = {13--24},
}

@article{kirmizialtinExploringGulfManumission2024,
	title = {Exploring {Gulf} {Manumission} {Documents} with {Word} {Vectors}},
	volume = {2},
	copyright = {https://creativecommons.org/licenses/by/4.0/},
	issn = {2773-2355, 2773-2363},
	url = {https://brill.com/view/journals/jdir/2/1-2/article-p1_1.xml},
	doi = {10.1163/27732363-bja00005},
	abstract = {Abstract
            
              In this article we analyze a corpus related to manumission and slavery in the Arabian Gulf in the late nineteenth- and early twentieth-century that we created using Handwritten Text Recognition (
              HTR
              ). The corpus comes from India Office Records (
              IOR
              )
              
                R/15/1/199
                File 5
              
              . Spanning the period from the 1890s to the early 1940s and composed of 977K words, it contains a variety of perspectives on manumission and slavery in the region from manumission requests to administrative documents relevant to colonial approaches to the institution of slavery. We use word2Vec with the WordVectors package in R to highlight how the method can uncover semantic relationships within historical texts, demonstrating some exploratory semantic queries, investigation of word analogies, and vector operations using the corpus content. We argue that advances in applied computer vision such as
              HTR
              are promising for historians working in colonial archives and that while our method is reproducible, there are still issues related to language representation and limitations of scale within smaller datasets. Even though
              HTR
              corpus creation is labor intensive, word vector analysis remains a powerful tool of computational analysis for corpora where
              HTR
              error is present.},
	number = {1-2},
	urldate = {2025-02-22},
	journal = {Journal of Digital Islamicate Research},
	author = {Kirmizialtin, Suphan and Wrisley, David Joseph},
	month = dec,
	year = {2024},
	pages = {1--29},
}

@article{badawyTopicDiscoveryDigital2025,
	title = {Topic {Discovery} in the {Digital} {Quran}: {A} {Text} {Mining} {Approach}},
	volume = {10},
	issn = {2468-4376},
	shorttitle = {Topic {Discovery} in the {Digital} {Quran}},
	url = {https://jisem-journal.com/index.php/journal/article/view/2976},
	doi = {10.52783/jisem.v10i18s.2976},
	abstract = {The research addresses the thematic analysis of the Quran, using for this purpose the Surah Al-Kahf and Surah An-Naml via computational approaches. This work outlines the design of a systematic method for understanding the deeply intricate moral, ethical, and spiritual understandings in those chapters. With the help of a Latent Dirichlet Allocation (LDA)-a type of topic modeling algorithm-the current study will extract and then analyze underlying themes from the chosen surahs. It basically involves text filtering, preprocessing, and tokenization before the application of the LDA algorithm. The identified topics were further validated by the Quranic scholars in order to validate their accuracy and theological consistency. Results show the effectiveness of topic modeling in religious text analysis, providing new insights into Quranic themes. This research not only furthers our understanding of the selected surahs but also provides a framework for applying computational techniques to religious text analysis, bridging traditional Islamic studies with modern data science approaches.},
	number = {18s},
	urldate = {2025-04-04},
	journal = {Journal of Information Systems Engineering and Management},
	author = {Badawy, Amro Ali},
	month = mar,
	year = {2025},
	pages = {642--649},
}

@misc{nigstOpenITIMachineReadableCorpus2023,
	title = {{OpenITI}: a {Machine}-{Readable} {Corpus} of {Islamicate} {Texts}},
	copyright = {Creative Commons Attribution Non Commercial Share Alike 4.0 International},
	shorttitle = {{OpenITI}},
	url = {https://zenodo.org/doi/10.5281/zenodo.10007820},
	doi = {10.5281/ZENODO.10007820},
	abstract = {Co-PIs: Matthew Thomas Miller (University of Maryland, College Park), Maxim G. Romanov (University of Hamburg), Sarah Bowen Savant (Aga Khan University—ISMC, London).

Open Islamicate Texts Initiative (OpenITI, see https://openiti.org/) is a multi-institutional effort to construct the first machine-actionable scholarly corpus of premodern Islamicate texts. Led by researchers at the Aga Khan University, Institute for the Study of Muslim Civilisations (AKU-ISMC), University of Hamburg (UH), and the Roshan Institute for Persian Studies at the University of Maryland (College Park) and an interdisciplinary advisory board of leading digital humanists and Islamic, Persian, and Arabic studies scholars, OpenITI aims to provide the essential textual infrastructure in Arabic, Persian and other Islamicate languages for new forms of textual analysis and digital scholarship. In the process, OpenITI will enable new synergies between Digital Humanities and the inter-related Islamicate fields of Islamic, Persian, and Arabic Studies. In addition to support from the researchers’ home institutions, it is supported by funding from the European Research Council under the European Union’s Horizon 2020 research and innovation programme, awarded to the KITAB project (Grant Agreement No. 772989, PI Sarah Bowen Savant) and the Qatar National Library.

Currently, OpenITI contains almost exclusively Arabic texts, which were first assembled into a corpus within the OpenArabic project, developed first at Tufts University (at The Perseus Project, 2013–2015) and then at Leipzig University (at the Alexander von Humboldt Chair for Digital Humanities, 2015–2017)—in both cases with the support and under the patronage of Prof. Gregory Crane. The much more limited number of Persian texts were compiled during 2015–2016 in the Persian Digital Library (PDL) pilot (see Persian Digital Library by PersDigUMD) at Roshan Institute for Persian Studies at the University of Maryland. These texts have not been made fully compatible with OpenITI mARkdown yet and will be made fully available in next releases.

This release contains all digital versions of the same text that are available in the OpenITI corpus . We also release a 'primary' version of the corpus that contains a single digital version for each text in the corpus that is marked as 'PRI' in the corpus metadata and may be more convenient for some use cases.

Note on Release Numbering: Version 2019.1.1—where 2019 is the year of the release, the first dotted number—.1—is the ordinal release number in 2019, and the second dotted number—.1—is the overall release number; the first dotted number will reset every year, while the second one will continue on increasing.

For more details: https://github.com/OpenITI/RELEASE

Note: In case of any issues with unzipping the files on Windows using built-in utilities, please use free softwares, such as WinRAR and 7zip.},
	urldate = {2025-04-04},
	publisher = {Zenodo},
	author = {Nigst, Lorenz and Romanov, Maxim and Savant, Sarah Bowen and Seydi, Masoumeh and Verkinderen, Peter and Hakimi, Hamidreza},
	month = oct,
	year = {2023},
	keywords = {Arabic; Classical Arabic; Corpus, Classical Arabic courpus, Islamicate texts},
}

@article{alraddadiAntiIslamicArabicText2021,
	title = {Anti-{Islamic} {Arabic} {Text} {Categorization} using {Text} {Mining} and {Sentiment} {Analysis} {Techniques}},
	volume = {12},
	issn = {21565570, 2158107X},
	url = {http://thesai.org/Publications/ViewPaper?Volume=12&Issue=8&Code=IJACSA&SerialNo=89},
	doi = {10.14569/IJACSA.2021.0120889},
	language = {en},
	number = {8},
	urldate = {2025-04-04},
	journal = {International Journal of Advanced Computer Science and Applications},
	author = {Alraddadi, Rawan Abdullah and Ghembaza, Moulay Ibrahim El-Khalil},
	year = {2021},
	file = {Volltext:/home/thomas/Zotero/storage/K38EZI6I/Alraddadi und Ghembaza - 2021 - Anti-Islamic Arabic Text Categorization using Text Mining and Sentiment Analysis Techniques.pdf:application/pdf},
}

@inproceedings{antounAraBERTTransformerbasedModel2020,
	address = {Marseille, France},
	title = {{AraBERT}: {Transformer}-based {Model} for {Arabic} {Language} {Understanding}},
	isbn = {979-10-95546-51-1},
	url = {https://aclanthology.org/2020.osact-1.2/},
	abstract = {The Arabic language is a morphologically rich language with relatively few resources and a less explored syntax compared to English. Given these limitations, Arabic Natural Language Processing (NLP) tasks like Sentiment Analysis (SA), Named Entity Recognition (NER), and Question Answering (QA), have proven to be very challenging to tackle. Recently, with the surge of transformers based models, language-specific BERT based models have proven to be very efficient at language understanding, provided they are pre-trained on a very large corpus. Such models were able to set new standards and achieve state-of-the-art results for most NLP tasks. In this paper, we pre-trained BERT specifically for the Arabic language in the pursuit of achieving the same success that BERT did for the English language. The performance of AraBERT is compared to multilingual BERT from Google and other state-of-the-art approaches. The results showed that the newly developed AraBERT achieved state-of-the-art performance on most tested Arabic NLP tasks. The pretrained araBERT models are publicly available on https://github.com/aub-mind/araBERT hoping to encourage research and applications for Arabic NLP.},
	language = {eng},
	booktitle = {Proceedings of the 4th {Workshop} on {Open}-{Source} {Arabic} {Corpora} and {Processing} {Tools}, with a {Shared} {Task} on {Offensive} {Language} {Detection}},
	publisher = {European Language Resource Association},
	author = {Antoun, Wissam and Baly, Fady and Hajj, Hazem},
	editor = {Al-Khalifa, Hend and Magdy, Walid and Darwish, Kareem and Elsayed, Tamer and Mubarak, Hamdy},
	month = may,
	year = {2020},
	pages = {9--15},
}

@article{sabbehComparativeAnalysisWord2023,
	title = {A {Comparative} {Analysis} of {Word} {Embedding} and {Deep} {Learning} for {Arabic} {Sentiment} {Classification}},
	volume = {12},
	copyright = {https://creativecommons.org/licenses/by/4.0/},
	issn = {2079-9292},
	url = {https://www.mdpi.com/2079-9292/12/6/1425},
	doi = {10.3390/electronics12061425},
	abstract = {Sentiment analysis on social media platforms (i.e., Twitter or Facebook) has become an important tool to learn about users’ opinions and preferences. However, the accuracy of sentiment analysis is disrupted by the challenges of natural language processing (NLP). Recently, deep learning models have proved superior performance over statistical- and lexical-based approaches in NLP-related tasks. Word embedding is an important layer of deep learning models to generate input features. Many word embedding models have been presented for text representation of both classic and context-based word embeddings. In this paper, we present a comparative analysis to evaluate both classic and contextualized word embeddings for sentiment analysis. The four most frequently used word embedding techniques were used in their trained and pre-trained versions. The selected embedding represents classical and contextualized techniques. Classical word embedding includes algorithms such as GloVe, Word2vec, and FastText. By contrast, ARBERT is used as a contextualized embedding model. Since word embedding is more typically employed as the input layer in deep networks, we used deep learning architectures BiLSTM and CNN for sentiment classification. To achieve these goals, the experiments were applied to a series of benchmark datasets: HARD, Khooli, AJGT, ArSAS, and ASTD. Finally, a comparative analysis was conducted on the results obtained for the experimented models. Our outcomes indicate that, generally, generated embedding by one technique achieves higher performance than its pretrained version for the same technique by around 0.28 to 1.8\% accuracy, 0.33 to 2.17\% precision, and 0.44 to 2\% recall. Moreover, the contextualized transformer-based embedding model BERT achieved the highest performance in its pretrained and trained versions. Additionally, the results indicate that BiLSTM outperforms CNN by approximately 2\% in 3 datasets, HARD, Khooli, and ArSAS, while CNN achieved around 2\% higher performance in the smaller datasets, AJGT and ASTD.},
	language = {en},
	number = {6},
	urldate = {2025-03-10},
	journal = {Electronics},
	author = {Sabbeh, Sahar F. and Fasihuddin, Heba A.},
	month = mar,
	year = {2023},
	pages = {1425},
}

@book{elazizRecentAdvancesNLP2020,
	address = {Cham},
	series = {Studies in computational intelligence},
	title = {Recent advances in {NLP}: the case of {Arabic} language},
	isbn = {978-3-030-34614-0 978-3-030-34613-3},
	shorttitle = {Recent advances in {NLP}},
	language = {eng},
	number = {volume 874},
	publisher = {Springer},
	editor = {Elaziz, Mohamed Abd and Al-qaness, Mohammed A. A. and Ewees, Ahmed A. and Dagou, Abdelghani},
	year = {2020},
	file = {Table of Contents PDF:/home/thomas/Zotero/storage/I2BPBCRV/Elaziz et al. - 2020 - Recent advances in NLP the case of Arabic languag.pdf:application/pdf},
}

@incollection{wardiniArabicComputationalLinguistics2022,
	address = {Cham},
	title = {Arabic {Computational} {Linguistics}: {Potential}, {Pitfalls} and {Challenges}},
	volume = {999},
	isbn = {978-3-030-90137-0 978-3-030-90138-7},
	shorttitle = {Arabic {Computational} {Linguistics}},
	url = {https://link.springer.com/10.1007/978-3-030-90138-7_4},
	doi = {10.1007/978-3-030-90138-7_4},
	language = {en},
	urldate = {2025-02-03},
	booktitle = {Natural {Language} {Processing} in {Artificial} {Intelligence} — {NLPinAI} 2021},
	publisher = {Springer International Publishing},
	author = {Wardini, Elie},
	editor = {Loukanova, Roussanka},
	year = {2022},
	note = {Series Title: Studies in Computational Intelligence},
	pages = {105--117},
}

@inproceedings{khondaker-etal-2023-gptaraeval,
	address = {Singapore},
	title = {{GPTAraEval}: {A} {Comprehensive} {Evaluation} of {ChatGPT} on {Arabic} {NLP}},
	url = {https://aclanthology.org/2023.emnlp-main.16},
	doi = {10.18653/v1/2023.emnlp-main.16},
	abstract = {ChatGPT's emergence heralds a transformative phase in NLP, particularly demonstrated through its excellent performance on many English benchmarks. However, the model's efficacy across diverse linguistic contexts remains largely uncharted territory. This work aims to bridge this knowledge gap, with a primary focus on assessing ChatGPT's capabilities on Arabic languages and dialectal varieties. Our comprehensive study conducts a large-scale automated and human evaluation of ChatGPT, encompassing 44 distinct language understanding and generation tasks on over 60 different datasets. To our knowledge, this marks the first extensive performance analysis of ChatGPT's deployment in Arabic NLP. Our findings indicate that, despite its remarkable performance in English, ChatGPT is consistently surpassed by smaller models that have undergone finetuning on Arabic. We further undertake a meticulous comparison of ChatGPT and GPT-4's Modern Standard Arabic (MSA) and Dialectal Arabic (DA), unveiling the relative shortcomings of both models in handling Arabic dialects compared to MSA. Although we further explore and confirm the utility of employing GPT-4 as a potential alternative for human evaluation, our work adds to a growing body of research underscoring the limitations of ChatGPT.},
	booktitle = {Proceedings of the 2023 conference on empirical methods in natural language processing},
	publisher = {Association for Computational Linguistics},
	author = {Khondaker, Md Tawkat Islam and Waheed, Abdul and Nagoudi, El Moatez Billah and Abdul-Mageed, Muhammad},
	editor = {Bouamor, Houda and Pino, Juan and Bali, Kalika},
	month = dec,
	year = {2023},
	pages = {220--247},
}

@book{lachkarArabicLanguageProcessing2018,
	address = {Cham},
	series = {Communications in {Computer} and {Information} {Science}},
	title = {Arabic {Language} {Processing}: {From} {Theory} to {Practice}},
	volume = {782},
	copyright = {http://www.springer.com/tdm},
	isbn = {978-3-319-73499-6 978-3-319-73500-9},
	shorttitle = {Arabic {Language} {Processing}},
	url = {http://link.springer.com/10.1007/978-3-319-73500-9},
	doi = {10.1007/978-3-319-73500-9},
	urldate = {2024-11-19},
	publisher = {Springer International Publishing},
	editor = {Lachkar, Abdelmonaime and Bouzoubaa, Karim and Mazroui, Azzedine and Hamdani, Abdelfettah and Lekhouaja, Abdelhak},
	year = {2018},
}

@article{solimanAraVecSetArabic2017,
	title = {{AraVec}: {A} set of {Arabic} {Word} {Embedding} {Models} for use in {Arabic} {NLP}},
	volume = {117},
	issn = {18770509},
	shorttitle = {{AraVec}},
	url = {https://linkinghub.elsevier.com/retrieve/pii/S1877050917321749},
	doi = {10.1016/j.procs.2017.10.117},
	language = {en},
	urldate = {2024-10-30},
	journal = {Procedia Computer Science},
	author = {Soliman, Abu Bakr and Eissa, Kareem and El-Beltagy, Samhaa R.},
	year = {2017},
	pages = {256--265},
}

@article{ayishDigitalHumanitiesArab2025,
	title = {Digital humanities for {Arab} media studies opportunities and challenges},
	volume = {7},
	issn = {2524-7840},
	url = {https://link.springer.com/10.1007/s42803-025-00105-9},
	doi = {10.1007/s42803-025-00105-9},
	language = {en},
	number = {2},
	urldate = {2026-02-10},
	journal = {International Journal of Digital Humanities},
	author = {Ayish, Mohammad},
	month = jul,
	year = {2025},
	pages = {249--266},
}

@article{karamArabicReallyWellResourced2026,
	title = {Is {Arabic} {Really} a {Well}-{Resourced} {Language}? {Digital} {Exploration} of a {Corpus} in {Middle} {Arabic}, a {Family} of {Varieties} {Omitted} by {Arabic} {Natural} {Language} {Processing}},
	shorttitle = {Is {Arabic} {Really} a {Well}-{Resourced} {Language}?},
	url = {https://www.connections.clio-online.net/article/id/fda-161011},
	doi = {10.60693/67A7-WB88},
	language = {en},
	urldate = {2026-03-20},
	publisher = {Connections (Clio-online)},
	author = {Karam, Rimane},
	year = {2026},
	note = {Medium: text/html,application/pdf},
	keywords = {Area Studies, FOS: History and archaeology},
}

@article{saraPersianArabLifeTrajectories2026,
	title = {Persian-{Arab} {Life} {Trajectories} in the {Red} {Sea} (19th-20th {Centuries}): {Tracing} {Mixedness} through {Digital} {Humanities}},
	shorttitle = {Persian-{Arab} {Life} {Trajectories} in the {Red} {Sea} (19th-20th {Centuries})},
	url = {https://www.connections.clio-online.net/article/id/fda-160645},
	doi = {10.60693/9RCJ-RJ41},
	language = {en},
	urldate = {2026-03-20},
	publisher = {Connections (Clio-online)},
	author = {Sara, Zanotta},
	year = {2026},
	note = {Medium: text/html,application/pdf},
	keywords = {Area Studies, FOS: History and archaeology},
}

@incollection{bednarkiewicz_studying_2023,
	address = {Edinburgh},
	title = {Studying hadith commentaries in the digital age},
	isbn = {978-1-4744-6106-1},
	url = {https://doi.org/10.1515/9781474461061-014},
	doi = {doi:10.1515/9781474461061-014},
	urldate = {2026-09-02},
	booktitle = {Hadith commentary},
	publisher = {Edinburgh University Press},
	author = {Bednarkiewicz, Maroussia and Qurboniev, Aslisho and Bossche, Gowaart Van Den},
	editor = {Blecher, Joel and Brinkmann, Stefanie},
	year = {2023},
	note = {tex.booktitle+duplicate-1: Continuity and Change},
	pages = {263--280},
	file = {PDF:/home/thomas/Zotero/storage/28MGSRZ4/Bednarkiewicz et al. - 2023 - Studying hadith commentaries in the digital age.pdf:application/pdf},
}

@inproceedings{obeid_camel_2020,
	address = {Marseille, France},
	title = {{CAMeL} {Tools}: {An} {Open} {Source} {Python} {Toolkit} for {Arabic} {Natural} {Language} {Processing}},
	isbn = {979-10-95546-34-4},
	shorttitle = {{CAMeL} {Tools}},
	url = {https://aclanthology.org/2020.lrec-1.868},
	abstract = {We present CAMeL Tools, a collection of open-source tools for Arabic natural language processing in Python. CAMeL Tools currently provides utilities for pre-processing, morphological modeling, Dialect Identification, Named Entity Recognition and Sentiment Analysis. In this paper, we describe the design of CAMeL Tools and the functionalities it provides.},
	language = {English},
	urldate = {2022-10-05},
	booktitle = {Proceedings of the {Twelfth} {Language} {Resources} and {Evaluation} {Conference}},
	publisher = {European Language Resources Association},
	author = {Obeid, Ossama and Zalmout, Nasser and Khalifa, Salam and Taji, Dima and Oudah, Mai and Alhafni, Bashar and Inoue, Go and Eryani, Fadhl and Erdmann, Alexander and Habash, Nizar},
	month = may,
	year = {2020},
	pages = {7022--7032},
	file = {Obeid et al. - 2020 - CAMeL Tools An Open Source Python Toolkit for Ara.pdf:/home/thomas/Zotero/storage/MA3Q5MQM/Obeid et al. - 2020 - CAMeL Tools An Open Source Python Toolkit for Ara.pdf:application/pdf},
}

@inproceedings{shahid_computational_2025,
	title = {Computational {Analysis} of {Quran} {Text} {Using} {Machine} {Learning} and {Large} {Language} {Models}},
	url = {https://ieeexplore.ieee.org/document/10908764/?arnumber=10908764},
	doi = {10.1109/CDMA61895.2025.00009},
	abstract = {The Quran verses are foundational for Muslims worldwide. Significant research has been dedicated to information retrieval (IR) from Quran; however, multiple studies have focused on descriptive analysis and topic modelling of the Quran in Arabic and translated versions. This study presents a comprehensive framework for analysing large textual data using an English translation of the Quran. Initially, it conducts a descriptive analysis of the verses to uncover various features, including readability, word clouds, significant n-grams, and network graphs illustrating word associations. The framework then applies machine learning techniques, specifically clustering models based on numerical vectors from text-embedding-3-large, to identify effective groupings of verses. Additionally, GPT-4-turbo is used for topic modelling within each cluster through prompt engineering, aiming to enhance the understanding of these clusters. The results include statistical information graphs and concise knowledge summaries that are beneficial to both domain experts and wider populace.},
	urldate = {2025-04-04},
	booktitle = {2025 8th {International} {Conference} on {Data} {Science} and {Machine} {Learning} {Applications} ({CDMA})},
	author = {Shahid, Usama and Hussain, Muhammad Zunnurain and Sayers, William},
	month = feb,
	year = {2025},
	keywords = {large language models, Data mining, Machine learning, machine learning, natural language processing, text mining, Natural language processing, Vectors, data science, Multiaccess communication, Numerical models, Prompt engineering, quran, Semantic search, Tag clouds, Translation},
	pages = {18--24},
	file = {IEEE Xplore Abstract Record:/home/thomas/Zotero/storage/WD98N6Q6/10908764.html:text/html;Shahid et al. - 2025 - Computational Analysis of Quran Text Using Machine.pdf:/home/thomas/Zotero/storage/PFVKBFDZ/Shahid et al. - 2025 - Computational Analysis of Quran Text Using Machine.pdf:application/pdf},
}

@article{lange_text_2021,
	title = {Text {Mining} {Islamic} {Law}},
	volume = {28},
	issn = {0928-9380, 1568-5195},
	url = {https://brill.com/view/journals/ils/28/3/article-p234_234.xml},
	doi = {10.1163/15685195-bja10009},
	abstract = {Abstract
            
              Digital humanities has a venerable pedigree, stretching back to the middle of the twentieth century, but despite noteworthy pioneering contributions it has not become a mainstream practice in Islamic Studies. This essay applies humanities computing to the study of Islamic law. We analyze a representative corpus of works of Islamic substantive law (
              furūʿ al-fiqh
              ) from the beginnings of Islamic legal jurisprudence to the early modern period (2nd/8th-13th/19th c.) using several computational tools and methods: text-reuse network analysis based on plain-text annotations and
              html
              tags, clustered frequency-based analysis, word clouds, and topic modeling. Applying machine-guided distant reading to Islamic legal texts over the
              longue-dureé
              , we study (1) the role of the Qurʾān, (2) patterns of normative qualifications (
              aḥkām
              ), and (3) the distribution of topics in our corpus. In certain instances the analysis confirms claims made in the scholarly literature on Islamic law, in other instances it corrects such claims.},
	number = {3},
	urldate = {2022-10-05},
	journal = {Islamic Law and Society},
	author = {Lange, Christian and Latif, Maksim Abdul and Çelik, Yusuf and Lyklema, A. Melle and van Kuppevelt, Dafne E. and van der Zwaan, Janneke},
	month = jul,
	year = {2021},
	keywords = {digital humanities, Islamic law, Qurʾān, schools of law in Islam},
	pages = {234--281},
	file = {Lange et al. - 2021 - Text Mining Islamic Law:/home/thomas/Zotero/storage/YLRDDGI5/Lange et al. - 2021 - Text Mining Islamic Law.pdf:application/pdf;Volltext:/home/thomas/Zotero/storage/TCW25PUR/Lange et al. - 2021 - Text Mining Islamic Law.pdf:application/pdf},
}

@article{guellil_arabic_2021,
	title = {Arabic natural language processing: {An} overview},
	volume = {33},
	issn = {13191578},
	shorttitle = {Arabic natural language processing},
	url = {https://linkinghub.elsevier.com/retrieve/pii/S1319157818310553},
	doi = {10.1016/j.jksuci.2019.02.006},
	abstract = {Arabic is recognised as the 4th most used language of the Internet. Arabic has three main varieties: (1) classical Arabic (CA), (2) Modern Standard Arabic (MSA), (3) Arabic Dialect (AD). MSA and AD could be written either in Arabic or in Roman script (Arabizi), which corresponds to Arabic written with Latin letters, numerals and punctuation. Due to the complexity of this language and the number of corresponding challenges for NLP, many surveys have been conducted, in order to synthesise the work done on Arabic. However these surveys principally focus on two varieties of Arabic (MSA and AD, written in Arabic letters only), they are slightly old (no such survey since 2015) and therefore do not cover recent resources and tools. To bridge the gap, we propose a survey focusing on 90 recent research papers (74\% of which were published after 2015). Our study presents and classiﬁes the work done on the three varieties of Arabic, by concentrating on both Arabic and Arabizi, and associates each work to its publicly available resources whenever available.},
	language = {en},
	number = {5},
	urldate = {2022-10-05},
	journal = {Journal of King Saud University - Computer and Information Sciences},
	author = {Guellil, Imane and Saâdane, Houda and Azouaou, Faical and Gueni, Billel and Nouvel, Damien},
	month = jun,
	year = {2021},
	pages = {497--507},
	file = {Eingereichte Version:/home/thomas/Zotero/storage/AMHKFR5F/Guellil et al. - 2021 - Arabic natural language processing An overview.pdf:application/pdf;Guellil et al. - 2021 - Arabic natural language processing An overview.pdf:/home/thomas/Zotero/storage/7PT79M6Q/Guellil et al. - 2021 - Arabic natural language processing An overview.pdf:application/pdf},
}

@article{ayu_text_2022,
	title = {Text mining approaches for analyzing an {Indonesian} tafseer and translation of the holy {Quran}},
	volume = {25},
	copyright = {Copyright (c) 2022 Institute of Advanced Engineering and Science},
	issn = {2502-4760},
	url = {https://ijeecs.iaescore.com/index.php/IJEECS/article/view/27212},
	doi = {10.11591/ijeecs.v25.i3.pp1469-1480},
	abstract = {The Indonesian tafseer and translation of Holy Quran is an important source of information and knowledge for Indonesian muslims, since not many Indonesian muslims understand Arabic language in the Quran.  However, the tafseer is full of the commentaries and explanation of each surah (chapter) and/or ayah (verse), which form a large document and not so easy to be accessed. Thus, the challenge is how to refer to both tafseer and translation in faster and accurate ways as one needs to always refer to them back and forth. Hence, this study proposes several text mining approaches, i.e.  most frequent words, K-means clustering, and association rules, to analyze an Indonesian tafseer and translation of Quran and provide insights of hidden knowledge and relationships based on statistical information derived from it.   These insights could be useful for muslims in general and for people that doing research in related areas.  This study shows interesting results from combined analysis of the approaches used which can help people accessing information in tafseer more efficiently.  As well, interesting relationships have been drawn from terms in the tafseer which could provide further and deeper knowledge on messages in the Quran.},
	language = {en},
	number = {3},
	urldate = {2025-04-04},
	author = {Ayu, Media Anugerah and Irawan, Edi and Mantoro, Teddy},
	month = mar,
	year = {2022},
	note = {Number: 3},
	keywords = {Association rule, K-Means clustering, Most frequent words mining, Tafseer text mining, Text mining},
	pages = {1469--1480},
	file = {Ayu et al. - 2022 - Text mining approaches for analyzing an Indonesian.pdf:/home/thomas/Zotero/storage/3QBDZGQE/Ayu et al. - 2022 - Text mining approaches for analyzing an Indonesian.pdf:application/pdf},
}
