@mastersthesis{digilib78779, month = {August}, title = {OPTIMALISASI ARSITEKTUR RETRIEVAL-AUGMENTED GENERATION MELALUI FINE-TUNING MODEL EMBEDDING DAN LARGE LANGUAGE MODEL UNTUK ANALISIS NAHWU-SHARAF KITAB KUNING}, school = {UIN SUNAN KALIJAGA YOGYAKARTA}, author = {NIM.: 24206051015 Subhan Ghufron}, year = {2026}, note = {Prof. Dr. Ir. Shofwatul 'Uyun,S.T.,M.Kom., IPM., ASEAN Eng.}, keywords = {Retrieval-Augmented Generation, Nahwu-Sharaf, Kitab Kuning, Fine-tuning, Model Embedding, Large Language Model, BGE-M3, E5-Large, Llama3-8B, Mistral-7B}, url = {https://digilib.uin-suka.ac.id/id/eprint/78779/}, abstract = {Arabic grammatical analysis (Nahwu-Sharaf) of classical Islamic texts (kitab kuning) is a cornerstone of the pesantren education tradition in Indonesia. This process still relies on manual analysis requiring deep expertise and extensive time. This study aims to evaluate the effectiveness of fine-tuning embedding models and Large Language Models (LLMs) within a Retrieval-Augmented Generation (RAG) architecture for automated Nahwu-Sharaf analysis of kitab kuning. Twelve configurations were tested comparatively, involving two embedding models (BGE-M3 and E5-Large) and two LLMs (Llama3-8B and Mistral-7B), each in baseline and fine-tuned conditions using QLoRA on the NahwuAI dataset. Evaluation was conducted using automatic metrics (Recall@K, BLEU, ROUGE-L, F1, Cosine Similarity) and human evaluation by three Arabic grammar experts. Results showed that fine-tuned BGE-M3 outperformed as the embedding model with a Recall@1 of 0.8109 (a 37.7\% increase). Fine-tuned Mistral-7B was selected as the best LLM due to its stronger generalization capability on kitab kuning texts compared to Llama3-8B, which tended to overfit to training data. The RAG + fine-tuned Mistral configuration achieved the highest F1 (0.1176) and the highest expert score (3.80 out of 5). RAG was demonstrated to provide the greatest added value when paired with a model already strengthened through fine-tuning.} }