@inproceedings{hewapathirana2025domain,
  title={{Domain Adaptation for Multi-document Summarisation: A Case Study in the Medical Research Domain}},
  author={Hewapathirana, K M and de Silva, Nisansa and Athuraliya, C D and Kandanaarachchi, Piumi},
  booktitle={Proceedings of the 39th Pacific Asia Conference on Language, Information and Computation},
  pages={791--802},
  year={2025},
  abstract={Effectively summarising medical research is critical for supporting evidence-based decision making in healthcare. While fine-tuning task-specific models on domain data is established practice, the comparative advantages over increasingly capable general-purpose LLMs remain an open question. This study systematically evaluates domain-adapted PRIMERA against several open-source large language models (LLaMA 3.2 3B, Mistral 7B, OpenChat 7B, and Gemma 7B) in zero-shot settings using the MS^2 dataset, which includes 20, 000 systematic reviews summarising over 470, 000 medical studies. Fine-tuning leads to notable improvements in ROUGE scoresâROUGE-1 from 12.8 to 33.0, ROUGE-2 from 2.0 to 6.5, and ROUGE-L from 8.1 to 22.6. Comparative evaluation indicates that the fine-tuned model consistently achieves stronger performance across all three ROUGE metrics, human evaluations, and LLM-as-a-judge assessments. These results suggest that domain-adapted models can offer advantages over general-purpose LLMs in specialised settings, particularly where factual accuracy and coverage are critical, though at the cost of reduced flexibility across domains.},
  misc={https://aclanthology.org/2025.paclic-1.73/,NLP:BIO:ML,https://goo.gl/iY6aTr}
}
