@techreport {pub6275,
	title = {LADFA: A Framework of Using Large Language Models and Retrieval-Augmented Generation for Personal Data Flow Analysis in Privacy Policies },
	author = {Haiyue Yuan AND Nikolay Matyunin AND Ali Raza AND Shujun Li},
	year = {2026},
	month = {January},
	abstract = {A privacy policy serves as an essential way to inform consumers about an organisation{\textquoteright}s data practises, including the collection, use, and sharing of personal data. Despite regulatory mandates such as GDPR, these privacy policies often remain difficult for consumers to fully comprehend due to the lengthy and complex legal language and inconsistent implementation. Previous research has applied machine learning and natural language processing techniques to conduct privacy policy analysis to improve the transparency and readability. More recently, researchers have also explored the capabilities of large language models (LLMs) to automate this process. However, limited attention has been paid to the automated extraction of complete data flows involving in data collection and data sharing from privacy policies using LLMs. This paper presents our work on developing an end-to-end framework, consisting of a pre-processor, an LLM-based analyser, and a post-processor, that leverages LLMs to process unstructured text, extract and construct data flows, and conduct analysis for insights discovery. We demonstrated and validated the framework{\textquoteright}s effectiveness and accuracy by conducting a case study that involved examining a number of selected privacy policies from the automotive industry. Moreover, it is worth noting that the framework is designed to be flexible and customisable, making it suitable for a range of text-based analysis tasks beyond privacy policy analysis. Finally, we discuss the limitations of this work and propose directions for future research.},
	publisher = {arXiv},
	booktitle = {arXiv}
}
