{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,24]],"date-time":"2026-03-24T00:10:50Z","timestamp":1774311050828,"version":"3.50.1"},"publisher-location":"Cham","reference-count":28,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032213204","type":"print"},{"value":"9783032213211","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-3-032-21321-1_17","type":"book-chapter","created":{"date-parts":[[2026,3,23]],"date-time":"2026-03-23T11:07:27Z","timestamp":1774264047000},"page":"118-125","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Accelerating Personalization Signal Learning via\u00a0Synthetic Data"],"prefix":"10.1007","author":[{"given":"Daraksha","family":"Parveen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Doug","family":"Kang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Anwitha","family":"Paruchuri","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Deep","family":"Kayal","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pavan","family":"Mallapragada","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,24]]},"reference":[{"issue":"7","key":"17_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3544558","volume":"55","author":"M Bayer","year":"2022","unstructured":"Bayer, M., Kaufhold, M.A., Reuter, C.: A survey on data augmentation for text classification. ACM Comput. Surv. 55(7), 1\u201339 (2022)","journal-title":"ACM Comput. Surv."},{"key":"17_CR2","doi-asserted-by":"publisher","unstructured":"Bukharin, A., et al.: Data diversity matters for robust instruction tuning. In: Findings of the Association for Computational Linguistics: EMNLP 2024, pp. 3411\u20133425. Association for Computational Linguistics (2024). https:\/\/doi.org\/10.18653\/v1\/2024.findings-emnlp.195","DOI":"10.18653\/v1\/2024.findings-emnlp.195"},{"key":"17_CR3","doi-asserted-by":"publisher","unstructured":"Chen, M., et al.: PLACES: Prompting language models for social conversation synthesis. In: Findings of the Association for Computational Linguistics: EACL 2023, pp. 844\u2013868. Association for Computational Linguistics (2023). https:\/\/doi.org\/10.18653\/v1\/2023.findings-eacl.63","DOI":"10.18653\/v1\/2023.findings-eacl.63"},{"key":"17_CR4","doi-asserted-by":"crossref","first-page":"346","DOI":"10.1162\/tacl_a_00370","volume":"9","author":"D Deutsch","year":"2021","unstructured":"Deutsch, D., Roth, D., Durrett, G.: Towards question-answering as an automatic metric for evaluating the content quality of summaries. Trans. Assoc. Comput. Linguist. 9, 346\u2013361 (2021)","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"17_CR5","doi-asserted-by":"crossref","unstructured":"Durmus, E., He, H., Diab, M.: Feqa: a question answering evaluation framework for faithfulness assessment in abstractive summarization. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics (ACL), pp. 5055\u20135070 (2020)","DOI":"10.18653\/v1\/2020.acl-main.454"},{"key":"17_CR6","doi-asserted-by":"publisher","unstructured":"Fabbri, A.R., Wu, C.S., Liu, W., Xiong, C.: QAFactEval: Improved qa-based factual consistency evaluation for summarization. In: Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies (NAACL-HLT), pp. 2587\u20132601. Association for Computational Linguistics (2022). https:\/\/doi.org\/10.18653\/v1\/2022.naacl-main.187,","DOI":"10.18653\/v1\/2022.naacl-main.187"},{"key":"17_CR7","doi-asserted-by":"crossref","unstructured":"Fu, X., Chen, N., Gao, P., Li, Y.: Privacy-preserving personalized recommender systems. SSRN Working Paper (2022). https:\/\/ssrn.com\/abstract=4202576","DOI":"10.2139\/ssrn.4202576"},{"key":"17_CR8","unstructured":"Hinton, G., Vinyals, O., Dean, J.: Distilling the knowledge in a neural network. In: NeurIPS Deep Learning Workshop (2015)"},{"key":"17_CR9","doi-asserted-by":"crossref","unstructured":"Hoffmann, J., et\u00a0al.: Training compute-optimal large language models. In: Advances in Neural Information Processing Systems (NeurIPS), vol.\u00a035, pp. 30016\u201330030 (2022)","DOI":"10.52202\/068431-2176"},{"key":"17_CR10","unstructured":"Hu, E.J., et al.: Lora: Low-rank adaptation of large language models. In: International Conference on Learning Representations (ICLR) (2022)"},{"key":"17_CR11","doi-asserted-by":"crossref","unstructured":"Jandaghi, P., Sheng, X., Bai, X., Pujara, J., Sidahmed, H.: Faithful persona-based conversational dataset generation with large language models. arXiv preprint arXiv:2312.10007 (2023)","DOI":"10.18653\/v1\/2024.findings-acl.904"},{"key":"17_CR12","unstructured":"Kaplan, J., et al.: Scaling laws for neural language models. arXiv preprint arXiv:2001.08361 (2020)"},{"key":"17_CR13","doi-asserted-by":"publisher","unstructured":"Kim, H., et al.: SODA: Million-scale dialogue distillation with social commonsense contextualization. In: Proceedings of the 2023 Conference on Empirical Methods in Natural Language Processing, pp. 12930\u201312949. Association for Computational Linguistics (2023). https:\/\/doi.org\/10.18653\/v1\/2023.emnlp-main.799","DOI":"10.18653\/v1\/2023.emnlp-main.799"},{"key":"17_CR14","doi-asserted-by":"crossref","unstructured":"Kim, S., et al.: The cot collection: Improving zero-shot and few-shot learning of language models via chain-of-thought fine-tuning. arXiv preprint arXiv:2305.14045 (2023)","DOI":"10.18653\/v1\/2023.emnlp-main.782"},{"key":"17_CR15","unstructured":"Lee, Y.J., Lim, C.G., Choi, Y., Im, J.H., Choi, H.J.: Personachatgen: generating personalized dialogues using gpt-3. In: Proceedings of the 1st Workshop on Customized Chat Grounding Persona and Knowledge, vol.\u00a01, pp. 29\u201348 (2022)"},{"key":"17_CR16","unstructured":"Magister, L., Fried, D., Belinkov, Y.: Teaching small models to reason: distilling chain-of-thought (2023). https:\/\/arxiv.org\/abs\/2305.06350"},{"key":"17_CR17","series-title":"Springer Optimization and Its Applications","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-75178-4","volume-title":"Synthetic Data for Deep Learning","author":"SI Nikolenko","year":"2021","unstructured":"Nikolenko, S.I.: Synthetic Data for Deep Learning. SOIA, vol. 174. Springer, Cham (2021). https:\/\/doi.org\/10.1007\/978-3-030-75178-4"},{"key":"17_CR18","doi-asserted-by":"crossref","unstructured":"Reimers, N., Gurevych, I.: Sentence-bert: Sentence embeddings using siamese bert-networks. In: Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 3982\u20133992 (2019)","DOI":"10.18653\/v1\/D19-1410"},{"issue":"1","key":"17_CR19","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1186\/s40537-019-0197-0","volume":"6","author":"C Shorten","year":"2019","unstructured":"Shorten, C., Khoshgoftaar, T.M.: A survey on image data augmentation for deep learning. J. Big Data 6(1), 1\u201348 (2019)","journal-title":"J. Big Data"},{"key":"17_CR20","doi-asserted-by":"crossref","unstructured":"Tigunova, A., Yates, A., Mirza, P., Weikum, G.: Charm: Inferring personal attributes from conversations. In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), vol.\u00a01, pp. 5391\u20135404 (2020)","DOI":"10.18653\/v1\/2020.emnlp-main.434"},{"key":"17_CR21","doi-asserted-by":"crossref","unstructured":"Wang, Z., Zhou, X., Koncel-Kedziorski, R., Marin, A., Xia, F.: Extracting and inferring personal attributes from dialogue. In: Proceedings of the 4th Workshop on NLP for Conversational AI, vol.\u00a01, pp. 58\u201369 (2022)","DOI":"10.18653\/v1\/2022.nlp4convai-1.6"},{"key":"17_CR22","unstructured":"Wen, P., et al.: Thinkpatterns-21k: A systematic study on the impact of thinking patterns in llms. arXiv preprint arXiv:2503.12918 (2025)"},{"key":"17_CR23","unstructured":"Wu, C.S., Madotto, A., Lin, Z., Xu, P., Fung, P.: Getting to know you: user attribute extraction from dialogues. In: Proceedings of the 12th Language Resources and Evaluation Conference (LREC), vol.\u00a01, pp. 581\u2013589 (2020)"},{"key":"17_CR24","unstructured":"Xu, C., et\u00a0al.: How does synthetic data generation impact machine learning performance? a comprehensive study. arXiv preprint arXiv:2301.09286 (2023)"},{"key":"17_CR25","unstructured":"Yao, Y., et\u00a0al.: Evaluating the text generation capabilities of large-scale language models. arXiv preprint arXiv:2207.07411 (2022)"},{"key":"17_CR26","unstructured":"Yukhymenko, H., Staab, R., Vero, M., Vechev, M.: A synthetic dataset for personal attribute inference. In: Advances in Neural Information Processing Systems (2024). https:\/\/arxiv.org\/abs\/2406.07217"},{"key":"17_CR27","doi-asserted-by":"crossref","unstructured":"Zhang, S., Dinan, E., Urbanek, J., Szlam, A., Kiela, D., Weston, J.: Personalizing dialogue agents: I have a dog, do you have pets too? In: Proceedings of the 56th Annual Meeting of the Association for Computational Linguistics (ACL), vol.\u00a01, pp. 2204\u20132213 (2018)","DOI":"10.18653\/v1\/P18-1205"},{"key":"17_CR28","doi-asserted-by":"crossref","unstructured":"Zhu, L., Li, W., Mao, R., Pandelea, V., Cambria, E.: Paed: Zero-shot persona attribute extraction in dialogues. In: Proceedings of the 61st Annual Meeting of the Association for Computational Linguistics (ACL), vol.\u00a01, pp. 9771\u20139787 (2023)","DOI":"10.18653\/v1\/2023.acl-long.544"}],"container-title":["Lecture Notes in Computer Science","Advances in Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-21321-1_17","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,23]],"date-time":"2026-03-23T23:16:57Z","timestamp":1774307817000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-21321-1_17"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9783032213204","9783032213211"],"references-count":28,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-21321-1_17","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"24 March 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Disclosure of Interests"}},{"value":"ECIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Information Retrieval","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Delft","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"The Netherlands","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 March 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 April 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"48","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecir2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecir2026.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}