{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T04:47:25Z","timestamp":1761367645482,"version":"build-2065373602"},"reference-count":49,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012226","name":"Fundamental Research Funds for the Central Universities","doi-asserted-by":"publisher","award":["500422813"],"award-info":[{"award-number":["500422813"]}],"id":[{"id":"10.13039\/501100012226","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62472046","82274685"],"award-info":[{"award-number":["62472046","82274685"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE Internet Things J."],"published-print":{"date-parts":[[2025]]},"DOI":"10.1109\/jiot.2025.3600573","type":"journal-article","created":{"date-parts":[[2025,8,19]],"date-time":"2025-08-19T18:17:12Z","timestamp":1755627432000},"page":"1-1","source":"Crossref","is-referenced-by-count":0,"title":["MT-Agent: Constructing a GUI Agent via Modality Enhancement and Text-Guided Fusion"],"prefix":"10.1109","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-8575-6378","authenticated-orcid":false,"given":"Jinhan","family":"Dong","sequence":"first","affiliation":[{"name":"Beijing University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4855-2464","authenticated-orcid":false,"given":"Lei","family":"Jin","sequence":"additional","affiliation":[{"name":"Beijing University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhihong","family":"Zhang","sequence":"additional","affiliation":[{"name":"China Mobile, Suzhou Software Technology Co., Ltd, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Tang","sequence":"additional","affiliation":[{"name":"China Mobile, Suzhou Software Technology Co., Ltd, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Runqing","family":"Zhang","sequence":"additional","affiliation":[{"name":"China Mobile, Suzhou Software Technology Co., Ltd, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liqiang","family":"Xu","sequence":"additional","affiliation":[{"name":"China Mobile, Suzhou Software Technology Co., Ltd, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6801-0510","authenticated-orcid":false,"given":"Junliang","family":"Xing","sequence":"additional","affiliation":[{"name":"Tsinghua University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","article-title":"AgentBench: Evaluating LLMs as agents","author":"Liu","year":"2023","journal-title":"arXiv:2308.03688"},{"key":"ref2","article-title":"Multimodal Web navigation with instruction-finetuned foundation models","author":"Furuta","year":"2023","journal-title":"arXiv:2305.11854"},{"key":"ref3","article-title":"A real-world Webagent with planning, long context understanding, and program synthesis","author":"Gur","year":"2023","journal-title":"arXiv:2307.12856"},{"key":"ref4","article-title":"Mind2Web: Towards a generalist agent for the Web","author":"Deng","year":"2023","journal-title":"arXiv:2306.06070"},{"key":"ref5","first-page":"1","article-title":"GPT-4V(ision) is a generalist Web agent, if grounded","volume-title":"Proc. 41st Int. Conf. Mach. Learn.","author":"Zheng"},{"key":"ref6","article-title":"Set-of-mark prompting unleashes extraordinary visual grounding in GPT-4V","author":"Yang","year":"2023","journal-title":"arXiv:2310.11441"},{"key":"ref7","article-title":"Towards better semantic understanding of mobile interfaces","author":"Sunkara","year":"2022","journal-title":"arXiv:2210.02663"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.505"},{"key":"ref9","article-title":"Navigating the digital world as humans do: Universal visual grounding for GUI agents","author":"Gou","year":"2024","journal-title":"arXiv:2410.05243"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73039-9_14"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.599"},{"volume-title":"GUICourse: From general vision language models to versatile GUI agents","year":"2024","author":"Chen","key":"ref12"},{"key":"ref13","article-title":"GUI odyssey: A comprehensive dataset for cross-App GUI navigation on mobile devices","author":"Lu","year":"2024","journal-title":"arXiv:2406.08451"},{"key":"ref14","article-title":"GUI testing arena: A unified benchmark for advancing autonomous GUI testing agent","author":"Zhao","year":"2024","journal-title":"arXiv:2412.18426"},{"key":"ref15","article-title":"Improving generalization in task-oriented dialogues with workflows and action plans","author":"Raimondo","year":"2023","journal-title":"arXiv:2306.01729"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.1158"},{"key":"ref17","first-page":"9097","article-title":"CoCo-agent: A comprehensive cognitive MLLM agent for smartphone GUI automation","volume-title":"Findings of the Association for Computational Linguistics","author":"Ma","year":"2024"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.186"},{"key":"ref19","article-title":"ReAct: Synergizing reasoning and acting in language models","author":"Yao","year":"2022","journal-title":"arXiv:2210.03629"},{"key":"ref20","first-page":"1","article-title":"GPT4Tools: Teaching large language model to use tools via self-instruction","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"36","author":"Yang"},{"key":"ref21","article-title":"AssistGPT: A general multi-modal assistant that can plan, execute, inspect, and learn","author":"Gao","year":"2023","journal-title":"arXiv:2306.08640"},{"key":"ref22","article-title":"Android in the wild: A large-scale dataset for android device control","author":"Rawles","year":"2023","journal-title":"arXiv:2307.10088"},{"key":"ref23","article-title":"ScreenAgent: A vision language model-driven computer control agent","author":"Niu","year":"2024","journal-title":"arXiv:2402.07945"},{"key":"ref24","first-page":"24824","article-title":"Chain-of-thought prompting elicits reasoning in large language models","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Wei"},{"key":"ref25","first-page":"22199","article-title":"Large language models are zero-shot reasoners","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Kojima"},{"key":"ref26","article-title":"Automatic chain of thought prompting in large language models","author":"Zhang","year":"2022","journal-title":"arXiv:2210.03493"},{"key":"ref27","article-title":"GPT-4 technical report","volume-title":"arXiv:2303.08774","author":"Achiam","year":"2023"},{"key":"ref28","first-page":"18893","article-title":"Pix2Struct: Screenshot parsing as pretraining for visual language understanding","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Lee"},{"key":"ref29","article-title":"GPT-4V in wonderland: Large multimodal models for zero-shot smartphone GUI navigation","author":"Yan","year":"2023","journal-title":"arXiv:2311.07562"},{"key":"ref30","article-title":"ClickAgent: Enhancing UI location capabilities of autonomous agents","author":"Hoscilowicz","year":"2024","journal-title":"arXiv:2410.11872"},{"key":"ref31","article-title":"Mobile-agent: Autonomous multi-modal mobile device agent with visual perception","author":"Wang","year":"2024","journal-title":"arXiv:2401.16158"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01354"},{"key":"ref33","article-title":"Spotlight: Mobile UI understanding using vision-language models with a focus","author":"Li","year":"2022","journal-title":"arXiv:2209.14927"},{"key":"ref34","article-title":"Ferret-UI 2: Mastering universal user interface understanding across platforms","author":"Li","year":"2024","journal-title":"arXiv:2410.18967"},{"key":"ref35","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Radford"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-72970-6_3"},{"key":"ref37","first-page":"23716","article-title":"Flamingo: A visual language model for few-shot learning","volume-title":"Proc. Adv. Neural Inf. Process. Syst.","volume":"35","author":"Alayrac"},{"key":"ref38","article-title":"Visual instruction tuning","author":"Liu","year":"2023","journal-title":"arXiv:2304.08485"},{"article-title":"BLIP-2: Bootstrapping language-image pre-training with frozen image encoders and large language models","volume-title":"Proc. Int. Conf. Mach. Learn. (ICML)","author":"Li","key":"ref39"},{"key":"ref40","article-title":"CogVLM: Visual expert for pretrained language models","author":"Wang","year":"2023","journal-title":"arXiv:2311.03079"},{"key":"ref41","article-title":"Mobile-agent-v2: Mobile device operation assistant with effective navigation via multi-agent collaboration","author":"Wang","year":"2024","journal-title":"arXiv:2406.01014"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-emnlp.702"},{"key":"ref43","article-title":"EVA-CLIP: Improved training techniques for CLIP at scale","author":"Sun","year":"2023","journal-title":"arXiv:2303.15389"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"ref45","article-title":"ScreenAI: A vision-language model for UI and infographics understanding","author":"Baechler","year":"2024","journal-title":"arXiv:2402.04615"},{"volume-title":"Flan-Alpaca: A fine-tuned language model combining flan-T5 and alpaca","year":"2024","key":"ref46"},{"key":"ref47","article-title":"OS-ATLAS: A foundation action model for generalist gui agents","author":"Wu","year":"2024","journal-title":"arXiv:2410.23218"},{"key":"ref48","article-title":"Qwen technical report","volume-title":"arXiv:2309.16609","author":"Bai","year":"2023"},{"key":"ref49","article-title":"Qwen2-VL: Enhancing vision-language model\u2019s perception of the world at any resolution","author":"Wang","year":"2024","journal-title":"arXiv:2409.12191"}],"container-title":["IEEE Internet of Things Journal"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/6488907\/6702522\/11130526.pdf?arnumber=11130526","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T04:43:28Z","timestamp":1761367408000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11130526\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"references-count":49,"URL":"https:\/\/doi.org\/10.1109\/jiot.2025.3600573","relation":{},"ISSN":["2327-4662","2372-2541"],"issn-type":[{"type":"electronic","value":"2327-4662"},{"type":"electronic","value":"2372-2541"}],"subject":[],"published":{"date-parts":[[2025]]}}}