@inproceedings{2026-acl-long-520,
  title = {{The Confidence Dichotomy: Analyzing and Mitigating Miscalibration in Tool-Use Agents}},
  author = {Weihao Xuan and Qingcheng Zeng and Heli Qi and Yunze Xiao and Junjue Wang and Naoto Yokoya},
  year = {2026},
  booktitle = {ACL 2026},
  url = {https://aclanthology.org/2026.acl-long.520/}
}

@inproceedings{2026-findings-acl-368,
  title = {{Sentipolis: Emotion-Aware Agents for Social Simulations}},
  author = {Chiyuan Fu and Lyuhao Chen and Yunze Xiao and Weihao Xuan and Carlos Busso and Mona Diab},
  year = {2026},
  booktitle = {Findings of ACL 2026},
  url = {https://aclanthology.org/2026.findings-acl.368/}
}

@inproceedings{2026-findings-acl-1270,
  title = {{Can LLMs Estimate Student Struggles? Human-AI Difficulty Alignment with Proficiency Simulation for Item Difficulty Prediction}},
  author = {Ming Li and Han Chen and Yunze Xiao and Jian Chen and Hong Jiao and Tianyi Zhou},
  year = {2026},
  booktitle = {Findings of ACL 2026},
  url = {https://aclanthology.org/2026.findings-acl.1270/}
}

@inproceedings{2026-acl-demo-83,
  title = {{TartanMaroon: Multi-Agent Academic Advising with Iterative Negotiation and Transparent Collaboration}},
  author = {Peidi Dong and Houda Bouamor and Yunze Xiao and Devi G Kurup},
  year = {2026},
  booktitle = {ACL 2026, System Demonstrations},
  url = {https://aclanthology.org/2026.acl-demo.83/}
}

@misc{ai-welfare,
  title = {{Position: AI Welfare Is Bullshit}},
  author = {Yunze Xiao and Gordon Dai and Shahan Ali Memon and Jen-tse Huang and Maarten Sap and Mona T. Diab},
  year = {2026},
  howpublished = {ICML 2026, Position Paper Track},
  url = {https://papers.ssrn.com/sol3/papers.cfm?abstract_id=6574439}
}

@inproceedings{2026-eacl-long-23,
  title = {{JiraiBench: A Cross-lingual Benchmark for Evaluating Large Language Models' Detection of Human Risky Health Behavior Content in Jirai Community}},
  author = {Yunze Xiao and Tingyu He and Lionel Z. Wang and Yiming Ma and Xingyu Song and Xiaohang Xu and Mona Diab and Irene Li and Ka Chung Ng},
  year = {2026},
  booktitle = {EACL 2026. Oral},
  url = {https://aclanthology.org/2026.eacl-long.23/}
}

@inproceedings{culture,
  title = {{Hire Your Anthropologist! Rethinking Culture Benchmarks Through an Anthropological Lens}},
  author = {Mai Alkhamissi and Yunze Xiao and Badr AlKhamissi and Mona Diab},
  year = {2026},
  booktitle = {Findings of EACL 2026},
  url = {https://aclanthology.org/2026.findings-eacl.63/}
}

@misc{2607-13924,
  title = {{ExpressionCueLens: a cross-cultural analysis of human-AI companion conversations on social media}},
  author = {Lynnette Hui Xian Ng and Yunze Xiao and Lionel Z. Wang and Weihao Xuan and Mona Diab},
  year = {2026},
  howpublished = {Journal of Ambient Intelligence and Humanized Computing, 17, 1567–1578},
  url = {https://link.springer.com/article/10.1007/s12652-026-05109-z}
}

@misc{2504-02956,
  title = {{Understanding Aha Moments: From External Observations to Internal Mechanisms}},
  author = {Shu Yang and Junchao Wu and Xin Chen and Yunze Xiao and Xinyi Yang and Derek F. Wong and Di Wang},
  year = {2026},
  howpublished = {TACL},
  url = {https://arxiv.org/abs/2504.02956}
}

@misc{10-1038-s41586-025-09962-4,
  title = {{A benchmark of expert-level academic questions to assess AI capabilities}},
  author = {Center for AI Safety and Scale AI and HLE Contributors Consortium (including Yunze Xiao)},
  year = {2026},
  howpublished = {(Humanity's Last Exam). Nature, 649, 1139–1146},
  url = {https://www.nature.com/articles/s41586-025-09962-4}
}

@inproceedings{humanizing,
  title = {{Humanizing Machines: Rethinking LLM Anthropomorphism Through a Multi-Level Framework of Design}},
  author = {Yunze Xiao and Lynnette Hui Xian Ng and Jiarui Liu and Mona Diab},
  year = {2025},
  booktitle = {EMNLP 2025. Oral},
  url = {https://aclanthology.org/2025.emnlp-main.164/}
}

@inproceedings{2025-emnlp-main-79,
  title = {{MMLU-ProX: A Multilingual Benchmark for Advanced Large Language Model Evaluation}},
  author = {Weihao Xuan and Rui Yang and Heli Qi and Qingcheng Zeng and Yunze Xiao and Aosong Feng and Dairui Liu and Yun Xing and Junjue Wang and Fan Gao and Jinghui Lu and Yuang Jiang and Huitao Li and Xin Li and Kunyu Yu and Ruihai Dong and Shangding Gu and Yuekang Li and Xiaofei Xie and Felix Juefei-Xu and Foutse Khomh and Osamu Yoshie and Qingyu Chen and Douglas Teodoro and Nan Liu and Randy Goebel and Lei Ma and Edison Marrese-Taylor and Shijian Lu and Yusuke Iwasawa and Yutaka Matsuo and Irene Li},
  year = {2025},
  booktitle = {EMNLP 2025},
  url = {https://aclanthology.org/2025.emnlp-main.79/}
}

@inproceedings{2025-emnlp-main-831,
  title = {{Synthetic Socratic Debates: Examining Persona Effects on Moral Decision and Persuasion Dynamics}},
  author = {Jiarui Liu and Yueqi Song and Yunze Xiao and Mingqian Zheng and Lindia Tjuatja and Jana Schaich Borg and Mona Diab and Maarten Sap},
  year = {2025},
  booktitle = {EMNLP 2025},
  url = {https://aclanthology.org/2025.emnlp-main.831/}
}

@inproceedings{2505-18139,
  title = {{Embracing Contradiction: Theoretical Inconsistency Will Not Impede the Road of Building Responsible AI Systems}},
  author = {Gordon Dai and Yunze Xiao},
  year = {2025},
  booktitle = {NeurIPS 2025, Position Paper Track},
  url = {https://proceedings.neurips.cc/paper_files/paper/2025/file/dd4a4bc7a7ba0b197b679c0025cb2df8-Paper-Position_Paper_Track.pdf}
}

@inproceedings{toxicloak,
  title = {{ToxiCloakCN: Evaluating Robustness of Offensive Language Detection in Chinese with Cloaking Perturbations}},
  author = {Yunze Xiao and Yujia Hu and Kenny Tsu Wei Choo and Roy Ka-Wei Lee},
  year = {2024},
  booktitle = {EMNLP 2024},
  url = {https://aclanthology.org/2024.emnlp-main.345/}
}

@inproceedings{2024-acl-long-102,
  title = {{InCharacter: Evaluating Personality Fidelity in Role-Playing Agents through Psychological Interviews}},
  author = {Xintao Wang and Yunze Xiao and Jen-tse Huang and Siyu Yuan and Rui Xu and Haoran Guo and Quan Tu and Yaying Fei and Ziang Leng and Wei Wang and Jiangjie Chen and Cheng Li and Yanghua Xiao},
  year = {2024},
  booktitle = {ACL 2024},
  url = {https://aclanthology.org/2024.acl-long.102/}
}

@inproceedings{2024-lrec-main-1508,
  title = {{Verbing Weirds Language (Models): Evaluation of English Zero-Derivation in Five LLMs}},
  author = {David R. Mortensen and Valentina Izrailevitch and Yunze Xiao and Hinrich Schütze and Leonie Weissweiler},
  year = {2024},
  booktitle = {LREC-COLING 2024},
  url = {https://aclanthology.org/2024.lrec-main.1508/}
}

@inproceedings{nexus,
  title = {{Nexus at ArAIEval Shared Task: Fine-Tuning Arabic Language Models for Propaganda and Disinformation Detection}},
  author = {Yunze Xiao and Firoj Alam},
  year = {2023},
  booktitle = {ArabicNLP 2023},
  url = {https://aclanthology.org/2023.arabicnlp-1.58/}
}

@misc{10-1109-ICCRD54409-2022-9730454,
  title = {{A Transformer-based Attention Flow Model for Intelligent Question and Answering Chatbot}},
  author = {Yunze Xiao},
  year = {2022},
  howpublished = {ICCRD 2022, 167–170},
  url = {https://doi.org/10.1109/ICCRD54409.2022.9730454}
}

@misc{2608-11528,
  title = {{Group Alignment-Induced Sycophancy: A Two-Sided Evaluation of Steerable Pluralistic Alignment}},
  author = {Haokai Zhao and Yunze Xiao and Weihao Xuan and Flora Salim and Benjamin Tag and Aditya Joshi},
  year = {2026},
  howpublished = {arXiv:2608.11528},
  url = {https://arxiv.org/abs/2608.11528}
}

@misc{2607-02032,
  title = {{PACE: A Proxy for Agentic Capability Evaluation}},
  author = {Yueqi Song and Lintang Sutawika and Jiarui Liu and Lindia Tjuatja and Jiayi Geng and Yunze Xiao and Daniel Lee and Aditya Bharat Soni and Vincent Lo and Xiang Yue and Graham Neubig},
  year = {2026},
  howpublished = {  arXiv:2607.02032},
  url = {https://arxiv.org/abs/2607.02032}
}

@misc{2606-11232,
  title = {{Every Act Has Its Price: Compressed Moral Composition in Frontier LLMs}},
  author = {Weijia Zhang and Ruiqi Chen and Yunze Xiao and Weihao Xuan},
  year = {2026},
  howpublished = {arXiv:2606.11232},
  url = {https://arxiv.org/abs/2606.11232}
}

@misc{chameleon,
  title = {{The Chameleon's Limit: Investigating Persona Collapse and Homogenization in Large Language Models}},
  author = {Yunze Xiao and Vivienne J. Zhang and Chenghao Yang and Ningshan Ma and Weihao Xuan and Jen-tse Huang},
  year = {2026},
  howpublished = {  arXiv:2604.24698},
  url = {https://arxiv.org/abs/2604.24698}
}

@misc{2604-22452,
  title = {{Superminds Test: Actively Evaluating Collective Intelligence of Agent Society via Probing Agents}},
  author = {Xirui Li and Ming Li and Yunze Xiao and Ryan Wong and Dianqi Li and Timothy Baldwin and Tianyi Zhou},
  year = {2026},
  howpublished = {arXiv:2604.22452},
  url = {https://arxiv.org/abs/2604.22452}
}

@misc{2604-06409,
  title = {{Say Something Else: Rethinking Contextual Privacy as Information Sufficiency}},
  author = {Yunze Xiao and Wenkai Li and Xiaoyuan Wu and Ningshan Ma and Yueqi Song and Weihao Xuan},
  year = {2026},
  howpublished = {arXiv:2604.06409},
  url = {https://arxiv.org/abs/2604.06409}
}

@misc{2602-00685,
  title = {{HumanStudy-Bench: Towards AI Agent Design for Participant Simulation}},
  author = {Xuan Liu and Haoyang Shang and Zizhang Liu and Xinyan Liu and Yunze Xiao and Yiwen Tu and Haojian Jin},
  year = {2026},
  howpublished = {arXiv:2602.00685},
  url = {https://arxiv.org/abs/2602.00685}
}

@misc{student,
  title = {{Towards Valid Student Simulation with Large Language Models}},
  author = {Zhihao Yuan and Yunze Xiao and Ming Li and Weihao Xuan and Richard Tong and Mona Diab and Tom Mitchell},
  year = {2026},
  howpublished = {arXiv:2601.05473},
  url = {https://arxiv.org/abs/2601.05473}
}

@misc{2601-02186,
  title = {{Toward Global Large Language Models in Medicine}},
  author = {Rui Yang and Huitao Li and Weihao Xuan and Heli Qi and Xin Li and Kunyu Yu and Yingjian Chen and Rongrong Wang and Jacques Behmoaras and Tianxi Cai and Bibhas Chakraborty and Qingyu Chen and Lionel Tim-Ee Cheng and Marie-Louise Damwanza and Chido Dzinotyiwei and Aosong Feng and Chuan Hong and Yusuke Iwasawa and Yuhe Ke and Linah Kitala and Taehoon Ko and Jisan Lee and Irene Li and Jonathan Chong Kai Liew and Hongfang Liu and Lian Leng Low and Edison Marrese-Taylor and Yutaka Matsuo and Isheanesu Misi and Yilin Ning and Jasmine Chiat Ling Ong and Marcus Eng Hock Ong and Enrico Petretto and Hossein Rouhizadeh and Abiram Sandralegar and Oren Schreier and Iain Bee Huat Tan and Patrick Tan and Daniel Shu Wei Ting and Junjue Wang and Chunhua Weng and Matthew Yu Heng Wong and Fang Wu and Yunze Xiao and Xuhai Xu and Qingcheng Zeng and Zhuo Zheng and Yifan Peng and Douglas Teodoro and Nan Liu},
  year = {2026},
  howpublished = {arXiv:2601.02186},
  url = {https://arxiv.org/abs/2601.02186}
}

@misc{2404-13885,
  title = {{Surveying Attitudinal Alignment Between Large Language Models Vs. Humans Towards 17 Sustainable Development Goals}},
  author = {Qingyang Wu and Ying Xu and Tingsong Xiao and Yunze Xiao and Yitong Li and Tianyang Wang and Yichi Zhang and Shanghai Zhong and Yuwei Zhang and Wei Lu and Yifan Yang},
  year = {2024},
  howpublished = {arXiv:2404.13885},
  url = {https://arxiv.org/abs/2404.13885}
}

@misc{survey,
  title = {{Chinese Offensive Language Detection: Current Status and Future Directions}},
  author = {Yunze Xiao and Houda Bouamor and Wajdi Zaghouani},
  year = {2024},
  howpublished = {arXiv:2403.18314},
  url = {https://arxiv.org/abs/2403.18314}
}
