Publications
* denotes co-first authorship; † denotes corresponding authorship.
2026
- Evaluating the Robustness and Readiness of Large Frontier Models in Health AI Applications. . Nature Medicine. [paper ↗][arXiv ↗]
[bibtex]
@article{illusion26, year = {2026}, url = {https://doi.org/10.1038/s41591-026-04501-8}, author = {Gu, Yu and Fu, Jingjing and Liu, Xiaodong and Valanarasu, Jeya Maria Jose and Codella, Noel C. F. and Tan, Reuben and Liu, Qianchu and Jin, Ying and Zhang, Sheng and Wang, Jinyu and Wang, Rui and Song, Lei and Qin, Guanghui and Usuyama, Naoto and Wong, Cliff and Cheng, Hao and Lee, HoHin and Sanapathi, Praneeth and Hilado, Sarah and Naumann, Tristan and Alvarez-Valle, Javier and Bian, Jiang and Wei, Mu and Malik, Khalil and Zhou, Lidong and Gao, Jianfeng and Horvitz, Eric and Lungren, Matthew P. and Burger, Doug and Topol, Eric and Poon, Hoifung and Vozila, Paul}, journal = {Nature Medicine}, title = {Evaluating the Robustness and Readiness of Large Frontier Models in Health {{AI}} Applications} } - Masked-Diffusion Autoencoders for 3D Medical Vision Representation Learning. . Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR). [paper ↓][website ↗]
[bibtex]
@inproceedings{mdae26, year = {2026}, url = {https://openaccess.thecvf.com/content/CVPR2026/papers/Tu_Masked-Diffusion_Autoencoders_for_3D_Medical_Vision_Representation_Learning_CVPR_2026_paper.pdf}, author = {Tu, Jiachen and Qin, Guanghui and Zhao, Theodore Zhengde and Valanarasu, Jeya Maria Jose and Zhang, Sheng and Naumann, Tristan and Lam, Fan and Wang, Sheng and Poon, Hoifung}, booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)}, title = {Masked-{{Diffusion Autoencoders}} for {{3D Medical Vision Representation Learning}}} } - Scaling Medical Imaging Report Generation with Multimodal Reinforcement Learning. . arXiv. [paper ↗][blog ↗][twitter ↗]
[bibtex]
@misc{unirg26, year = {2026}, url = {https://doi.org/10.48550/arXiv.2601.17151}, author = {Liu, Qianchu and Zhang, Sheng and Qin, Guanghui and Gu, Yu and Jin, Ying and Preston, Sam and Xu, Yanbo and Kiblawi, Sid and Yim, Wen-wai and Ossowski, Tim and Naumann, Tristan and Wei, Mu and Poon, Hoifung}, title = {Scaling Medical Imaging Report Generation with Multimodal Reinforcement Learning} } - HealthAgentBench: A Unified Benchmark Suite of Realistic Agentic Healthcare Environments for Challenging Frontier AI Agents. . arXiv. [paper ↗][website ↗][codes ↗]
[bibtex]
@misc{healthagentbench26, year = {2026}, url = {https://doi.org/10.48550/ARXIV.2606.31179}, author = {Liu, Qianchu and Zhang, Sheng and Qin, Guanghui and Valanarasu, Jeya Maria Jose and Rokuss, Maximilian and Lu, Mingyu and Ossowski, Timothy and Chaves, Juan Manuel Zambrano and Wong, Cliff and Argaw, Peniel and Hasija, Yashna and Wei, Mu and Yim, Wen-wai and Liu, Qin and Jing, Zilin and Entenmann, Jason and Usuyama, Naoto and Naumann, Tristan and Poon, Hoifung}, title = {{{HealthAgentBench}}: {{A Unified Benchmark Suite}} of {{Realistic Agentic Healthcare Environments}} for {{Challenging Frontier AI Agents}}} } - OctoMed: Data Recipes for State-of-the-Art Multimodal Medical Reasoning. . Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR). [paper ↓][model ↗][data ↗][blog ↗]
[bibtex]
@inproceedings{octomed26, year = {2026}, url = {https://openaccess.thecvf.com/content/CVPR2026/papers/Ossowski_OctoMed_Data_Recipes_for_State-of-the-Art_Multimodal_Medical_Reasoning_CVPR_2026_paper.pdf}, author = {Ossowski, Timothy and Zhang, Sheng and Liu, Qianchu and Qin, Guanghui and Tan, Reuben and Naumann, Tristan and Hu, Junjie and Poon, Hoifung}, booktitle = {Proceedings of the IEEE/CVF Conference on Computer Vision and Pattern Recognition (CVPR)}, title = {{{OctoMed}}: {{Data Recipes}} for {{State-of-the-Art Multimodal Medical Reasoning}}} } - GigaPath-Flash and GigaTIME-Flash: Efficient Pathology Foundation Models for Whole-Slide and Tumor Microenvironment Analysis. . arXiv. [paper ↗][gigapath-flash ↗][gigatime-flash ↗]
[bibtex]
@misc{gigaflash26, year = {2026}, url = {https://doi.org/10.48550/arXiv.2607.18218}, author = {Usuyama, Naoto and Valanarasu, Jeya Maria Jose and Yao, Sicong and Xu, Hanwen and Bagga, Jaspreet and Qin, Guanghui and Kramer, Robert E. and Wong, Cliff and Lee, Soohee and Qiu, Hao and Zhao, Theodore Zhengde and Shimol, Racheli Ben and Crabtree, Angela and Matlock, Kevin and Garcia, Eduardo Alejandro Lozano and Sangani, Naiteek and Santamaria-Pang, Alberto and Rokuss, Maximilian and Hasija, Yashna and Patel, Naisargi Manishkumar and Entenmann, Jason and Bartlett, Alexandra Q. and Wright, Bill J. and Fox, Bernard A. and Piening, Brian and Zhang, Sheng and Wang, Sheng and Naumann, Tristan and Bifulco, Carlo and Poon, Hoifung}, title = {{{GigaPath-Flash}} and {{GigaTIME-Flash}}: {{Efficient Pathology Foundation Models}} for {{Whole-Slide}} and {{Tumor Microenvironment Analysis}}} }
2025
- Ras-Mediated Dynamic and Biphasic Regulation of Cell Migration. . Proceedings of the National Academy of Sciences of the United States of America (PNAS). [paper ↗][bioRxiv ↗]
[bibtex]
@article{dynamic25, year = {2025}, url = {https://doi.org/10.1073/pnas.2503847122}, author = {Lin, Yiyan and Paraj\'on, Eleana and Yuan, Qinling and Ye, Siyu and Qin, Guanghui and Deng, Yu and Borleis, Jane and Koyfman, Ariel and Iglesias, Pablo A and Konstantopoulos, Konstantinos and Robinson, Douglas N and Devreotes, Peter N}, journal = {Proceedings of the National Academy of Sciences of the United States of America (PNAS)}, issue = {30}, title = {Ras-Mediated Dynamic and Biphasic Regulation of Cell Migration}, volume = {122}, pages = {e2503847122} } - Be My Eyes: Extending Large Language Models to New Modalities Through Multi-Agent Collaboration. . arXiv. [paper ↗]
[bibtex]
@misc{bemyeyes25, year = {2025}, url = {https://doi.org/10.48550/arXiv.2511.19417}, author = {Huang, James Y. and Zhang, Sheng and Liu, Qianchu and Qin, Guanghui and Zhu, Tinghui and Naumann, Tristan and Chen, Muhao and Poon, Hoifung}, title = {Be {{My Eyes}}: {{Extending Large Language Models}} to {{New Modalities Through Multi-Agent Collaboration}}} } - Med-RLVR: Emerging Medical Reasoning from a 3B Base Model via Reinforcement Learning. . arXiv. [paper ↗]
[bibtex]
@misc{medrlvr25, year = {2025}, url = {https://doi.org/10.48550/arXiv.2502.19655}, author = {Zhang, Sheng and Liu, Qianchu and Qin, Guanghui and Naumann, Tristan and Poon, Hoifung}, title = {Med-{{RLVR}}: {{Emerging Medical Reasoning}} from a {{3B Base Model}} via {{Reinforcement Learning}}} } - X-Reasoner: Towards Generalizable Reasoning Across Modalities and Domains. . arXiv. [paper ↗][code ↗][model ↗]
[bibtex]
@misc{xreasoner25, year = {2025}, url = {https://doi.org/10.48550/arXiv.2505.03981}, author = {Liu, Qianchu and Zhang, Sheng and Qin, Guanghui and Ossowski, Timothy and Gu, Yu and Jin, Ying and Kiblawi, Sid and Preston, Sam and Wei, Mu and Vozila, Paul and Naumann, Tristan and Poon, Hoifung}, title = {X-{{Reasoner}}: {{Towards Generalizable Reasoning Across Modalities}} and {{Domains}}} } - CLERC: A Dataset for U.S. Legal Case Retrieval and Retrieval-Augmented Analysis Generation. . Proceedings of Annual Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics (NAACL). [paper ↗][data ↗]
[bibtex]
@inproceedings{clerc25, year = {2025}, url = {https://doi.org/10.18653/v1/2025.findings-naacl.441}, author = {Hou, Abe Bohan and Weller, Orion and Qin, Guanghui and Yang, Eugene and Lawrie, Dawn and Holzenberger, Nils and Blair-Stanek, Andrew and Van Durme, Benjamin}, booktitle = {Proceedings of Annual Conference of the Nations of the Americas Chapter of the Association for Computational Linguistics (NAACL)}, title = {{{CLERC}}: {{A Dataset}} for {{U.S. Legal Case Retrieval}} and {{Retrieval-Augmented Analysis Generation}}} } - Researchy Questions: A Dataset of Multi-Perspective, Decompositional Questions for Deep Research. . Proceedings of ACM Special Interest Group on Information Retrieval (SIGIR). [paper ↗][data ↗]
[bibtex]
@inproceedings{researchy25, year = {2025}, url = {https://doi.org/10.1145/3726302.3730275}, author = {Rosset, Corby and Chung, Ho-Lam and Qin, Guanghui and Chau, Ethan and Feng, Zhuo and Awadallah, Ahmed and Neville, Jennifer and Rao, Nikhil}, booktitle = {Proceedings of ACM Special Interest Group on Information Retrieval (SIGIR)}, title = {Researchy {{Questions}}: {{A Dataset}} of {{Multi-Perspective}}, {{Decompositional Questions}} for {{Deep Research}}} } - ERM Proteins Regulate the Shape and Number of Endoplasmic Reticulum-Plasma Membrane Junctions in Neurons. . bioRxiv. [paper ↗][biorxiv ↗]
[bibtex]
@misc{erm25, year = {2025}, url = {https://doi.org/10.1101/2025.06.18.660273}, author = {Deng, Huichao and Cheng, Jinbo and Fetter, Richard D and Qin, Guanghui and Zhang, Jianxiu and Liang, Xing and Taylor, Caitlin and Zhang, Mingjie and Wu, Xiandeng and Shen, Kang}, title = {{{ERM}} Proteins Regulate the Shape and Number of {{Endoplasmic Reticulum-Plasma Membrane Junctions}} in Neurons} } - Streaming Sequence Transduction through Dynamic Compression. . Proceedings of International Conference on Spoken Language Translation (IWSLT). [paper ↗][code ↗]
[bibtex]
@inproceedings{star25, year = {2025}, url = {https://doi.org/10.18653/v1/2025.iwslt-1.1}, author = {Tan, Weiting and Chen, Yunmo and Chen, Tongfei and Qin, Guanghui and Xu, Haoran and Zhang, Heidi C. and Van Durme, Benjamin and Koehn, Philipp}, booktitle = {Proceedings of International Conference on Spoken Language Translation (IWSLT)}, title = {Streaming {{Sequence Transduction}} through {{Dynamic Compression}}} } - KV-Distill: Nearly Lossless Context Compression for Transformers. . arXiv. [paper ↗]
[bibtex]
@misc{kvdistill25, year = {2025}, url = {https://doi.org/10.48550/arXiv.2503.10337}, author = {Chari, Vivek and Qin, Guanghui and Van Durme, Benjamin}, title = {{{KV-Distill}}: {{Nearly Lossless Context Compression}} for {{Transformers}}} }
2024
- Dodo: Dynamic Contextual Compression for Decoder-only LMs. . Proceedings of Annual Meeting of the Association for Computational Linguistics (ACL). [paper ↗][code ↗]
[bibtex]
@inproceedings{dodo24, year = {2024}, url = {https://doi.org/10.18653/v1/2024.acl-long.536}, author = {Qin, Guanghui and Rosset, Corby and Chau, Ethan C. and Rao, Nikhil and Van Durme, Benjamin}, booktitle = {Proceedings of Annual Meeting of the Association for Computational Linguistics (ACL)}, title = {Dodo: {{Dynamic Contextual Compression}} for {{Decoder-only LMs}}} } - Ras Suppression Potentiates Rear Actomyosin Contractility-Driven Cell Polarization and Migration. . Nature Cell Biology. [paper ↗][biorxiv ↗]
[bibtex]
@article{ras24, year = {2024}, url = {https://doi.org/10.1038/s41556-024-01453-4}, author = {Lin, Yiyan and Pal, Dhiman Sankar and Banerjee, Parijat and Banerjee, Tatsat and Qin, Guanghui and Deng, Yu and Borleis, Jane and Iglesias, Pablo A. and Devreotes, Peter N.}, journal = {Nature Cell Biology}, issue = {7}, title = {Ras Suppression Potentiates Rear Actomyosin Contractility-Driven Cell Polarization and Migration}, volume = {26}, pages = {1062--1076} } - L-Fresco: Factual Recall Evaluation Score for Legal Analysis Generation. . Proceedings of Generative AI + Law Workshop at International Conference on Machine Learning. [paper ↓]
[bibtex]
@inproceedings{lfresco24, year = {2024}, author = {Hou, Abe Bohan and Jiang, Zhengping and Qin, Guanghui and Weller, Orion and Blair-Stanek, Andrew and Van Durme, Benjamin}, booktitle = {Proceedings of Generative AI + Law Workshop at International Conference on Machine Learning}, title = {L-{{Fresco}}: {{Factual Recall Evaluation Score}} for {{Legal Analysis Generation}}} } - Towards Efficient Long-Context Natural Language Processing. . PhD thesis, Johns Hopkins University. [paper ↗]
[bibtex]
@thesis{thesis24, year = {2024}, url = {https://jscholarship.library.jhu.edu/items/68c24613-c500-4255-a15c-39d3dabbcee1}, author = {Qin, Guanghui}, type = {phdthesis}, title = {Towards {{Efficient Long-Context Natural Language Processing}}}, institution = {Johns Hopkins University} }
2023
- Nugget: Neural Agglomerative Embeddings of Text. . Proceedings of International Conference on Machine Learning (ICML). [paper ↗][data ↗][poster ↓][slides ↓][twitter ↗]
[bibtex]
@inproceedings{nugget23, year = {2023}, url = {https://proceedings.mlr.press/v202/qin23a.html}, author = {Qin, Guanghui and Van Durme, Benjamin}, booktitle = {Proceedings of International Conference on Machine Learning (ICML)}, title = {Nugget: {{Neural Agglomerative Embeddings}} of {{Text}}} } - The NLP Task Effectiveness of Long-Range Transformers. . Proceedings of Conference of the European Chapter of the Association for Computational Linguistics (EACL). [paper ↗][code ↗][slides ↓][poster ↓][video ↗]
[bibtex]
@inproceedings{nlpeffective23, year = {2023}, url = {https://doi.org/10.18653/v1/2023.eacl-main.273}, author = {Qin, Guanghui and Feng, Yukun and Van Durme, Benjamin}, booktitle = {Proceedings of Conference of the European Chapter of the Association for Computational Linguistics (EACL)}, title = {The {{NLP Task Effectiveness}} of {{Long-Range Transformers}}} }
2021
- Learning How to Ask: Querying LMs with Mixtures of Soft Prompts. . Proceedings of Conference of the North American Chapter of the Association for Computational Linguistics (NAACL) Best Short Paper. [paper ↗][poster ↓][slides ↓][code ↗][twitter ↗]
[bibtex]
@inproceedings{softprompt21, year = {2021}, url = {https://doi.org/10.18653/v1/2021.naacl-main.410}, author = {Qin, Guanghui and Eisner, Jason}, booktitle = {Proceedings of Conference of the North American Chapter of the Association for Computational Linguistics (NAACL)}, title = {Learning {{How}} to {{Ask}}: {{Querying LMs}} with {{Mixtures}} of {{Soft Prompts}}} } - LOME: Large Ontology Multilingual Extraction. . Proceedings of Conference of the European Chapter of the Association for Computational Linguistics (EACL). [paper ↗][demo ↗][code ↗][docker ↗][video ↗]
[bibtex]
@inproceedings{lome21, year = {2021}, url = {https://doi.org/10.18653/v1/2021.eacl-demos.19}, author = {Xia, Patrick and Qin, Guanghui and Vashishtha, Siddharth and Chen, Yunmo and Chen, Tongfei and May, Chandler and Harman, Craig and Rawlins, Kyle and White, Aaron Steven and Van Durme, Benjamin}, booktitle = {Proceedings of Conference of the European Chapter of the Association for Computational Linguistics (EACL)}, title = {{{LOME}}: {{Large Ontology Multilingual Extraction}}} } - Everything Is All It Takes: A Multipronged Strategy for Zero-Shot Cross-Lingual Information Extraction. . Proceedings of Conference on Empirical Methods in Natural Language Processing (EMNLP). [paper ↗][video ↗][code ↗]
[bibtex]
@inproceedings{everything21, year = {2021}, url = {https://doi.org/10.18653/v1/2021.emnlp-main.149}, author = {Yarmohammadi, Mahsa and Wu, Shijie and Marone, Marc and Xu, Haoran and Ebner, Seth and Qin, Guanghui and Chen, Yunmo and Guo, Jialiang and Harman, Craig and Murray, Kenton and White, Aaron Steven and Dredze, Mark and Van Durme, Benjamin}, booktitle = {Proceedings of Conference on Empirical Methods in Natural Language Processing (EMNLP)}, title = {Everything {{Is All It Takes}}: {{A Multipronged Strategy}} for {{Zero-Shot Cross-Lingual Information Extraction}}} } - Iterative Paraphrastic Augmentation with Discriminative Span Alignment. . Transactions of the Association for Computational Linguistics (TACL). [paper ↗]
[bibtex]
@article{iterative21, year = {2021}, url = {https://doi.org/10.1162/tacl_a_00380}, author = {Culkin, Ryan and Hu, J. Edward and Stengel-Eskin, Elias and Qin, Guanghui and Van Durme, Benjamin}, journal = {Transactions of the Association for Computational Linguistics (TACL)}, title = {Iterative {{Paraphrastic Augmentation}} with {{Discriminative Span Alignment}}}, volume = {9}, pages = {494--509} }
2020
- Neural Datalog Through Time: Informed Temporal Modeling via Logical Specification. . Proceedings of International Conference on Machine Learning (ICML). [paper ↗][blog ↗][slides ↓][code ↗][video ↗][press ↗]
[bibtex]
@inproceedings{datalog20, year = {2020}, url = {https://proceedings.mlr.press/v119/mei20a.html}, author = {Mei, Hongyuan and Qin, Guanghui and Xu, Minjie and Eisner, Jason}, booktitle = {Proceedings of International Conference on Machine Learning (ICML)}, title = {Neural {{Datalog Through Time}}: {{Informed Temporal Modeling}} via {{Logical Specification}}} } - CopyNext: Explicit Span Copying and Alignment in Sequence to Sequence Models. . Proceedings of The Forth Workshop on Structured Prediction for NLP. [paper ↗][video ↗][code ↗]
[bibtex]
@inproceedings{copynext20, year = {2020}, url = {https://doi.org/10.18653/v1/2020.spnlp-1.2}, author = {Singh, Abhinav and Xia, Patrick and Qin, Guanghui and Yarmohammadi, Mahsa and Van Durme, Benjamin}, booktitle = {Proceedings of The Forth Workshop on Structured Prediction for NLP}, title = {{{CopyNext}}: {{Explicit Span Copying}} and {{Alignment}} in {{Sequence}} to {{Sequence Models}}} }
2019
- Imputing Missing Events in Continuous-Time Event Streams. . Proceedings of International Conference on Machine Learning (ICML). [paper ↗][code ↗][poster ↓][slides ↓]
[bibtex]
@inproceedings{smoothing19, year = {2019}, url = {https://proceedings.mlr.press/v97/mei19a.html}, author = {Mei, Hongyuan and Qin, Guanghui and Eisner, Jason}, booktitle = {Proceedings of International Conference on Machine Learning (ICML)}, title = {Imputing {{Missing Events}} in {{Continuous-Time Event Streams}}} }
2018
- Learning Latent Semantic Annotations for Grounding Natural Language to Structured Data. . Proceedings of Conference on Empirical Methods in Natural Language Processing (EMNLP). [paper ↗][code ↗][slides ↓][video ↗]
[bibtex]
@inproceedings{latent18, year = {2018}, url = {https://doi.org/10.18653/v1/D18-1411}, author = {Qin, Guanghui and Yao, Jin-Ge and Wang, Xuening and Wang, Jinpeng and Lin, Chin-Yew}, booktitle = {Proceedings of Conference on Empirical Methods in Natural Language Processing (EMNLP)}, title = {Learning {{Latent Semantic Annotations}} for {{Grounding Natural Language}} to {{Structured Data}}} } - Data2Text Studio : Automated Text Generation from Structured Data. . Proceedings of Conference on Empirical Methods in Natural Language Processing (EMNLP). [paper ↗]
[bibtex]
@inproceedings{d2t18, year = {2018}, url = {https://doi.org/10.18653/v1/D18-2003}, author = {Dou, Longxu and Qin, Guanghui and Wang, Jinpeng and Yao, Jin-Ge and Lin, Chin-Yew}, booktitle = {Proceedings of Conference on Empirical Methods in Natural Language Processing (EMNLP)}, title = {{{Data2Text Studio}} : {{Automated Text Generation}} from {{Structured Data}}} }